@tanstack/ai 0.21.3 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +20 -0
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/index.js +184 -8
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +3 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/types.d.ts +20 -4
- package/package.json +2 -2
- package/skills/ai-core/adapter-configuration/SKILL.md +31 -0
- package/skills/ai-core/structured-outputs/SKILL.md +21 -9
- package/src/activities/chat/adapter.ts +23 -0
- package/src/activities/chat/index.ts +300 -13
- package/src/activities/generateSpeech/index.ts +3 -3
- package/src/types.ts +20 -4
|
@@ -95,6 +95,26 @@ export interface TextAdapter<TModel extends string, TProviderOptions extends Rec
|
|
|
95
95
|
* `{ object, raw, reasoning? }`.
|
|
96
96
|
*/
|
|
97
97
|
structuredOutputStream?: (options: StructuredOutputOptions<TProviderOptions>) => AsyncIterable<StreamChunk>;
|
|
98
|
+
/**
|
|
99
|
+
* Declares whether the adapter supports combining `tools` and a
|
|
100
|
+
* schema-constrained final answer in a single streaming request.
|
|
101
|
+
*
|
|
102
|
+
* When `true`, the engine wires `outputSchema` into the regular
|
|
103
|
+
* `chatStream()` call and skips the separate `runStructuredFinalization`
|
|
104
|
+
* round-trip. The model's natural final turn carries the
|
|
105
|
+
* schema-constrained JSON text and the engine harvests it from the agent
|
|
106
|
+
* loop's accumulated content.
|
|
107
|
+
*
|
|
108
|
+
* When `false`, `undefined`, or the method is omitted, the engine runs
|
|
109
|
+
* the agent loop without `outputSchema` and then issues a separate
|
|
110
|
+
* `structuredOutput` / `structuredOutputStream` call against the JSON
|
|
111
|
+
* schema for finalization (the legacy path).
|
|
112
|
+
*
|
|
113
|
+
* The method receives the per-call `modelOptions` so providers whose
|
|
114
|
+
* support depends on the resolved upstream model (e.g. OpenRouter) can
|
|
115
|
+
* answer per-request. Most adapters can return a constant.
|
|
116
|
+
*/
|
|
117
|
+
supportsCombinedToolsAndSchema?: (modelOptions?: TProviderOptions | undefined) => boolean;
|
|
98
118
|
}
|
|
99
119
|
/**
|
|
100
120
|
* A TextAdapter with any/unknown type parameters.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Declares whether the adapter supports combining `tools` and a\n * schema-constrained final answer in a single streaming request.\n *\n * When `true`, the engine wires `outputSchema` into the regular\n * `chatStream()` call and skips the separate `runStructuredFinalization`\n * round-trip. The model's natural final turn carries the\n * schema-constrained JSON text and the engine harvests it from the agent\n * loop's accumulated content.\n *\n * When `false`, `undefined`, or the method is omitted, the engine runs\n * the agent loop without `outputSchema` and then issues a separate\n * `structuredOutput` / `structuredOutputStream` call against the JSON\n * schema for finalization (the legacy path).\n *\n * The method receives the per-call `modelOptions` so providers whose\n * support depends on the resolved upstream model (e.g. OpenRouter) can\n * answer per-request. Most adapters can return a constant.\n */\n supportsCombinedToolsAndSchema?: (\n modelOptions?: TProviderOptions | undefined,\n ) => boolean\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"AAkKO,MAAe,gBAgBpB;AAAA,EACS,OAAO;AAAA,EAEP;AAAA,EAYC;AAAA,EAEV,YAAY,SAA4B,CAAA,GAAI,OAAe;AACzD,SAAK,SAAS;AACd,SAAK,QAAQ;AAAA,EACf;AAAA,EAcU,aAAqB;AAC7B,WAAO,GAAG,KAAK,IAAI,IAAI,KAAK,KAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,UAAU,CAAC,CAAC;AAAA,EAC9E;AACF;"}
|
|
@@ -59,6 +59,16 @@ class TextEngine {
|
|
|
59
59
|
logger;
|
|
60
60
|
// Structured-output finalization state (populated by runStructuredFinalization)
|
|
61
61
|
structuredOutputResult = null;
|
|
62
|
+
// Native combined mode: tracks whether we've already emitted the synthetic
|
|
63
|
+
// `structured-output.start` event before the schema-constrained final-turn
|
|
64
|
+
// text begins streaming. The event must precede the first
|
|
65
|
+
// TEXT_MESSAGE_START so the client-side StreamProcessor routes the JSON
|
|
66
|
+
// deltas into a StructuredOutputPart instead of a plain TextPart.
|
|
67
|
+
combinedStartEmitted = false;
|
|
68
|
+
// Native combined mode: messageId we want the synthetic
|
|
69
|
+
// `structured-output.start` (and any error emitted before deltas arrive)
|
|
70
|
+
// to carry, so the client matches it to the streaming text deltas.
|
|
71
|
+
combinedStructuredMessageId = null;
|
|
62
72
|
// Holds the validated value when `finalStructuredOutput.validate` is provided
|
|
63
73
|
// and succeeds. Distinct from `structuredOutputResult.data` (the raw,
|
|
64
74
|
// unvalidated payload from the structured-output.complete chunk).
|
|
@@ -184,7 +194,7 @@ class TextEngine {
|
|
|
184
194
|
if (pendingPhase === "wait") {
|
|
185
195
|
return;
|
|
186
196
|
}
|
|
187
|
-
const skipAgentLoop = !!this.finalStructuredOutput && this.tools.length === 0;
|
|
197
|
+
const skipAgentLoop = !!this.finalStructuredOutput && this.tools.length === 0 && this.finalStructuredOutput.nativeCombined !== true;
|
|
188
198
|
if (!skipAgentLoop) {
|
|
189
199
|
do {
|
|
190
200
|
if (this.earlyTermination || this.isCancelled()) {
|
|
@@ -198,11 +208,11 @@ class TextEngine {
|
|
|
198
208
|
this.middlewareCtx.phase = "beforeModel";
|
|
199
209
|
this.middlewareCtx.iteration = this.iterationCount;
|
|
200
210
|
const iterConfig = this.buildMiddlewareConfig();
|
|
201
|
-
const
|
|
211
|
+
const iterTransformedConfig = await this.middlewareRunner.runOnConfig(
|
|
202
212
|
this.middlewareCtx,
|
|
203
213
|
iterConfig
|
|
204
214
|
);
|
|
205
|
-
this.applyMiddlewareConfig(
|
|
215
|
+
this.applyMiddlewareConfig(iterTransformedConfig);
|
|
206
216
|
yield* this.streamModelResponse();
|
|
207
217
|
} else {
|
|
208
218
|
yield* this.processToolCalls();
|
|
@@ -214,7 +224,11 @@ class TextEngine {
|
|
|
214
224
|
finishReason: this.lastFinishReason
|
|
215
225
|
});
|
|
216
226
|
if (this.finalStructuredOutput && !this.isCancelled() && !this.finalizationError) {
|
|
217
|
-
|
|
227
|
+
if (this.finalStructuredOutput.nativeCombined === true) {
|
|
228
|
+
yield* this.harvestCombinedStructuredOutput();
|
|
229
|
+
} else {
|
|
230
|
+
yield* this.runStructuredFinalization();
|
|
231
|
+
}
|
|
218
232
|
}
|
|
219
233
|
if (!this.terminalHookCalled && this.toolPhase !== "wait" && !this.isCancelled()) {
|
|
220
234
|
if (this.finalizationError) {
|
|
@@ -338,6 +352,7 @@ class TextEngine {
|
|
|
338
352
|
toolCount: this.tools.length
|
|
339
353
|
}
|
|
340
354
|
);
|
|
355
|
+
const combinedSchema = this.finalStructuredOutput?.nativeCombined === true ? this.finalStructuredOutput.jsonSchema : void 0;
|
|
341
356
|
for await (const chunk of this.adapter.chatStream({
|
|
342
357
|
model: this.params.model,
|
|
343
358
|
messages: this.messages,
|
|
@@ -352,18 +367,41 @@ class TextEngine {
|
|
|
352
367
|
logger: this.logger,
|
|
353
368
|
threadId: this.threadId,
|
|
354
369
|
runId: this.runIdOverride,
|
|
355
|
-
parentRunId: this.parentRunIdOverride
|
|
370
|
+
parentRunId: this.parentRunIdOverride,
|
|
371
|
+
...combinedSchema ? { outputSchema: combinedSchema } : {}
|
|
356
372
|
})) {
|
|
357
373
|
if (this.isCancelled()) {
|
|
358
374
|
break;
|
|
359
375
|
}
|
|
360
376
|
this.totalChunkCount++;
|
|
361
377
|
this.handleStreamChunk(chunk);
|
|
378
|
+
if (this.finalStructuredOutput?.nativeCombined === true && this.finalStructuredOutput.yieldChunks && !this.combinedStartEmitted && chunk.type === EventType.TEXT_MESSAGE_START) {
|
|
379
|
+
this.combinedStartEmitted = true;
|
|
380
|
+
const messageId = typeof chunk.messageId === "string" && chunk.messageId !== "" ? chunk.messageId : generateMessageId();
|
|
381
|
+
this.combinedStructuredMessageId = messageId;
|
|
382
|
+
const synthStart = {
|
|
383
|
+
type: EventType.CUSTOM,
|
|
384
|
+
name: "structured-output.start",
|
|
385
|
+
value: { messageId },
|
|
386
|
+
model: this.params.model,
|
|
387
|
+
timestamp: Date.now(),
|
|
388
|
+
threadId: this.threadId,
|
|
389
|
+
...this.runIdOverride ? { runId: this.runIdOverride } : {}
|
|
390
|
+
};
|
|
391
|
+
const synthOutputs = await this.middlewareRunner.runOnChunk(
|
|
392
|
+
this.middlewareCtx,
|
|
393
|
+
synthStart
|
|
394
|
+
);
|
|
395
|
+
for (const outputChunk of synthOutputs) {
|
|
396
|
+
yield outputChunk;
|
|
397
|
+
this.middlewareCtx.chunkIndex++;
|
|
398
|
+
}
|
|
399
|
+
}
|
|
362
400
|
const outputChunks = await this.middlewareRunner.runOnChunk(
|
|
363
401
|
this.middlewareCtx,
|
|
364
402
|
chunk
|
|
365
403
|
);
|
|
366
|
-
const suppressAgentLifecycle = !!this.finalStructuredOutput && this.finalStructuredOutput.yieldChunks;
|
|
404
|
+
const suppressAgentLifecycle = !!this.finalStructuredOutput && this.finalStructuredOutput.yieldChunks && this.finalStructuredOutput.nativeCombined !== true;
|
|
367
405
|
for (const outputChunk of outputChunks) {
|
|
368
406
|
if (suppressAgentLifecycle && (outputChunk.type === EventType.RUN_STARTED || outputChunk.type === EventType.RUN_FINISHED)) {
|
|
369
407
|
continue;
|
|
@@ -1145,6 +1183,137 @@ class TextEngine {
|
|
|
1145
1183
|
}
|
|
1146
1184
|
}
|
|
1147
1185
|
}
|
|
1186
|
+
/**
|
|
1187
|
+
* Native combined mode: harvest the structured output from the agent
|
|
1188
|
+
* loop's accumulated final-turn text (no separate provider call).
|
|
1189
|
+
*
|
|
1190
|
+
* The adapter wired `outputSchema` into the regular `chatStream` request,
|
|
1191
|
+
* so the model's final-turn text is the schema-constrained JSON. We parse
|
|
1192
|
+
* `this.accumulatedContent`, populate `this.structuredOutputResult`, emit
|
|
1193
|
+
* a synthetic `structured-output.complete` (and a `structured-output.start`
|
|
1194
|
+
* if one wasn't emitted earlier — only happens on the streaming path when
|
|
1195
|
+
* the model returned no text at all), and run the validate callback when
|
|
1196
|
+
* present. Failures populate `this.finalizationError` so the engine's
|
|
1197
|
+
* terminal-hook chooser routes to `onError` (per spec §7.3).
|
|
1198
|
+
*
|
|
1199
|
+
* The `'structuredOutput'` middleware phase intentionally does NOT fire on
|
|
1200
|
+
* this path — middleware sees the run through `beforeModel` / `modelStream`
|
|
1201
|
+
* as usual. See PR #605 / issue #605 for the design rationale.
|
|
1202
|
+
*/
|
|
1203
|
+
async *harvestCombinedStructuredOutput() {
|
|
1204
|
+
if (!this.finalStructuredOutput) {
|
|
1205
|
+
throw new Error(
|
|
1206
|
+
"harvestCombinedStructuredOutput called without finalStructuredOutput config"
|
|
1207
|
+
);
|
|
1208
|
+
}
|
|
1209
|
+
const yieldChunks = this.finalStructuredOutput.yieldChunks;
|
|
1210
|
+
const rawText = this.accumulatedContent;
|
|
1211
|
+
if (rawText.length === 0) {
|
|
1212
|
+
this.finalizationError = {
|
|
1213
|
+
message: "missing structured result",
|
|
1214
|
+
code: "structured-output-missing-result"
|
|
1215
|
+
};
|
|
1216
|
+
} else {
|
|
1217
|
+
try {
|
|
1218
|
+
const parsed = JSON.parse(rawText);
|
|
1219
|
+
this.structuredOutputResult = { data: parsed, rawText };
|
|
1220
|
+
} catch (err) {
|
|
1221
|
+
const detail = rawText.slice(0, 200) + (rawText.length > 200 ? "..." : "");
|
|
1222
|
+
this.finalizationError = {
|
|
1223
|
+
message: `Failed to parse structured output as JSON. Content: ${detail}`,
|
|
1224
|
+
code: "structured-output-parse-failed",
|
|
1225
|
+
cause: err
|
|
1226
|
+
};
|
|
1227
|
+
}
|
|
1228
|
+
}
|
|
1229
|
+
if (this.structuredOutputResult && !this.finalizationError && this.finalStructuredOutput.validate) {
|
|
1230
|
+
try {
|
|
1231
|
+
const validated = this.finalStructuredOutput.validate(
|
|
1232
|
+
this.structuredOutputResult.data
|
|
1233
|
+
);
|
|
1234
|
+
this.validatedStructuredOutput = validated;
|
|
1235
|
+
this.hasValidatedStructuredOutput = true;
|
|
1236
|
+
} catch (err) {
|
|
1237
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
1238
|
+
this.finalizationError = {
|
|
1239
|
+
message,
|
|
1240
|
+
code: "structured-output-validation-failed",
|
|
1241
|
+
cause: err
|
|
1242
|
+
};
|
|
1243
|
+
}
|
|
1244
|
+
}
|
|
1245
|
+
if (!yieldChunks) {
|
|
1246
|
+
return;
|
|
1247
|
+
}
|
|
1248
|
+
if (!this.combinedStartEmitted) {
|
|
1249
|
+
this.combinedStartEmitted = true;
|
|
1250
|
+
const messageId = this.combinedStructuredMessageId ?? generateMessageId();
|
|
1251
|
+
this.combinedStructuredMessageId = messageId;
|
|
1252
|
+
const synthStart = {
|
|
1253
|
+
type: EventType.CUSTOM,
|
|
1254
|
+
name: "structured-output.start",
|
|
1255
|
+
value: { messageId },
|
|
1256
|
+
model: this.params.model,
|
|
1257
|
+
timestamp: Date.now(),
|
|
1258
|
+
threadId: this.threadId,
|
|
1259
|
+
...this.runIdOverride ? { runId: this.runIdOverride } : {}
|
|
1260
|
+
};
|
|
1261
|
+
const startOutputs = await this.middlewareRunner.runOnChunk(
|
|
1262
|
+
this.middlewareCtx,
|
|
1263
|
+
synthStart
|
|
1264
|
+
);
|
|
1265
|
+
for (const outputChunk of startOutputs) {
|
|
1266
|
+
yield outputChunk;
|
|
1267
|
+
this.middlewareCtx.chunkIndex++;
|
|
1268
|
+
}
|
|
1269
|
+
}
|
|
1270
|
+
if (this.structuredOutputResult && !this.finalizationError) {
|
|
1271
|
+
const completeChunk = {
|
|
1272
|
+
type: EventType.CUSTOM,
|
|
1273
|
+
name: "structured-output.complete",
|
|
1274
|
+
value: {
|
|
1275
|
+
object: this.structuredOutputResult.data,
|
|
1276
|
+
raw: this.structuredOutputResult.rawText,
|
|
1277
|
+
...this.combinedStructuredMessageId ? { messageId: this.combinedStructuredMessageId } : {}
|
|
1278
|
+
},
|
|
1279
|
+
model: this.params.model,
|
|
1280
|
+
timestamp: Date.now(),
|
|
1281
|
+
threadId: this.threadId,
|
|
1282
|
+
...this.runIdOverride ? { runId: this.runIdOverride } : {}
|
|
1283
|
+
};
|
|
1284
|
+
const completeOutputs = await this.middlewareRunner.runOnChunk(
|
|
1285
|
+
this.middlewareCtx,
|
|
1286
|
+
completeChunk
|
|
1287
|
+
);
|
|
1288
|
+
for (const outputChunk of completeOutputs) {
|
|
1289
|
+
yield outputChunk;
|
|
1290
|
+
this.middlewareCtx.chunkIndex++;
|
|
1291
|
+
}
|
|
1292
|
+
}
|
|
1293
|
+
if (this.finalizationError) {
|
|
1294
|
+
const errChunk = {
|
|
1295
|
+
type: EventType.RUN_ERROR,
|
|
1296
|
+
runId: this.runIdOverride ?? this.requestId,
|
|
1297
|
+
model: this.params.model,
|
|
1298
|
+
timestamp: Date.now(),
|
|
1299
|
+
threadId: this.threadId,
|
|
1300
|
+
message: this.finalizationError.message,
|
|
1301
|
+
...this.finalizationError.code ? { code: this.finalizationError.code } : {},
|
|
1302
|
+
error: {
|
|
1303
|
+
message: this.finalizationError.message,
|
|
1304
|
+
...this.finalizationError.code ? { code: this.finalizationError.code } : {}
|
|
1305
|
+
}
|
|
1306
|
+
};
|
|
1307
|
+
const errOutputs = await this.middlewareRunner.runOnChunk(
|
|
1308
|
+
this.middlewareCtx,
|
|
1309
|
+
errChunk
|
|
1310
|
+
);
|
|
1311
|
+
for (const outputChunk of errOutputs) {
|
|
1312
|
+
yield outputChunk;
|
|
1313
|
+
this.middlewareCtx.chunkIndex++;
|
|
1314
|
+
}
|
|
1315
|
+
}
|
|
1316
|
+
}
|
|
1148
1317
|
buildMiddlewareConfig() {
|
|
1149
1318
|
return {
|
|
1150
1319
|
messages: this.messages,
|
|
@@ -1283,6 +1452,7 @@ async function runAgenticStructuredOutput(options) {
|
|
|
1283
1452
|
throw new Error("Failed to convert output schema to JSON Schema");
|
|
1284
1453
|
}
|
|
1285
1454
|
const validate = isStandardSchema(outputSchema) ? (data) => parseWithStandardSchema(outputSchema, data) : void 0;
|
|
1455
|
+
const nativeCombined = adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true;
|
|
1286
1456
|
const engine = new TextEngine(
|
|
1287
1457
|
{
|
|
1288
1458
|
adapter,
|
|
@@ -1292,7 +1462,8 @@ async function runAgenticStructuredOutput(options) {
|
|
|
1292
1462
|
finalStructuredOutput: {
|
|
1293
1463
|
jsonSchema,
|
|
1294
1464
|
yieldChunks: false,
|
|
1295
|
-
...validate ? { validate } : {}
|
|
1465
|
+
...validate ? { validate } : {},
|
|
1466
|
+
...nativeCombined ? { nativeCombined: true } : {}
|
|
1296
1467
|
}
|
|
1297
1468
|
},
|
|
1298
1469
|
logger
|
|
@@ -1424,13 +1595,18 @@ async function* runStreamingStructuredOutputImpl(options, jsonSchema) {
|
|
|
1424
1595
|
const { adapter, outputSchema, middleware, context, debug, ...textOptions } = options;
|
|
1425
1596
|
const model = adapter.model;
|
|
1426
1597
|
const logger = resolveDebugOption(debug);
|
|
1598
|
+
const nativeCombined = adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true;
|
|
1427
1599
|
const engine = new TextEngine(
|
|
1428
1600
|
{
|
|
1429
1601
|
adapter,
|
|
1430
1602
|
params: { ...textOptions, model, logger },
|
|
1431
1603
|
middleware,
|
|
1432
1604
|
context,
|
|
1433
|
-
finalStructuredOutput: {
|
|
1605
|
+
finalStructuredOutput: {
|
|
1606
|
+
jsonSchema,
|
|
1607
|
+
yieldChunks: true,
|
|
1608
|
+
...nativeCombined ? { nativeCombined: true } : {}
|
|
1609
|
+
}
|
|
1434
1610
|
},
|
|
1435
1611
|
logger
|
|
1436
1612
|
);
|