@tanstack/ai 0.21.3 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +20 -0
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/index.js +185 -8
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/middleware/types.d.ts +2 -0
- package/dist/esm/activities/chat/stream/processor.d.ts +15 -0
- package/dist/esm/activities/chat/stream/processor.js +56 -0
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +3 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/types.d.ts +20 -4
- package/package.json +3 -3
- package/skills/ai-core/adapter-configuration/SKILL.md +32 -1
- package/skills/ai-core/debug-logging/SKILL.md +1 -1
- package/skills/ai-core/structured-outputs/SKILL.md +21 -9
- package/src/activities/chat/adapter.ts +23 -0
- package/src/activities/chat/index.ts +301 -13
- package/src/activities/chat/middleware/types.ts +2 -0
- package/src/activities/chat/stream/processor.ts +92 -0
- package/src/activities/generateSpeech/index.ts +3 -3
- package/src/types.ts +20 -4
|
@@ -95,6 +95,26 @@ export interface TextAdapter<TModel extends string, TProviderOptions extends Rec
|
|
|
95
95
|
* `{ object, raw, reasoning? }`.
|
|
96
96
|
*/
|
|
97
97
|
structuredOutputStream?: (options: StructuredOutputOptions<TProviderOptions>) => AsyncIterable<StreamChunk>;
|
|
98
|
+
/**
|
|
99
|
+
* Declares whether the adapter supports combining `tools` and a
|
|
100
|
+
* schema-constrained final answer in a single streaming request.
|
|
101
|
+
*
|
|
102
|
+
* When `true`, the engine wires `outputSchema` into the regular
|
|
103
|
+
* `chatStream()` call and skips the separate `runStructuredFinalization`
|
|
104
|
+
* round-trip. The model's natural final turn carries the
|
|
105
|
+
* schema-constrained JSON text and the engine harvests it from the agent
|
|
106
|
+
* loop's accumulated content.
|
|
107
|
+
*
|
|
108
|
+
* When `false`, `undefined`, or the method is omitted, the engine runs
|
|
109
|
+
* the agent loop without `outputSchema` and then issues a separate
|
|
110
|
+
* `structuredOutput` / `structuredOutputStream` call against the JSON
|
|
111
|
+
* schema for finalization (the legacy path).
|
|
112
|
+
*
|
|
113
|
+
* The method receives the per-call `modelOptions` so providers whose
|
|
114
|
+
* support depends on the resolved upstream model (e.g. OpenRouter) can
|
|
115
|
+
* answer per-request. Most adapters can return a constant.
|
|
116
|
+
*/
|
|
117
|
+
supportsCombinedToolsAndSchema?: (modelOptions?: TProviderOptions | undefined) => boolean;
|
|
98
118
|
}
|
|
99
119
|
/**
|
|
100
120
|
* A TextAdapter with any/unknown type parameters.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Declares whether the adapter supports combining `tools` and a\n * schema-constrained final answer in a single streaming request.\n *\n * When `true`, the engine wires `outputSchema` into the regular\n * `chatStream()` call and skips the separate `runStructuredFinalization`\n * round-trip. The model's natural final turn carries the\n * schema-constrained JSON text and the engine harvests it from the agent\n * loop's accumulated content.\n *\n * When `false`, `undefined`, or the method is omitted, the engine runs\n * the agent loop without `outputSchema` and then issues a separate\n * `structuredOutput` / `structuredOutputStream` call against the JSON\n * schema for finalization (the legacy path).\n *\n * The method receives the per-call `modelOptions` so providers whose\n * support depends on the resolved upstream model (e.g. OpenRouter) can\n * answer per-request. Most adapters can return a constant.\n */\n supportsCombinedToolsAndSchema?: (\n modelOptions?: TProviderOptions | undefined,\n ) => boolean\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"AAkKO,MAAe,gBAgBpB;AAAA,EACS,OAAO;AAAA,EAEP;AAAA,EAYC;AAAA,EAEV,YAAY,SAA4B,CAAA,GAAI,OAAe;AACzD,SAAK,SAAS;AACd,SAAK,QAAQ;AAAA,EACf;AAAA,EAcU,aAAqB;AAC7B,WAAO,GAAG,KAAK,IAAI,IAAI,KAAK,KAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,UAAU,CAAC,CAAC;AAAA,EAC9E;AACF;"}
|
|
@@ -59,6 +59,16 @@ class TextEngine {
|
|
|
59
59
|
logger;
|
|
60
60
|
// Structured-output finalization state (populated by runStructuredFinalization)
|
|
61
61
|
structuredOutputResult = null;
|
|
62
|
+
// Native combined mode: tracks whether we've already emitted the synthetic
|
|
63
|
+
// `structured-output.start` event before the schema-constrained final-turn
|
|
64
|
+
// text begins streaming. The event must precede the first
|
|
65
|
+
// TEXT_MESSAGE_START so the client-side StreamProcessor routes the JSON
|
|
66
|
+
// deltas into a StructuredOutputPart instead of a plain TextPart.
|
|
67
|
+
combinedStartEmitted = false;
|
|
68
|
+
// Native combined mode: messageId we want the synthetic
|
|
69
|
+
// `structured-output.start` (and any error emitted before deltas arrive)
|
|
70
|
+
// to carry, so the client matches it to the streaming text deltas.
|
|
71
|
+
combinedStructuredMessageId = null;
|
|
62
72
|
// Holds the validated value when `finalStructuredOutput.validate` is provided
|
|
63
73
|
// and succeeds. Distinct from `structuredOutputResult.data` (the raw,
|
|
64
74
|
// unvalidated payload from the structured-output.complete chunk).
|
|
@@ -103,6 +113,7 @@ class TextEngine {
|
|
|
103
113
|
this.middlewareCtx = {
|
|
104
114
|
requestId: this.requestId,
|
|
105
115
|
streamId: this.streamId,
|
|
116
|
+
runId: this.runIdOverride ?? this.requestId,
|
|
106
117
|
threadId: this.threadId,
|
|
107
118
|
// Legacy alias kept on the ctx so middleware that reads
|
|
108
119
|
// `ctx.conversationId` keeps working. Always equals `threadId`.
|
|
@@ -184,7 +195,7 @@ class TextEngine {
|
|
|
184
195
|
if (pendingPhase === "wait") {
|
|
185
196
|
return;
|
|
186
197
|
}
|
|
187
|
-
const skipAgentLoop = !!this.finalStructuredOutput && this.tools.length === 0;
|
|
198
|
+
const skipAgentLoop = !!this.finalStructuredOutput && this.tools.length === 0 && this.finalStructuredOutput.nativeCombined !== true;
|
|
188
199
|
if (!skipAgentLoop) {
|
|
189
200
|
do {
|
|
190
201
|
if (this.earlyTermination || this.isCancelled()) {
|
|
@@ -198,11 +209,11 @@ class TextEngine {
|
|
|
198
209
|
this.middlewareCtx.phase = "beforeModel";
|
|
199
210
|
this.middlewareCtx.iteration = this.iterationCount;
|
|
200
211
|
const iterConfig = this.buildMiddlewareConfig();
|
|
201
|
-
const
|
|
212
|
+
const iterTransformedConfig = await this.middlewareRunner.runOnConfig(
|
|
202
213
|
this.middlewareCtx,
|
|
203
214
|
iterConfig
|
|
204
215
|
);
|
|
205
|
-
this.applyMiddlewareConfig(
|
|
216
|
+
this.applyMiddlewareConfig(iterTransformedConfig);
|
|
206
217
|
yield* this.streamModelResponse();
|
|
207
218
|
} else {
|
|
208
219
|
yield* this.processToolCalls();
|
|
@@ -214,7 +225,11 @@ class TextEngine {
|
|
|
214
225
|
finishReason: this.lastFinishReason
|
|
215
226
|
});
|
|
216
227
|
if (this.finalStructuredOutput && !this.isCancelled() && !this.finalizationError) {
|
|
217
|
-
|
|
228
|
+
if (this.finalStructuredOutput.nativeCombined === true) {
|
|
229
|
+
yield* this.harvestCombinedStructuredOutput();
|
|
230
|
+
} else {
|
|
231
|
+
yield* this.runStructuredFinalization();
|
|
232
|
+
}
|
|
218
233
|
}
|
|
219
234
|
if (!this.terminalHookCalled && this.toolPhase !== "wait" && !this.isCancelled()) {
|
|
220
235
|
if (this.finalizationError) {
|
|
@@ -338,6 +353,7 @@ class TextEngine {
|
|
|
338
353
|
toolCount: this.tools.length
|
|
339
354
|
}
|
|
340
355
|
);
|
|
356
|
+
const combinedSchema = this.finalStructuredOutput?.nativeCombined === true ? this.finalStructuredOutput.jsonSchema : void 0;
|
|
341
357
|
for await (const chunk of this.adapter.chatStream({
|
|
342
358
|
model: this.params.model,
|
|
343
359
|
messages: this.messages,
|
|
@@ -352,18 +368,41 @@ class TextEngine {
|
|
|
352
368
|
logger: this.logger,
|
|
353
369
|
threadId: this.threadId,
|
|
354
370
|
runId: this.runIdOverride,
|
|
355
|
-
parentRunId: this.parentRunIdOverride
|
|
371
|
+
parentRunId: this.parentRunIdOverride,
|
|
372
|
+
...combinedSchema ? { outputSchema: combinedSchema } : {}
|
|
356
373
|
})) {
|
|
357
374
|
if (this.isCancelled()) {
|
|
358
375
|
break;
|
|
359
376
|
}
|
|
360
377
|
this.totalChunkCount++;
|
|
361
378
|
this.handleStreamChunk(chunk);
|
|
379
|
+
if (this.finalStructuredOutput?.nativeCombined === true && this.finalStructuredOutput.yieldChunks && !this.combinedStartEmitted && chunk.type === EventType.TEXT_MESSAGE_START) {
|
|
380
|
+
this.combinedStartEmitted = true;
|
|
381
|
+
const messageId = typeof chunk.messageId === "string" && chunk.messageId !== "" ? chunk.messageId : generateMessageId();
|
|
382
|
+
this.combinedStructuredMessageId = messageId;
|
|
383
|
+
const synthStart = {
|
|
384
|
+
type: EventType.CUSTOM,
|
|
385
|
+
name: "structured-output.start",
|
|
386
|
+
value: { messageId },
|
|
387
|
+
model: this.params.model,
|
|
388
|
+
timestamp: Date.now(),
|
|
389
|
+
threadId: this.threadId,
|
|
390
|
+
...this.runIdOverride ? { runId: this.runIdOverride } : {}
|
|
391
|
+
};
|
|
392
|
+
const synthOutputs = await this.middlewareRunner.runOnChunk(
|
|
393
|
+
this.middlewareCtx,
|
|
394
|
+
synthStart
|
|
395
|
+
);
|
|
396
|
+
for (const outputChunk of synthOutputs) {
|
|
397
|
+
yield outputChunk;
|
|
398
|
+
this.middlewareCtx.chunkIndex++;
|
|
399
|
+
}
|
|
400
|
+
}
|
|
362
401
|
const outputChunks = await this.middlewareRunner.runOnChunk(
|
|
363
402
|
this.middlewareCtx,
|
|
364
403
|
chunk
|
|
365
404
|
);
|
|
366
|
-
const suppressAgentLifecycle = !!this.finalStructuredOutput && this.finalStructuredOutput.yieldChunks;
|
|
405
|
+
const suppressAgentLifecycle = !!this.finalStructuredOutput && this.finalStructuredOutput.yieldChunks && this.finalStructuredOutput.nativeCombined !== true;
|
|
367
406
|
for (const outputChunk of outputChunks) {
|
|
368
407
|
if (suppressAgentLifecycle && (outputChunk.type === EventType.RUN_STARTED || outputChunk.type === EventType.RUN_FINISHED)) {
|
|
369
408
|
continue;
|
|
@@ -1145,6 +1184,137 @@ class TextEngine {
|
|
|
1145
1184
|
}
|
|
1146
1185
|
}
|
|
1147
1186
|
}
|
|
1187
|
+
/**
|
|
1188
|
+
* Native combined mode: harvest the structured output from the agent
|
|
1189
|
+
* loop's accumulated final-turn text (no separate provider call).
|
|
1190
|
+
*
|
|
1191
|
+
* The adapter wired `outputSchema` into the regular `chatStream` request,
|
|
1192
|
+
* so the model's final-turn text is the schema-constrained JSON. We parse
|
|
1193
|
+
* `this.accumulatedContent`, populate `this.structuredOutputResult`, emit
|
|
1194
|
+
* a synthetic `structured-output.complete` (and a `structured-output.start`
|
|
1195
|
+
* if one wasn't emitted earlier — only happens on the streaming path when
|
|
1196
|
+
* the model returned no text at all), and run the validate callback when
|
|
1197
|
+
* present. Failures populate `this.finalizationError` so the engine's
|
|
1198
|
+
* terminal-hook chooser routes to `onError` (per spec §7.3).
|
|
1199
|
+
*
|
|
1200
|
+
* The `'structuredOutput'` middleware phase intentionally does NOT fire on
|
|
1201
|
+
* this path — middleware sees the run through `beforeModel` / `modelStream`
|
|
1202
|
+
* as usual. See PR #605 / issue #605 for the design rationale.
|
|
1203
|
+
*/
|
|
1204
|
+
async *harvestCombinedStructuredOutput() {
|
|
1205
|
+
if (!this.finalStructuredOutput) {
|
|
1206
|
+
throw new Error(
|
|
1207
|
+
"harvestCombinedStructuredOutput called without finalStructuredOutput config"
|
|
1208
|
+
);
|
|
1209
|
+
}
|
|
1210
|
+
const yieldChunks = this.finalStructuredOutput.yieldChunks;
|
|
1211
|
+
const rawText = this.accumulatedContent;
|
|
1212
|
+
if (rawText.length === 0) {
|
|
1213
|
+
this.finalizationError = {
|
|
1214
|
+
message: "missing structured result",
|
|
1215
|
+
code: "structured-output-missing-result"
|
|
1216
|
+
};
|
|
1217
|
+
} else {
|
|
1218
|
+
try {
|
|
1219
|
+
const parsed = JSON.parse(rawText);
|
|
1220
|
+
this.structuredOutputResult = { data: parsed, rawText };
|
|
1221
|
+
} catch (err) {
|
|
1222
|
+
const detail = rawText.slice(0, 200) + (rawText.length > 200 ? "..." : "");
|
|
1223
|
+
this.finalizationError = {
|
|
1224
|
+
message: `Failed to parse structured output as JSON. Content: ${detail}`,
|
|
1225
|
+
code: "structured-output-parse-failed",
|
|
1226
|
+
cause: err
|
|
1227
|
+
};
|
|
1228
|
+
}
|
|
1229
|
+
}
|
|
1230
|
+
if (this.structuredOutputResult && !this.finalizationError && this.finalStructuredOutput.validate) {
|
|
1231
|
+
try {
|
|
1232
|
+
const validated = this.finalStructuredOutput.validate(
|
|
1233
|
+
this.structuredOutputResult.data
|
|
1234
|
+
);
|
|
1235
|
+
this.validatedStructuredOutput = validated;
|
|
1236
|
+
this.hasValidatedStructuredOutput = true;
|
|
1237
|
+
} catch (err) {
|
|
1238
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
1239
|
+
this.finalizationError = {
|
|
1240
|
+
message,
|
|
1241
|
+
code: "structured-output-validation-failed",
|
|
1242
|
+
cause: err
|
|
1243
|
+
};
|
|
1244
|
+
}
|
|
1245
|
+
}
|
|
1246
|
+
if (!yieldChunks) {
|
|
1247
|
+
return;
|
|
1248
|
+
}
|
|
1249
|
+
if (!this.combinedStartEmitted) {
|
|
1250
|
+
this.combinedStartEmitted = true;
|
|
1251
|
+
const messageId = this.combinedStructuredMessageId ?? generateMessageId();
|
|
1252
|
+
this.combinedStructuredMessageId = messageId;
|
|
1253
|
+
const synthStart = {
|
|
1254
|
+
type: EventType.CUSTOM,
|
|
1255
|
+
name: "structured-output.start",
|
|
1256
|
+
value: { messageId },
|
|
1257
|
+
model: this.params.model,
|
|
1258
|
+
timestamp: Date.now(),
|
|
1259
|
+
threadId: this.threadId,
|
|
1260
|
+
...this.runIdOverride ? { runId: this.runIdOverride } : {}
|
|
1261
|
+
};
|
|
1262
|
+
const startOutputs = await this.middlewareRunner.runOnChunk(
|
|
1263
|
+
this.middlewareCtx,
|
|
1264
|
+
synthStart
|
|
1265
|
+
);
|
|
1266
|
+
for (const outputChunk of startOutputs) {
|
|
1267
|
+
yield outputChunk;
|
|
1268
|
+
this.middlewareCtx.chunkIndex++;
|
|
1269
|
+
}
|
|
1270
|
+
}
|
|
1271
|
+
if (this.structuredOutputResult && !this.finalizationError) {
|
|
1272
|
+
const completeChunk = {
|
|
1273
|
+
type: EventType.CUSTOM,
|
|
1274
|
+
name: "structured-output.complete",
|
|
1275
|
+
value: {
|
|
1276
|
+
object: this.structuredOutputResult.data,
|
|
1277
|
+
raw: this.structuredOutputResult.rawText,
|
|
1278
|
+
...this.combinedStructuredMessageId ? { messageId: this.combinedStructuredMessageId } : {}
|
|
1279
|
+
},
|
|
1280
|
+
model: this.params.model,
|
|
1281
|
+
timestamp: Date.now(),
|
|
1282
|
+
threadId: this.threadId,
|
|
1283
|
+
...this.runIdOverride ? { runId: this.runIdOverride } : {}
|
|
1284
|
+
};
|
|
1285
|
+
const completeOutputs = await this.middlewareRunner.runOnChunk(
|
|
1286
|
+
this.middlewareCtx,
|
|
1287
|
+
completeChunk
|
|
1288
|
+
);
|
|
1289
|
+
for (const outputChunk of completeOutputs) {
|
|
1290
|
+
yield outputChunk;
|
|
1291
|
+
this.middlewareCtx.chunkIndex++;
|
|
1292
|
+
}
|
|
1293
|
+
}
|
|
1294
|
+
if (this.finalizationError) {
|
|
1295
|
+
const errChunk = {
|
|
1296
|
+
type: EventType.RUN_ERROR,
|
|
1297
|
+
runId: this.runIdOverride ?? this.requestId,
|
|
1298
|
+
model: this.params.model,
|
|
1299
|
+
timestamp: Date.now(),
|
|
1300
|
+
threadId: this.threadId,
|
|
1301
|
+
message: this.finalizationError.message,
|
|
1302
|
+
...this.finalizationError.code ? { code: this.finalizationError.code } : {},
|
|
1303
|
+
error: {
|
|
1304
|
+
message: this.finalizationError.message,
|
|
1305
|
+
...this.finalizationError.code ? { code: this.finalizationError.code } : {}
|
|
1306
|
+
}
|
|
1307
|
+
};
|
|
1308
|
+
const errOutputs = await this.middlewareRunner.runOnChunk(
|
|
1309
|
+
this.middlewareCtx,
|
|
1310
|
+
errChunk
|
|
1311
|
+
);
|
|
1312
|
+
for (const outputChunk of errOutputs) {
|
|
1313
|
+
yield outputChunk;
|
|
1314
|
+
this.middlewareCtx.chunkIndex++;
|
|
1315
|
+
}
|
|
1316
|
+
}
|
|
1317
|
+
}
|
|
1148
1318
|
buildMiddlewareConfig() {
|
|
1149
1319
|
return {
|
|
1150
1320
|
messages: this.messages,
|
|
@@ -1283,6 +1453,7 @@ async function runAgenticStructuredOutput(options) {
|
|
|
1283
1453
|
throw new Error("Failed to convert output schema to JSON Schema");
|
|
1284
1454
|
}
|
|
1285
1455
|
const validate = isStandardSchema(outputSchema) ? (data) => parseWithStandardSchema(outputSchema, data) : void 0;
|
|
1456
|
+
const nativeCombined = adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true;
|
|
1286
1457
|
const engine = new TextEngine(
|
|
1287
1458
|
{
|
|
1288
1459
|
adapter,
|
|
@@ -1292,7 +1463,8 @@ async function runAgenticStructuredOutput(options) {
|
|
|
1292
1463
|
finalStructuredOutput: {
|
|
1293
1464
|
jsonSchema,
|
|
1294
1465
|
yieldChunks: false,
|
|
1295
|
-
...validate ? { validate } : {}
|
|
1466
|
+
...validate ? { validate } : {},
|
|
1467
|
+
...nativeCombined ? { nativeCombined: true } : {}
|
|
1296
1468
|
}
|
|
1297
1469
|
},
|
|
1298
1470
|
logger
|
|
@@ -1424,13 +1596,18 @@ async function* runStreamingStructuredOutputImpl(options, jsonSchema) {
|
|
|
1424
1596
|
const { adapter, outputSchema, middleware, context, debug, ...textOptions } = options;
|
|
1425
1597
|
const model = adapter.model;
|
|
1426
1598
|
const logger = resolveDebugOption(debug);
|
|
1599
|
+
const nativeCombined = adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true;
|
|
1427
1600
|
const engine = new TextEngine(
|
|
1428
1601
|
{
|
|
1429
1602
|
adapter,
|
|
1430
1603
|
params: { ...textOptions, model, logger },
|
|
1431
1604
|
middleware,
|
|
1432
1605
|
context,
|
|
1433
|
-
finalStructuredOutput: {
|
|
1606
|
+
finalStructuredOutput: {
|
|
1607
|
+
jsonSchema,
|
|
1608
|
+
yieldChunks: true,
|
|
1609
|
+
...nativeCombined ? { nativeCombined: true } : {}
|
|
1610
|
+
}
|
|
1434
1611
|
},
|
|
1435
1612
|
logger
|
|
1436
1613
|
);
|