@tanstack/ai 0.21.3 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +20 -0
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/index.js +185 -8
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/middleware/types.d.ts +2 -0
- package/dist/esm/activities/chat/stream/processor.d.ts +15 -0
- package/dist/esm/activities/chat/stream/processor.js +56 -0
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +3 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/types.d.ts +20 -4
- package/package.json +3 -3
- package/skills/ai-core/adapter-configuration/SKILL.md +32 -1
- package/skills/ai-core/debug-logging/SKILL.md +1 -1
- package/skills/ai-core/structured-outputs/SKILL.md +21 -9
- package/src/activities/chat/adapter.ts +23 -0
- package/src/activities/chat/index.ts +301 -13
- package/src/activities/chat/middleware/types.ts +2 -0
- package/src/activities/chat/stream/processor.ts +92 -0
- package/src/activities/generateSpeech/index.ts +3 -3
- package/src/types.ts +20 -4
|
@@ -58,10 +58,10 @@ export type TTSActivityResult<TStream extends boolean = false> = TStream extends
|
|
|
58
58
|
* @example Generate speech from text
|
|
59
59
|
* ```ts
|
|
60
60
|
* import { generateSpeech } from '@tanstack/ai'
|
|
61
|
-
* import {
|
|
61
|
+
* import { openaiSpeech } from '@tanstack/ai-openai'
|
|
62
62
|
*
|
|
63
63
|
* const result = await generateSpeech({
|
|
64
|
-
* adapter:
|
|
64
|
+
* adapter: openaiSpeech('tts-1-hd'),
|
|
65
65
|
* text: 'Hello, welcome to TanStack AI!',
|
|
66
66
|
* voice: 'nova'
|
|
67
67
|
* })
|
|
@@ -72,7 +72,7 @@ export type TTSActivityResult<TStream extends boolean = false> = TStream extends
|
|
|
72
72
|
* @example With format and speed options
|
|
73
73
|
* ```ts
|
|
74
74
|
* const result = await generateSpeech({
|
|
75
|
-
* adapter:
|
|
75
|
+
* adapter: openaiSpeech('tts-1'),
|
|
76
76
|
* text: 'This is slower speech.',
|
|
77
77
|
* voice: 'alloy',
|
|
78
78
|
* format: 'wav',
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import {
|
|
1
|
+
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiSpeech } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
|
package/dist/esm/types.d.ts
CHANGED
|
@@ -650,10 +650,26 @@ export interface TextOptions<TProviderOptionsSuperset extends Record<string, any
|
|
|
650
650
|
request?: Request | RequestInit;
|
|
651
651
|
/**
|
|
652
652
|
* Schema for structured output.
|
|
653
|
-
*
|
|
654
|
-
*
|
|
655
|
-
*
|
|
656
|
-
*
|
|
653
|
+
*
|
|
654
|
+
* **Two distinct use sites:**
|
|
655
|
+
*
|
|
656
|
+
* 1. **User-facing (activity layer):** accepts any
|
|
657
|
+
* {@link SchemaInput} — Zod, ArkType, Valibot, or a raw JSON Schema.
|
|
658
|
+
* The activity layer converts to JSON Schema before handing off.
|
|
659
|
+
*
|
|
660
|
+
* 2. **Adapter-facing (`chatStream` call):** the engine populates this with
|
|
661
|
+
* a pre-converted JSON Schema **only** when the adapter declared
|
|
662
|
+
* `supportsCombinedToolsAndSchema(modelOptions) === true`. The adapter
|
|
663
|
+
* should then wire the schema into the upstream request (e.g.
|
|
664
|
+
* `response_format: { type: 'json_schema', ... }`, `text.format`,
|
|
665
|
+
* `output_format`) alongside any `tools`. The model's natural final
|
|
666
|
+
* turn carries the schema-constrained JSON text and the engine
|
|
667
|
+
* harvests it from the agent loop without a separate finalization
|
|
668
|
+
* round-trip.
|
|
669
|
+
*
|
|
670
|
+
* Adapters that did NOT declare the capability never see this field
|
|
671
|
+
* populated — the engine instead invokes `structuredOutput` /
|
|
672
|
+
* `structuredOutputStream` after the agent loop.
|
|
657
673
|
*/
|
|
658
674
|
outputSchema?: SchemaInput;
|
|
659
675
|
/**
|
package/package.json
CHANGED
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.22.1",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
7
7
|
"repository": {
|
|
8
8
|
"type": "git",
|
|
9
9
|
"url": "git+https://github.com/TanStack/ai.git",
|
|
10
|
-
"directory": "packages/
|
|
10
|
+
"directory": "packages/ai"
|
|
11
11
|
},
|
|
12
12
|
"type": "module",
|
|
13
13
|
"module": "./dist/esm/index.js",
|
|
@@ -64,7 +64,7 @@
|
|
|
64
64
|
"@ag-ui/core": "^0.0.52",
|
|
65
65
|
"@standard-schema/spec": "^1.1.0",
|
|
66
66
|
"partial-json": "^0.1.7",
|
|
67
|
-
"@tanstack/ai-event-client": "0.
|
|
67
|
+
"@tanstack/ai-event-client": "0.4.0"
|
|
68
68
|
},
|
|
69
69
|
"peerDependencies": {
|
|
70
70
|
"@opentelemetry/api": ">=1.9.0"
|
|
@@ -26,7 +26,7 @@ sources:
|
|
|
26
26
|
|
|
27
27
|
> **Before implementing:** Ask the user which provider and model they want.
|
|
28
28
|
> Then fetch the latest available models from the provider's source code
|
|
29
|
-
> (check the adapter's model metadata file, e.g. `packages/
|
|
29
|
+
> (check the adapter's model metadata file, e.g. `packages/ai-openai/src/model-meta.ts`)
|
|
30
30
|
> or from the provider's API/docs to recommend the most current model.
|
|
31
31
|
> The model lists in this skill and its reference files may be outdated.
|
|
32
32
|
> Always verify against the source before recommending a specific model.
|
|
@@ -221,6 +221,37 @@ const custom = myOpenai('ft:gpt-5.2:my-org:custom-model:abc123')
|
|
|
221
221
|
At runtime, `extendAdapter` simply passes through to the original factory.
|
|
222
222
|
The `_customModels` parameter is only used for type inference.
|
|
223
223
|
|
|
224
|
+
### 5. Capability Flag: `supportsCombinedToolsAndSchema`
|
|
225
|
+
|
|
226
|
+
Adapters can declare an optional capability method:
|
|
227
|
+
|
|
228
|
+
```ts
|
|
229
|
+
supportsCombinedToolsAndSchema?(modelOptions?: TProviderOptions): boolean
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
When `true`, the engine wires `outputSchema` into the regular
|
|
233
|
+
`chatStream` call alongside `tools` and harvests the schema-constrained
|
|
234
|
+
JSON from the agent loop's final-turn text — skipping the separate
|
|
235
|
+
`structuredOutput` / `structuredOutputStream` finalization round-trip.
|
|
236
|
+
When `false` (or the method is omitted), the legacy finalization path
|
|
237
|
+
runs.
|
|
238
|
+
|
|
239
|
+
Current per-adapter status (#605):
|
|
240
|
+
|
|
241
|
+
| Adapter | Returns |
|
|
242
|
+
| -------------------------------------------- | ------------------------------------------------------------------------------------------------- |
|
|
243
|
+
| `openaiText` / `openaiChatCompletions` | `true` (all supported models) |
|
|
244
|
+
| `anthropicText` | `true` for Claude 4.5+ (gated by `ANTHROPIC_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
|
|
245
|
+
| `geminiText` | `true` for Gemini 3.x (gated by `GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
|
|
246
|
+
| `grokText` | `true` for Grok 4 family (gated by `GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
|
|
247
|
+
| `groqText` | `false` (Groq API rejects schema + tools + stream) |
|
|
248
|
+
| `openRouterText` / `openRouterResponsesText` | `false` (per-call resolution is a follow-up) |
|
|
249
|
+
| `ollamaText` | `false` (constrained-decoding vs tool-call grammar conflict) |
|
|
250
|
+
|
|
251
|
+
Subclasses can override to narrow the capability. When extending an
|
|
252
|
+
adapter for a custom model that doesn't support the combination, return
|
|
253
|
+
`false` explicitly.
|
|
254
|
+
|
|
224
255
|
## Common Mistakes
|
|
225
256
|
|
|
226
257
|
### a. HIGH: Confusing legacy monolithic with tree-shakeable adapter
|
|
@@ -181,15 +181,27 @@ The terminal event is a `CUSTOM` chunk: `{ type: 'CUSTOM', name: 'structured-out
|
|
|
181
181
|
|
|
182
182
|
**Adapter coverage for streaming:**
|
|
183
183
|
|
|
184
|
-
| Adapter
|
|
185
|
-
|
|
|
186
|
-
| `@tanstack/ai-openai`
|
|
187
|
-
| `@tanstack/ai-
|
|
188
|
-
| `@tanstack/ai-
|
|
189
|
-
| `@tanstack/ai-
|
|
190
|
-
|
|
|
191
|
-
|
|
192
|
-
|
|
184
|
+
| Adapter | `outputSchema` + `stream: true` |
|
|
185
|
+
| --------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
186
|
+
| `@tanstack/ai-openai` (Responses + Chat Completions) | **Native combined mode (#605)** — schema wired into the regular `chatStream` call alongside `tools`; engine harvests JSON, no finalization round-trip |
|
|
187
|
+
| `@tanstack/ai-anthropic` (Claude 4.5+ only) | **Native combined mode (#605)** — `output_config.format` + `tools` in one beta Messages call. Older Claude models fall back |
|
|
188
|
+
| `@tanstack/ai-gemini` (Gemini 3.x only) | **Native combined mode (#605)** — `responseSchema` + `tools` in one `generateContentStream`. Gemini 2.x falls back |
|
|
189
|
+
| `@tanstack/ai-grok` (Grok 4 family only) | **Native combined mode (#605)** — `response_format: json_schema` + `tools`. Grok 2 / 3 fall back |
|
|
190
|
+
| `@tanstack/ai-openrouter` | Native single-request stream (legacy `structuredOutputStream` path; per-call combined-mode lookup is a follow-up) |
|
|
191
|
+
| `@tanstack/ai-groq` | Legacy `structuredOutputStream` only (no tools — Groq's API rejects schema + tools + stream) |
|
|
192
|
+
| All other adapters (ollama, older Claude, Gemini 2.x, Grok 2/3) | Fallback: runs non-streaming `structuredOutput`, emits one `structured-output.complete` event |
|
|
193
|
+
|
|
194
|
+
**Native combined mode vs fallback** is signaled by the adapter's
|
|
195
|
+
optional `supportsCombinedToolsAndSchema(modelOptions)` method. When
|
|
196
|
+
it returns `true`, the engine wires the JSON Schema into the regular
|
|
197
|
+
`chatStream` call and harvests the final-turn text — middleware sees
|
|
198
|
+
the run through `beforeModel` / `modelStream` as usual, and the
|
|
199
|
+
`'structuredOutput'` middleware phase does **not** fire. When it
|
|
200
|
+
returns `false` (or is omitted), the engine takes the legacy
|
|
201
|
+
finalization path: agent loop, then a separate `structuredOutput` /
|
|
202
|
+
`structuredOutputStream` call with `'structuredOutput'` phase tagging.
|
|
203
|
+
|
|
204
|
+
Consumer code is identical across providers — always read the final object off `structured-output.complete`.
|
|
193
205
|
|
|
194
206
|
### Pattern 4: useChat with outputSchema (progressive UI)
|
|
195
207
|
|
|
@@ -123,6 +123,29 @@ export interface TextAdapter<
|
|
|
123
123
|
structuredOutputStream?: (
|
|
124
124
|
options: StructuredOutputOptions<TProviderOptions>,
|
|
125
125
|
) => AsyncIterable<StreamChunk>
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Declares whether the adapter supports combining `tools` and a
|
|
129
|
+
* schema-constrained final answer in a single streaming request.
|
|
130
|
+
*
|
|
131
|
+
* When `true`, the engine wires `outputSchema` into the regular
|
|
132
|
+
* `chatStream()` call and skips the separate `runStructuredFinalization`
|
|
133
|
+
* round-trip. The model's natural final turn carries the
|
|
134
|
+
* schema-constrained JSON text and the engine harvests it from the agent
|
|
135
|
+
* loop's accumulated content.
|
|
136
|
+
*
|
|
137
|
+
* When `false`, `undefined`, or the method is omitted, the engine runs
|
|
138
|
+
* the agent loop without `outputSchema` and then issues a separate
|
|
139
|
+
* `structuredOutput` / `structuredOutputStream` call against the JSON
|
|
140
|
+
* schema for finalization (the legacy path).
|
|
141
|
+
*
|
|
142
|
+
* The method receives the per-call `modelOptions` so providers whose
|
|
143
|
+
* support depends on the resolved upstream model (e.g. OpenRouter) can
|
|
144
|
+
* answer per-request. Most adapters can return a constant.
|
|
145
|
+
*/
|
|
146
|
+
supportsCombinedToolsAndSchema?: (
|
|
147
|
+
modelOptions?: TProviderOptions | undefined,
|
|
148
|
+
) => boolean
|
|
126
149
|
}
|
|
127
150
|
|
|
128
151
|
/**
|