@tanstack/ai 0.21.3 → 0.22.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -58,10 +58,10 @@ export type TTSActivityResult<TStream extends boolean = false> = TStream extends
58
58
  * @example Generate speech from text
59
59
  * ```ts
60
60
  * import { generateSpeech } from '@tanstack/ai'
61
- * import { openaiTTS } from '@tanstack/ai-openai'
61
+ * import { openaiSpeech } from '@tanstack/ai-openai'
62
62
  *
63
63
  * const result = await generateSpeech({
64
- * adapter: openaiTTS('tts-1-hd'),
64
+ * adapter: openaiSpeech('tts-1-hd'),
65
65
  * text: 'Hello, welcome to TanStack AI!',
66
66
  * voice: 'nova'
67
67
  * })
@@ -72,7 +72,7 @@ export type TTSActivityResult<TStream extends boolean = false> = TStream extends
72
72
  * @example With format and speed options
73
73
  * ```ts
74
74
  * const result = await generateSpeech({
75
- * adapter: openaiTTS('tts-1'),
75
+ * adapter: openaiSpeech('tts-1'),
76
76
  * text: 'This is slower speech.',
77
77
  * voice: 'alloy',
78
78
  * format: 'wav',
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiTTS } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiTTS('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiTTS('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
1
+ {"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiSpeech } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
@@ -650,10 +650,26 @@ export interface TextOptions<TProviderOptionsSuperset extends Record<string, any
650
650
  request?: Request | RequestInit;
651
651
  /**
652
652
  * Schema for structured output.
653
- * When provided, the adapter should use the provider's native structured output API
654
- * to ensure the response conforms to this schema.
655
- * The schema will be converted to JSON Schema format before being sent to the provider.
656
- * Supports any Standard JSON Schema compliant library (Zod, ArkType, Valibot, etc.).
653
+ *
654
+ * **Two distinct use sites:**
655
+ *
656
+ * 1. **User-facing (activity layer):** accepts any
657
+ * {@link SchemaInput} — Zod, ArkType, Valibot, or a raw JSON Schema.
658
+ * The activity layer converts to JSON Schema before handing off.
659
+ *
660
+ * 2. **Adapter-facing (`chatStream` call):** the engine populates this with
661
+ * a pre-converted JSON Schema **only** when the adapter declared
662
+ * `supportsCombinedToolsAndSchema(modelOptions) === true`. The adapter
663
+ * should then wire the schema into the upstream request (e.g.
664
+ * `response_format: { type: 'json_schema', ... }`, `text.format`,
665
+ * `output_format`) alongside any `tools`. The model's natural final
666
+ * turn carries the schema-constrained JSON text and the engine
667
+ * harvests it from the agent loop without a separate finalization
668
+ * round-trip.
669
+ *
670
+ * Adapters that did NOT declare the capability never see this field
671
+ * populated — the engine instead invokes `structuredOutput` /
672
+ * `structuredOutputStream` after the agent loop.
657
673
  */
658
674
  outputSchema?: SchemaInput;
659
675
  /**
package/package.json CHANGED
@@ -1,13 +1,13 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.21.3",
3
+ "version": "0.22.1",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",
9
9
  "url": "git+https://github.com/TanStack/ai.git",
10
- "directory": "packages/typescript/ai"
10
+ "directory": "packages/ai"
11
11
  },
12
12
  "type": "module",
13
13
  "module": "./dist/esm/index.js",
@@ -64,7 +64,7 @@
64
64
  "@ag-ui/core": "^0.0.52",
65
65
  "@standard-schema/spec": "^1.1.0",
66
66
  "partial-json": "^0.1.7",
67
- "@tanstack/ai-event-client": "0.3.10"
67
+ "@tanstack/ai-event-client": "0.4.0"
68
68
  },
69
69
  "peerDependencies": {
70
70
  "@opentelemetry/api": ">=1.9.0"
@@ -26,7 +26,7 @@ sources:
26
26
 
27
27
  > **Before implementing:** Ask the user which provider and model they want.
28
28
  > Then fetch the latest available models from the provider's source code
29
- > (check the adapter's model metadata file, e.g. `packages/typescript/ai-openai/src/model-meta.ts`)
29
+ > (check the adapter's model metadata file, e.g. `packages/ai-openai/src/model-meta.ts`)
30
30
  > or from the provider's API/docs to recommend the most current model.
31
31
  > The model lists in this skill and its reference files may be outdated.
32
32
  > Always verify against the source before recommending a specific model.
@@ -221,6 +221,37 @@ const custom = myOpenai('ft:gpt-5.2:my-org:custom-model:abc123')
221
221
  At runtime, `extendAdapter` simply passes through to the original factory.
222
222
  The `_customModels` parameter is only used for type inference.
223
223
 
224
+ ### 5. Capability Flag: `supportsCombinedToolsAndSchema`
225
+
226
+ Adapters can declare an optional capability method:
227
+
228
+ ```ts
229
+ supportsCombinedToolsAndSchema?(modelOptions?: TProviderOptions): boolean
230
+ ```
231
+
232
+ When `true`, the engine wires `outputSchema` into the regular
233
+ `chatStream` call alongside `tools` and harvests the schema-constrained
234
+ JSON from the agent loop's final-turn text — skipping the separate
235
+ `structuredOutput` / `structuredOutputStream` finalization round-trip.
236
+ When `false` (or the method is omitted), the legacy finalization path
237
+ runs.
238
+
239
+ Current per-adapter status (#605):
240
+
241
+ | Adapter | Returns |
242
+ | -------------------------------------------- | ------------------------------------------------------------------------------------------------- |
243
+ | `openaiText` / `openaiChatCompletions` | `true` (all supported models) |
244
+ | `anthropicText` | `true` for Claude 4.5+ (gated by `ANTHROPIC_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
245
+ | `geminiText` | `true` for Gemini 3.x (gated by `GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
246
+ | `grokText` | `true` for Grok 4 family (gated by `GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
247
+ | `groqText` | `false` (Groq API rejects schema + tools + stream) |
248
+ | `openRouterText` / `openRouterResponsesText` | `false` (per-call resolution is a follow-up) |
249
+ | `ollamaText` | `false` (constrained-decoding vs tool-call grammar conflict) |
250
+
251
+ Subclasses can override to narrow the capability. When extending an
252
+ adapter for a custom model that doesn't support the combination, return
253
+ `false` explicitly.
254
+
224
255
  ## Common Mistakes
225
256
 
226
257
  ### a. HIGH: Confusing legacy monolithic with tree-shakeable adapter
@@ -253,7 +253,7 @@ const safe: Logger = {
253
253
  }
254
254
  ```
255
255
 
256
- Source: packages/typescript/ai/src/logger/internal-logger.ts
256
+ Source: packages/ai/src/logger/internal-logger.ts
257
257
 
258
258
  ## Cross-References
259
259
 
@@ -181,15 +181,27 @@ The terminal event is a `CUSTOM` chunk: `{ type: 'CUSTOM', name: 'structured-out
181
181
 
182
182
  **Adapter coverage for streaming:**
183
183
 
184
- | Adapter | `outputSchema` + `stream: true` |
185
- | ------------------------------------------------- | --------------------------------------------------------------------------------------------- |
186
- | `@tanstack/ai-openai` | Native single-request stream (Responses API) |
187
- | `@tanstack/ai-openrouter` | Native single-request stream |
188
- | `@tanstack/ai-grok` | Native single-request stream (Chat Completions) |
189
- | `@tanstack/ai-groq` | Native single-request stream (Chat Completions) |
190
- | All other adapters (anthropic, gemini, ollama, …) | Fallback: runs non-streaming `structuredOutput`, emits one `structured-output.complete` event |
191
-
192
- Consumer code is identical across providers — always read the final object off `structured-output.complete`. You only see incremental `TEXT_MESSAGE_CONTENT` deltas when the adapter implements `structuredOutputStream` natively.
184
+ | Adapter | `outputSchema` + `stream: true` |
185
+ | --------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- |
186
+ | `@tanstack/ai-openai` (Responses + Chat Completions) | **Native combined mode (#605)** — schema wired into the regular `chatStream` call alongside `tools`; engine harvests JSON, no finalization round-trip |
187
+ | `@tanstack/ai-anthropic` (Claude 4.5+ only) | **Native combined mode (#605)** — `output_config.format` + `tools` in one beta Messages call. Older Claude models fall back |
188
+ | `@tanstack/ai-gemini` (Gemini 3.x only) | **Native combined mode (#605)** — `responseSchema` + `tools` in one `generateContentStream`. Gemini 2.x falls back |
189
+ | `@tanstack/ai-grok` (Grok 4 family only) | **Native combined mode (#605)** — `response_format: json_schema` + `tools`. Grok 2 / 3 fall back |
190
+ | `@tanstack/ai-openrouter` | Native single-request stream (legacy `structuredOutputStream` path; per-call combined-mode lookup is a follow-up) |
191
+ | `@tanstack/ai-groq` | Legacy `structuredOutputStream` only (no tools — Groq's API rejects schema + tools + stream) |
192
+ | All other adapters (ollama, older Claude, Gemini 2.x, Grok 2/3) | Fallback: runs non-streaming `structuredOutput`, emits one `structured-output.complete` event |
193
+
194
+ **Native combined mode vs fallback** is signaled by the adapter's
195
+ optional `supportsCombinedToolsAndSchema(modelOptions)` method. When
196
+ it returns `true`, the engine wires the JSON Schema into the regular
197
+ `chatStream` call and harvests the final-turn text — middleware sees
198
+ the run through `beforeModel` / `modelStream` as usual, and the
199
+ `'structuredOutput'` middleware phase does **not** fire. When it
200
+ returns `false` (or is omitted), the engine takes the legacy
201
+ finalization path: agent loop, then a separate `structuredOutput` /
202
+ `structuredOutputStream` call with `'structuredOutput'` phase tagging.
203
+
204
+ Consumer code is identical across providers — always read the final object off `structured-output.complete`.
193
205
 
194
206
  ### Pattern 4: useChat with outputSchema (progressive UI)
195
207
 
@@ -123,6 +123,29 @@ export interface TextAdapter<
123
123
  structuredOutputStream?: (
124
124
  options: StructuredOutputOptions<TProviderOptions>,
125
125
  ) => AsyncIterable<StreamChunk>
126
+
127
+ /**
128
+ * Declares whether the adapter supports combining `tools` and a
129
+ * schema-constrained final answer in a single streaming request.
130
+ *
131
+ * When `true`, the engine wires `outputSchema` into the regular
132
+ * `chatStream()` call and skips the separate `runStructuredFinalization`
133
+ * round-trip. The model's natural final turn carries the
134
+ * schema-constrained JSON text and the engine harvests it from the agent
135
+ * loop's accumulated content.
136
+ *
137
+ * When `false`, `undefined`, or the method is omitted, the engine runs
138
+ * the agent loop without `outputSchema` and then issues a separate
139
+ * `structuredOutput` / `structuredOutputStream` call against the JSON
140
+ * schema for finalization (the legacy path).
141
+ *
142
+ * The method receives the per-call `modelOptions` so providers whose
143
+ * support depends on the resolved upstream model (e.g. OpenRouter) can
144
+ * answer per-request. Most adapters can return a constant.
145
+ */
146
+ supportsCombinedToolsAndSchema?: (
147
+ modelOptions?: TProviderOptions | undefined,
148
+ ) => boolean
126
149
  }
127
150
 
128
151
  /**