@tanstack/ai 0.24.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
- import { DefaultMessageMetadataByModality, JSONSchema, Modality, StreamChunk, TextOptions } from '../../types.js';
1
+ import { DefaultMessageMetadataByModality, JSONSchema, Modality, StreamChunk, TextOptions, TokenUsage } from '../../types.js';
2
2
  /**
3
3
  * Configuration for adapter instances
4
4
  */
@@ -31,6 +31,8 @@ export interface StructuredOutputResult<T = unknown> {
31
31
  data: T;
32
32
  /** The raw text response from the model before parsing */
33
33
  rawText: string;
34
+ /** Token usage information (if provided by the adapter) */
35
+ usage?: TokenUsage;
34
36
  }
35
37
  /**
36
38
  * Text adapter interface with pre-resolved generics.
@@ -1 +1 @@
1
- {"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Declares whether the adapter supports combining `tools` and a\n * schema-constrained final answer in a single streaming request.\n *\n * When `true`, the engine wires `outputSchema` into the regular\n * `chatStream()` call and skips the separate `runStructuredFinalization`\n * round-trip. The model's natural final turn carries the\n * schema-constrained JSON text and the engine harvests it from the agent\n * loop's accumulated content.\n *\n * When `false`, `undefined`, or the method is omitted, the engine runs\n * the agent loop without `outputSchema` and then issues a separate\n * `structuredOutput` / `structuredOutputStream` call against the JSON\n * schema for finalization (the legacy path).\n *\n * The method receives the per-call `modelOptions` so providers whose\n * support depends on the resolved upstream model (e.g. OpenRouter) can\n * answer per-request. Most adapters can return a constant.\n */\n supportsCombinedToolsAndSchema?: (\n modelOptions?: TProviderOptions | undefined,\n ) => boolean\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"AAkKO,MAAe,gBAgBpB;AAAA,EACS,OAAO;AAAA,EAEP;AAAA,EAYC;AAAA,EAEV,YAAY,SAA4B,CAAA,GAAI,OAAe;AACzD,SAAK,SAAS;AACd,SAAK,QAAQ;AAAA,EACf;AAAA,EAcU,aAAqB;AAC7B,WAAO,GAAG,KAAK,IAAI,IAAI,KAAK,KAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,UAAU,CAAC,CAAC;AAAA,EAC9E;AACF;"}
1
+ {"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n TokenUsage,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n /** Token usage information (if provided by the adapter) */\n usage?: TokenUsage\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Declares whether the adapter supports combining `tools` and a\n * schema-constrained final answer in a single streaming request.\n *\n * When `true`, the engine wires `outputSchema` into the regular\n * `chatStream()` call and skips the separate `runStructuredFinalization`\n * round-trip. The model's natural final turn carries the\n * schema-constrained JSON text and the engine harvests it from the agent\n * loop's accumulated content.\n *\n * When `false`, `undefined`, or the method is omitted, the engine runs\n * the agent loop without `outputSchema` and then issues a separate\n * `structuredOutput` / `structuredOutputStream` call against the JSON\n * schema for finalization (the legacy path).\n *\n * The method receives the per-call `modelOptions` so providers whose\n * support depends on the resolved upstream model (e.g. OpenRouter) can\n * answer per-request. Most adapters can return a constant.\n */\n supportsCombinedToolsAndSchema?: (\n modelOptions?: TProviderOptions | undefined,\n ) => boolean\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"AAqKO,MAAe,gBAgBpB;AAAA,EACS,OAAO;AAAA,EAEP;AAAA,EAYC;AAAA,EAEV,YAAY,SAA4B,CAAA,GAAI,OAAe;AACzD,SAAK,SAAS;AACd,SAAK,QAAQ;AAAA,EACf;AAAA,EAcU,aAAqB;AAC7B,WAAO,GAAG,KAAK,IAAI,IAAI,KAAK,KAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,UAAU,CAAC,CAAC;AAAA,EAC9E;AACF;"}
@@ -1,4 +1,4 @@
1
- import { JSONSchema, ModelMessage, StreamChunk, Tool, ToolCall, UsageTotals } from '../../../types.js';
1
+ import { JSONSchema, ModelMessage, StreamChunk, TokenUsage, Tool, ToolCall } from '../../../types.js';
2
2
  import { SystemPrompt } from '../../../system-prompts.js';
3
3
  /**
4
4
  * Phase of the chat middleware lifecycle.
@@ -204,11 +204,11 @@ export interface ToolPhaseCompleteInfo {
204
204
  * Token usage statistics passed to the onUsage hook.
205
205
  * Extracted from the RUN_FINISHED chunk when usage data is present.
206
206
  *
207
- * Includes optional provider-reported `cost`/`costDetails` (see {@link UsageTotals}).
208
- * Kept as an interface extending `UsageTotals` to preserve declaration merging for
207
+ * Includes optional provider-reported `cost`/`costDetails` (see {@link TokenUsage}).
208
+ * Kept as an interface extending `TokenUsage` to preserve declaration merging for
209
209
  * this publicly exported type.
210
210
  */
211
- export interface UsageInfo extends UsageTotals {
211
+ export interface UsageInfo extends TokenUsage {
212
212
  }
213
213
  /**
214
214
  * Information passed to onFinish.
@@ -221,7 +221,7 @@ export interface FinishInfo {
221
221
  /** Final accumulated text content */
222
222
  content: string;
223
223
  /** Final usage totals, if available (optionally including provider-reported cost) */
224
- usage?: UsageTotals | undefined;
224
+ usage?: TokenUsage | undefined;
225
225
  }
226
226
  /**
227
227
  * Information passed to onAbort.
@@ -45,6 +45,15 @@ async function runGenerateAudio(options) {
45
45
  modelOptions: rest.modelOptions,
46
46
  timestamp: Date.now()
47
47
  });
48
+ if (result.usage) {
49
+ aiEventClient.emit("audio:usage", {
50
+ requestId,
51
+ model,
52
+ usage: result.usage,
53
+ modelOptions: rest.modelOptions,
54
+ timestamp: Date.now()
55
+ });
56
+ }
48
57
  logger.output(`activity=generateAudio provider=${providerName}`, {
49
58
  contentType: result.audio.contentType,
50
59
  audioDuration: result.audio.duration
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":["../../../../src/activities/generateAudio/index.ts"],"sourcesContent":["/**\n * Audio Generation Activity\n *\n * Generates audio (music, sound effects, etc.) from text prompts.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { AudioAdapter } from './adapter'\nimport type { AudioGenerationResult, StreamChunk } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'audio' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from an AudioAdapter via ~types.\n */\nexport type AudioProviderOptions<TAdapter> = TAdapter extends {\n '~types': { providerOptions: infer P extends object }\n}\n ? P\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the audio generation activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The audio adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface AudioActivityOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The audio adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** Text description of the desired audio */\n prompt: string\n /** Desired duration in seconds */\n duration?: number\n /** Provider-specific options for audio generation */\n modelOptions?: AudioProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<AudioGenerationResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the audio generation activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<AudioGenerationResult>\n */\nexport type AudioActivityResult<TStream extends boolean = false> =\n TStream extends true\n ? AsyncIterable<StreamChunk>\n : Promise<AudioGenerationResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * Audio generation activity - generates audio from text prompts.\n *\n * Uses AI models to create music, sound effects, and other audio content.\n *\n * @example Generate music from a prompt\n * ```ts\n * import { generateAudio } from '@tanstack/ai'\n * import { falAudio } from '@tanstack/ai-fal'\n *\n * const result = await generateAudio({\n * adapter: falAudio('fal-ai/diffrhythm'),\n * prompt: 'An upbeat electronic track with synths',\n * duration: 10\n * })\n *\n * console.log(result.audio.url) // URL to generated audio\n * ```\n */\nexport function generateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateAudio(options),\n ) as AudioActivityResult<TStream>\n }\n return runGenerateAudio(options) as AudioActivityResult<TStream>\n}\n\n/**\n * Run the core audio generation logic (non-streaming).\n */\nasync function runGenerateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n>(\n options: AudioActivityOptions<TAdapter, boolean>,\n): Promise<AudioGenerationResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('audio')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('audio:request:started', {\n requestId,\n provider: adapter.name,\n model,\n prompt: rest.prompt,\n duration: rest.duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateAudio provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateAudio({ ...rest, model, logger })\n const elapsedMs = Date.now() - startTime\n\n aiEventClient.emit('audio:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateAudio provider=${providerName}`, {\n contentType: result.audio.contentType,\n audioDuration: result.audio.duration,\n })\n\n return result\n } catch (error) {\n const elapsedMs = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('audio:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateAudio activity failed', {\n error,\n source: 'generateAudio',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateAudio() function without executing.\n */\nexport function createAudioOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type {\n AudioAdapter,\n AudioAdapterConfig,\n AnyAudioAdapter,\n} from './adapter'\nexport { BaseAudioAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAoEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAyBO,SAAS,cAId,SAC8B;AAC9B,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,iBAAiB,OAAO;AAAA,IAAA;AAAA,EAE5B;AACA,SAAO,iBAAiB,OAAO;AACjC;AAKA,eAAe,iBAGb,SACgC;AAChC,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,OAAO;AAClC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,yBAAyB;AAAA,IAC1C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,UAAU,KAAK;AAAA,IACf,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,mCAAmC,YAAY,IAAI;AAAA,IAChE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,cAAc,EAAE,GAAG,MAAM,OAAO,QAAQ;AACrE,UAAM,YAAY,KAAK,IAAA,IAAQ;AAE/B,kBAAc,KAAK,2BAA2B;AAAA,MAC5C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,mCAAmC,YAAY,IAAI;AAAA,MAC/D,aAAa,OAAO,MAAM;AAAA,MAC1B,eAAe,OAAO,MAAM;AAAA,IAAA,CAC7B;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,YAAY,KAAK,IAAA,IAAQ;AAC/B,UAAM,MAAM;AACZ,kBAAc,KAAK,uBAAuB;AAAA,MACxC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,iCAAiC;AAAA,MAC7C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,mBAId,SACyC;AACzC,SAAO;AACT;"}
1
+ {"version":3,"file":"index.js","sources":["../../../../src/activities/generateAudio/index.ts"],"sourcesContent":["/**\n * Audio Generation Activity\n *\n * Generates audio (music, sound effects, etc.) from text prompts.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { AudioAdapter } from './adapter'\nimport type { AudioGenerationResult, StreamChunk } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'audio' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from an AudioAdapter via ~types.\n */\nexport type AudioProviderOptions<TAdapter> = TAdapter extends {\n '~types': { providerOptions: infer P extends object }\n}\n ? P\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the audio generation activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The audio adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface AudioActivityOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The audio adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** Text description of the desired audio */\n prompt: string\n /** Desired duration in seconds */\n duration?: number\n /** Provider-specific options for audio generation */\n modelOptions?: AudioProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<AudioGenerationResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the audio generation activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<AudioGenerationResult>\n */\nexport type AudioActivityResult<TStream extends boolean = false> =\n TStream extends true\n ? AsyncIterable<StreamChunk>\n : Promise<AudioGenerationResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * Audio generation activity - generates audio from text prompts.\n *\n * Uses AI models to create music, sound effects, and other audio content.\n *\n * @example Generate music from a prompt\n * ```ts\n * import { generateAudio } from '@tanstack/ai'\n * import { falAudio } from '@tanstack/ai-fal'\n *\n * const result = await generateAudio({\n * adapter: falAudio('fal-ai/diffrhythm'),\n * prompt: 'An upbeat electronic track with synths',\n * duration: 10\n * })\n *\n * console.log(result.audio.url) // URL to generated audio\n * ```\n */\nexport function generateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateAudio(options),\n ) as AudioActivityResult<TStream>\n }\n return runGenerateAudio(options) as AudioActivityResult<TStream>\n}\n\n/**\n * Run the core audio generation logic (non-streaming).\n */\nasync function runGenerateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n>(\n options: AudioActivityOptions<TAdapter, boolean>,\n): Promise<AudioGenerationResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('audio')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('audio:request:started', {\n requestId,\n provider: adapter.name,\n model,\n prompt: rest.prompt,\n duration: rest.duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateAudio provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateAudio({ ...rest, model, logger })\n const elapsedMs = Date.now() - startTime\n\n aiEventClient.emit('audio:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n if (result.usage) {\n aiEventClient.emit('audio:usage', {\n requestId,\n model,\n usage: result.usage,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n }\n\n logger.output(`activity=generateAudio provider=${providerName}`, {\n contentType: result.audio.contentType,\n audioDuration: result.audio.duration,\n })\n\n return result\n } catch (error) {\n const elapsedMs = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('audio:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateAudio activity failed', {\n error,\n source: 'generateAudio',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateAudio() function without executing.\n */\nexport function createAudioOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type {\n AudioAdapter,\n AudioAdapterConfig,\n AnyAudioAdapter,\n} from './adapter'\nexport { BaseAudioAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAoEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAyBO,SAAS,cAId,SAC8B;AAC9B,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,iBAAiB,OAAO;AAAA,IAAA;AAAA,EAE5B;AACA,SAAO,iBAAiB,OAAO;AACjC;AAKA,eAAe,iBAGb,SACgC;AAChC,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,OAAO;AAClC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,yBAAyB;AAAA,IAC1C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,UAAU,KAAK;AAAA,IACf,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,mCAAmC,YAAY,IAAI;AAAA,IAChE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,cAAc,EAAE,GAAG,MAAM,OAAO,QAAQ;AACrE,UAAM,YAAY,KAAK,IAAA,IAAQ;AAE/B,kBAAc,KAAK,2BAA2B;AAAA,MAC5C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,QAAI,OAAO,OAAO;AAChB,oBAAc,KAAK,eAAe;AAAA,QAChC;AAAA,QACA;AAAA,QACA,OAAO,OAAO;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,WAAW,KAAK,IAAA;AAAA,MAAI,CACrB;AAAA,IACH;AAEA,WAAO,OAAO,mCAAmC,YAAY,IAAI;AAAA,MAC/D,aAAa,OAAO,MAAM;AAAA,MAC1B,eAAe,OAAO,MAAM;AAAA,IAAA,CAC7B;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,YAAY,KAAK,IAAA,IAAQ;AAC/B,UAAM,MAAM;AACZ,kBAAc,KAAK,uBAAuB;AAAA,MACxC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,iCAAiC;AAAA,MAC7C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,mBAId,SACyC;AACzC,SAAO;AACT;"}
@@ -50,6 +50,15 @@ async function runGenerateSpeech(options) {
50
50
  modelOptions: rest.modelOptions,
51
51
  timestamp: Date.now()
52
52
  });
53
+ if (result.usage) {
54
+ aiEventClient.emit("speech:usage", {
55
+ requestId,
56
+ model,
57
+ usage: result.usage,
58
+ modelOptions: rest.modelOptions,
59
+ timestamp: Date.now()
60
+ });
61
+ }
53
62
  logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {
54
63
  bytes: result.audio.length,
55
64
  contentType: result.contentType
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiSpeech } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
1
+ {"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiSpeech } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n if (result.usage) {\n aiEventClient.emit('speech:usage', {\n requestId,\n model,\n usage: result.usage,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n }\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,QAAI,OAAO,OAAO;AAChB,oBAAc,KAAK,gBAAgB;AAAA,QACjC;AAAA,QACA;AAAA,QACA,OAAO,OAAO;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,WAAW,KAAK,IAAA;AAAA,MAAI,CACrB;AAAA,IACH;AAEA,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
@@ -13,13 +13,24 @@ import { Modality } from './types.js';
13
13
  * ] as const
14
14
  * ```
15
15
  */
16
- export interface ExtendedModelDef<TName extends string = string, TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>, TOptions = unknown> {
16
+ export interface ExtendedModelDef<TName extends string = string, TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>, TOptions = unknown, TFeatures extends ReadonlyArray<string> = ReadonlyArray<string>, TTools extends ReadonlyArray<string> = ReadonlyArray<string>> {
17
17
  /** The model name identifier */
18
18
  name: TName;
19
19
  /** Supported input modalities for this model */
20
20
  input: TInput;
21
21
  /** Type brand for provider options - use `{} as YourOptionsType` */
22
22
  modelOptions: TOptions;
23
+ /** Optional declared features (e.g. 'reasoning', 'structured_outputs') */
24
+ features?: TFeatures;
25
+ /** Optional declared provider tools (e.g. 'web_search') */
26
+ tools?: TTools;
27
+ }
28
+ /** Capability bag accepted by the object form of `createModel`. */
29
+ export interface ModelCapabilities<TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>, TFeatures extends ReadonlyArray<string> = ReadonlyArray<string>, TTools extends ReadonlyArray<string> = ReadonlyArray<string>, TOptions = unknown> {
30
+ input?: TInput;
31
+ features?: TFeatures;
32
+ tools?: TTools;
33
+ modelOptions?: TOptions;
23
34
  }
24
35
  /**
25
36
  * Creates a custom model definition for use with `extendAdapter`.
@@ -47,8 +58,19 @@ export interface ExtendedModelDef<TName extends string = string, TInput extends
47
58
  *
48
59
  * const myOpenai = extendAdapter(openaiText, customModels)
49
60
  * ```
61
+ *
62
+ * @example
63
+ * ```typescript
64
+ * // Capabilities object form - declare features and provider tools
65
+ * const reasoner = createModel('reasoner', {
66
+ * input: ['text'],
67
+ * features: ['reasoning', 'structured_outputs'],
68
+ * tools: ['web_search'],
69
+ * })
70
+ * ```
50
71
  */
51
72
  export declare function createModel<const TName extends string, const TInput extends ReadonlyArray<Modality>>(name: TName, input: TInput): ExtendedModelDef<TName, TInput>;
73
+ export declare function createModel<const TName extends string, const TCaps extends ModelCapabilities>(name: TName, capabilities: TCaps): ExtendedModelDef<TName, TCaps['input'] extends ReadonlyArray<Modality> ? TCaps['input'] : ReadonlyArray<Modality>, TCaps['modelOptions'], TCaps['features'] extends ReadonlyArray<string> ? TCaps['features'] : ReadonlyArray<string>, TCaps['tools'] extends ReadonlyArray<string> ? TCaps['tools'] : ReadonlyArray<string>>;
52
74
  /**
53
75
  * Extract the model name union from an array of model definitions.
54
76
  */
@@ -1,8 +1,14 @@
1
- function createModel(name, input) {
1
+ function createModel(name, second) {
2
+ if (Array.isArray(second)) {
3
+ return { name, input: second, modelOptions: {} };
4
+ }
5
+ const caps = second;
2
6
  return {
3
7
  name,
4
- input,
5
- modelOptions: {}
8
+ input: caps.input ?? ["text"],
9
+ modelOptions: caps.modelOptions ?? {},
10
+ features: caps.features,
11
+ tools: caps.tools
6
12
  };
7
13
  }
8
14
  function extendAdapter(factory, _customModels) {
@@ -1 +1 @@
1
- {"version":3,"file":"extend-adapter.js","sources":["../../src/extend-adapter.ts"],"sourcesContent":["import type { Modality } from './types'\n\n// ===========================\n// Extended Model Definition\n// ===========================\n\n/**\n * Definition for a custom model to add to an adapter.\n *\n * @template TName - The model name as a literal string type\n * @template TInput - Array of supported input modalities\n * @template TOptions - Provider options type for this model\n *\n * @example\n * ```typescript\n * const customModels = [\n * createModel('my-custom-model', ['text', 'image']),\n * ] as const\n * ```\n */\nexport interface ExtendedModelDef<\n TName extends string = string,\n TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>,\n TOptions = unknown,\n> {\n /** The model name identifier */\n name: TName\n /** Supported input modalities for this model */\n input: TInput\n /** Type brand for provider options - use `{} as YourOptionsType` */\n modelOptions: TOptions\n}\n\n/**\n * Creates a custom model definition for use with `extendAdapter`.\n *\n * This is a helper function that provides proper type inference without\n * requiring manual `as const` casts on individual properties.\n *\n * @template TName - The model name (inferred from argument)\n * @template TInput - The input modalities array (inferred from argument)\n *\n * @param name - The model name identifier (literal string)\n * @param input - Array of supported input modalities\n * @returns A properly typed model definition for use with `extendAdapter`\n *\n * @example\n * ```typescript\n * import { extendAdapter, createModel } from '@tanstack/ai'\n * import { openaiText } from '@tanstack/ai-openai'\n *\n * // Define custom models with full type inference\n * const customModels = [\n * createModel('my-fine-tuned-gpt4', ['text', 'image']),\n * createModel('local-llama', ['text']),\n * ] as const\n *\n * const myOpenai = extendAdapter(openaiText, customModels)\n * ```\n */\nexport function createModel<\n const TName extends string,\n const TInput extends ReadonlyArray<Modality>,\n>(name: TName, input: TInput): ExtendedModelDef<TName, TInput> {\n return {\n name,\n input,\n modelOptions: {},\n }\n}\n\n// ===========================\n// Type Extraction Utilities\n// ===========================\n\n/**\n * Extract the model name union from an array of model definitions.\n */\ntype ExtractCustomModelNames<TDefs extends ReadonlyArray<ExtendedModelDef>> =\n TDefs[number]['name']\n\n// ===========================\n// Factory Type Inference\n// ===========================\n\n/**\n * Infer the model parameter type from an adapter factory function.\n * For generic functions like `<T extends Union>(model: T)`, this gets `T` which\n * TypeScript treats as the constraint union when used in parameter position.\n */\ntype InferFactoryModels<TFactory> = TFactory extends (\n model: infer TModel,\n ...args: Array<any>\n) => any\n ? TModel extends string\n ? TModel\n : string\n : string\n\n/**\n * Infer the config parameter type from an adapter factory function.\n */\ntype InferConfig<TFactory> = TFactory extends (\n model: any,\n config?: infer TConfig,\n) => any\n ? TConfig\n : undefined\n\n/**\n * Infer the adapter return type from a factory function.\n */\ntype InferAdapterReturn<TFactory> = TFactory extends (\n ...args: Array<any>\n) => infer TReturn\n ? TReturn\n : never\n\n// ===========================\n// extendAdapter Function\n// ===========================\n\n/**\n * Extends an existing adapter factory with additional custom models.\n *\n * The extended adapter accepts both original models (with full original type inference)\n * and custom models (with types from your definitions).\n *\n * At runtime, this simply passes through to the original factory - no validation is performed.\n * The original factory's signature is fully preserved, including any config parameters.\n *\n * @param factory - The original adapter factory function (e.g., `openaiText`, `anthropicText`)\n * @param models - Array of custom model definitions with `name` and `input`\n * @returns A new factory function that accepts both original and custom models\n *\n * @example\n * ```typescript\n * import { extendAdapter, createModel } from '@tanstack/ai'\n * import { openaiText } from '@tanstack/ai-openai'\n *\n * // Define custom models\n * const customModels = [\n * createModel('my-fine-tuned-gpt4', ['text', 'image']),\n * createModel('local-llama', ['text']),\n * ] as const\n *\n * // Create extended adapter\n * const myOpenai = extendAdapter(openaiText, customModels)\n *\n * // Use with original models - full type inference preserved\n * const gpt4 = myOpenai('gpt-4o')\n *\n * // Use with custom models\n * const custom = myOpenai('my-fine-tuned-gpt4')\n *\n * // Type error: 'invalid-model' is not a valid model\n * // myOpenai('invalid-model')\n *\n * // Works with chat()\n * chat({\n * adapter: myOpenai('my-fine-tuned-gpt4'),\n * messages: [...]\n * })\n * ```\n */\nexport function extendAdapter<\n TFactory extends (...args: Array<any>) => any,\n const TDefs extends ReadonlyArray<ExtendedModelDef>,\n>(\n factory: TFactory,\n _customModels: TDefs,\n): (\n model: InferFactoryModels<TFactory> | ExtractCustomModelNames<TDefs>,\n ...args: InferConfig<TFactory> extends undefined\n ? []\n : [config?: InferConfig<TFactory>]\n) => InferAdapterReturn<TFactory> {\n // At runtime, we simply pass through to the original factory.\n // The _customModels parameter is only used for type inference.\n // No runtime validation - users are trusted to pass valid model names.\n return factory as any\n}\n"],"names":[],"mappings":"AA4DO,SAAS,YAGd,MAAa,OAAgD;AAC7D,SAAO;AAAA,IACL;AAAA,IACA;AAAA,IACA,cAAc,CAAA;AAAA,EAAC;AAEnB;AAgGO,SAAS,cAId,SACA,eAMgC;AAIhC,SAAO;AACT;"}
1
+ {"version":3,"file":"extend-adapter.js","sources":["../../src/extend-adapter.ts"],"sourcesContent":["import type { Modality } from './types'\n\n// ===========================\n// Extended Model Definition\n// ===========================\n\n/**\n * Definition for a custom model to add to an adapter.\n *\n * @template TName - The model name as a literal string type\n * @template TInput - Array of supported input modalities\n * @template TOptions - Provider options type for this model\n *\n * @example\n * ```typescript\n * const customModels = [\n * createModel('my-custom-model', ['text', 'image']),\n * ] as const\n * ```\n */\nexport interface ExtendedModelDef<\n TName extends string = string,\n TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>,\n TOptions = unknown,\n TFeatures extends ReadonlyArray<string> = ReadonlyArray<string>,\n TTools extends ReadonlyArray<string> = ReadonlyArray<string>,\n> {\n /** The model name identifier */\n name: TName\n /** Supported input modalities for this model */\n input: TInput\n /** Type brand for provider options - use `{} as YourOptionsType` */\n modelOptions: TOptions\n /** Optional declared features (e.g. 'reasoning', 'structured_outputs') */\n features?: TFeatures\n /** Optional declared provider tools (e.g. 'web_search') */\n tools?: TTools\n}\n\n/** Capability bag accepted by the object form of `createModel`. */\nexport interface ModelCapabilities<\n TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>,\n TFeatures extends ReadonlyArray<string> = ReadonlyArray<string>,\n TTools extends ReadonlyArray<string> = ReadonlyArray<string>,\n TOptions = unknown,\n> {\n input?: TInput\n features?: TFeatures\n tools?: TTools\n modelOptions?: TOptions\n}\n\n/**\n * Creates a custom model definition for use with `extendAdapter`.\n *\n * This is a helper function that provides proper type inference without\n * requiring manual `as const` casts on individual properties.\n *\n * @template TName - The model name (inferred from argument)\n * @template TInput - The input modalities array (inferred from argument)\n *\n * @param name - The model name identifier (literal string)\n * @param input - Array of supported input modalities\n * @returns A properly typed model definition for use with `extendAdapter`\n *\n * @example\n * ```typescript\n * import { extendAdapter, createModel } from '@tanstack/ai'\n * import { openaiText } from '@tanstack/ai-openai'\n *\n * // Define custom models with full type inference\n * const customModels = [\n * createModel('my-fine-tuned-gpt4', ['text', 'image']),\n * createModel('local-llama', ['text']),\n * ] as const\n *\n * const myOpenai = extendAdapter(openaiText, customModels)\n * ```\n *\n * @example\n * ```typescript\n * // Capabilities object form - declare features and provider tools\n * const reasoner = createModel('reasoner', {\n * input: ['text'],\n * features: ['reasoning', 'structured_outputs'],\n * tools: ['web_search'],\n * })\n * ```\n */\n// Overload 1 — legacy positional input array (unchanged behavior)\nexport function createModel<\n const TName extends string,\n const TInput extends ReadonlyArray<Modality>,\n>(name: TName, input: TInput): ExtendedModelDef<TName, TInput>\n// Overload 2 — capabilities object\nexport function createModel<\n const TName extends string,\n const TCaps extends ModelCapabilities,\n>(\n name: TName,\n capabilities: TCaps,\n): ExtendedModelDef<\n TName,\n TCaps['input'] extends ReadonlyArray<Modality>\n ? TCaps['input']\n : ReadonlyArray<Modality>,\n TCaps['modelOptions'],\n TCaps['features'] extends ReadonlyArray<string>\n ? TCaps['features']\n : ReadonlyArray<string>,\n TCaps['tools'] extends ReadonlyArray<string>\n ? TCaps['tools']\n : ReadonlyArray<string>\n>\n// Implementation\nexport function createModel(\n name: string,\n second: ReadonlyArray<Modality> | ModelCapabilities,\n): ExtendedModelDef {\n if (Array.isArray(second)) {\n return { name, input: second, modelOptions: {} }\n }\n const caps = second as ModelCapabilities\n return {\n name,\n input: caps.input ?? (['text'] as ReadonlyArray<Modality>),\n modelOptions: caps.modelOptions ?? {},\n features: caps.features,\n tools: caps.tools,\n }\n}\n\n// ===========================\n// Type Extraction Utilities\n// ===========================\n\n/**\n * Extract the model name union from an array of model definitions.\n */\ntype ExtractCustomModelNames<TDefs extends ReadonlyArray<ExtendedModelDef>> =\n TDefs[number]['name']\n\n// ===========================\n// Factory Type Inference\n// ===========================\n\n/**\n * Infer the model parameter type from an adapter factory function.\n * For generic functions like `<T extends Union>(model: T)`, this gets `T` which\n * TypeScript treats as the constraint union when used in parameter position.\n */\ntype InferFactoryModels<TFactory> = TFactory extends (\n model: infer TModel,\n ...args: Array<any>\n) => any\n ? TModel extends string\n ? TModel\n : string\n : string\n\n/**\n * Infer the config parameter type from an adapter factory function.\n */\ntype InferConfig<TFactory> = TFactory extends (\n model: any,\n config?: infer TConfig,\n) => any\n ? TConfig\n : undefined\n\n/**\n * Infer the adapter return type from a factory function.\n */\ntype InferAdapterReturn<TFactory> = TFactory extends (\n ...args: Array<any>\n) => infer TReturn\n ? TReturn\n : never\n\n// ===========================\n// extendAdapter Function\n// ===========================\n\n/**\n * Extends an existing adapter factory with additional custom models.\n *\n * The extended adapter accepts both original models (with full original type inference)\n * and custom models (with types from your definitions).\n *\n * At runtime, this simply passes through to the original factory - no validation is performed.\n * The original factory's signature is fully preserved, including any config parameters.\n *\n * @param factory - The original adapter factory function (e.g., `openaiText`, `anthropicText`)\n * @param models - Array of custom model definitions with `name` and `input`\n * @returns A new factory function that accepts both original and custom models\n *\n * @example\n * ```typescript\n * import { extendAdapter, createModel } from '@tanstack/ai'\n * import { openaiText } from '@tanstack/ai-openai'\n *\n * // Define custom models\n * const customModels = [\n * createModel('my-fine-tuned-gpt4', ['text', 'image']),\n * createModel('local-llama', ['text']),\n * ] as const\n *\n * // Create extended adapter\n * const myOpenai = extendAdapter(openaiText, customModels)\n *\n * // Use with original models - full type inference preserved\n * const gpt4 = myOpenai('gpt-4o')\n *\n * // Use with custom models\n * const custom = myOpenai('my-fine-tuned-gpt4')\n *\n * // Type error: 'invalid-model' is not a valid model\n * // myOpenai('invalid-model')\n *\n * // Works with chat()\n * chat({\n * adapter: myOpenai('my-fine-tuned-gpt4'),\n * messages: [...]\n * })\n * ```\n */\nexport function extendAdapter<\n TFactory extends (...args: Array<any>) => any,\n const TDefs extends ReadonlyArray<ExtendedModelDef>,\n>(\n factory: TFactory,\n _customModels: TDefs,\n): (\n model: InferFactoryModels<TFactory> | ExtractCustomModelNames<TDefs>,\n ...args: InferConfig<TFactory> extends undefined\n ? []\n : [config?: InferConfig<TFactory>]\n) => InferAdapterReturn<TFactory> {\n // At runtime, we simply pass through to the original factory.\n // The _customModels parameter is only used for type inference.\n // No runtime validation - users are trusted to pass valid model names.\n return factory as any\n}\n"],"names":[],"mappings":"AAmHO,SAAS,YACd,MACA,QACkB;AAClB,MAAI,MAAM,QAAQ,MAAM,GAAG;AACzB,WAAO,EAAE,MAAM,OAAO,QAAQ,cAAc,CAAA,EAAC;AAAA,EAC/C;AACA,QAAM,OAAO;AACb,SAAO;AAAA,IACL;AAAA,IACA,OAAO,KAAK,SAAU,CAAC,MAAM;AAAA,IAC7B,cAAc,KAAK,gBAAgB,CAAA;AAAA,IACnC,UAAU,KAAK;AAAA,IACf,OAAO,KAAK;AAAA,EAAA;AAEhB;AAgGO,SAAS,cAId,SACA,eAMgC;AAIhC,SAAO;AACT;"}
@@ -17,6 +17,7 @@ export { maxIterations, untilFinishReason, combineStrategies, } from './activiti
17
17
  export { createToolRegistry, createFrozenRegistry, type ToolRegistry, } from './tool-registry.js';
18
18
  export type { ChatMiddleware, ChatMiddlewareContext, ChatMiddlewarePhase, ChatMiddlewareConfig, StructuredOutputMiddlewareConfig, ToolCallHookContext, BeforeToolCallDecision, AfterToolCallInfo, IterationInfo, ToolPhaseCompleteInfo, UsageInfo, FinishInfo, AbortInfo, ErrorInfo, } from './activities/chat/middleware/index.js';
19
19
  export * from './types.js';
20
+ export { buildBaseUsage, type BaseUsageInput } from './utilities/usage.js';
20
21
  export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts.js';
21
22
  export { normalizeSystemPrompts } from './system-prompts.js';
22
23
  export { detectImageMimeType } from './utils.js';
@@ -30,6 +31,6 @@ export { uiMessagesToWire } from './utilities/ag-ui-wire.js';
30
31
  export type { WireMessage } from './utilities/ag-ui-wire.js';
31
32
  export { isContentPart, isContentPartArray, normalizeToolResult, } from './utilities/tool-result.js';
32
33
  export { createModel, extendAdapter } from './extend-adapter.js';
33
- export type { ExtendedModelDef } from './extend-adapter.js';
34
+ export type { ExtendedModelDef, ModelCapabilities } from './extend-adapter.js';
34
35
  export type { Logger, DebugCategories, DebugConfig, DebugOption, } from './logger/types.js';
35
36
  export { ConsoleLogger } from './logger/console-logger.js';
package/dist/esm/index.js CHANGED
@@ -12,6 +12,7 @@ import { ToolCallManager } from "./activities/chat/tools/tool-calls.js";
12
12
  import { brandProviderTool } from "./tools/provider-tool.js";
13
13
  import { combineStrategies, maxIterations, untilFinishReason } from "./activities/chat/agent-loop-strategies.js";
14
14
  import { createFrozenRegistry, createToolRegistry } from "./tool-registry.js";
15
+ import { buildBaseUsage } from "./utilities/usage.js";
15
16
  import { normalizeSystemPrompts } from "./system-prompts.js";
16
17
  import { detectImageMimeType } from "./utils.js";
17
18
  import { realtimeToken } from "./realtime/index.js";
@@ -38,6 +39,7 @@ export {
38
39
  ToolCallManager,
39
40
  WordBoundaryStrategy,
40
41
  brandProviderTool,
42
+ buildBaseUsage,
41
43
  chat,
42
44
  chatParamsFromRequest,
43
45
  chatParamsFromRequestBody,
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;"}
1
+ {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;"}
@@ -1,6 +1,7 @@
1
1
  import { StandardJSONSchemaV1, StandardSchemaV1 } from '@standard-schema/spec';
2
2
  import { InternalLogger } from './logger/internal-logger.js';
3
3
  import { SystemPrompt } from './system-prompts.js';
4
+ import { CompletionTokensDetails, PromptTokensDetails, ProviderUsageDetails, TokenUsage, UsageCostBreakdown } from '@tanstack/ai-event-client';
4
5
  import { BaseEvent as AGUIBaseEvent, CustomEvent as AGUICustomEvent, MessagesSnapshotEvent as AGUIMessagesSnapshotEvent, ReasoningEncryptedValueEvent as AGUIReasoningEncryptedValueEvent, ReasoningEndEvent as AGUIReasoningEndEvent, ReasoningMessageContentEvent as AGUIReasoningMessageContentEvent, ReasoningMessageEndEvent as AGUIReasoningMessageEndEvent, ReasoningMessageStartEvent as AGUIReasoningMessageStartEvent, ReasoningStartEvent as AGUIReasoningStartEvent, RunErrorEvent as AGUIRunErrorEvent, RunFinishedEvent as AGUIRunFinishedEvent, RunStartedEvent as AGUIRunStartedEvent, StateDeltaEvent as AGUIStateDeltaEvent, StateSnapshotEvent as AGUIStateSnapshotEvent, StepFinishedEvent as AGUIStepFinishedEvent, StepStartedEvent as AGUIStepStartedEvent, TextMessageContentEvent as AGUITextMessageContentEvent, TextMessageEndEvent as AGUITextMessageEndEvent, TextMessageStartEvent as AGUITextMessageStartEvent, ToolCallArgsEvent as AGUIToolCallArgsEvent, ToolCallEndEvent as AGUIToolCallEndEvent, ToolCallResultEvent as AGUIToolCallResultEvent, ToolCallStartEvent as AGUIToolCallStartEvent, EventType } from '@ag-ui/core';
5
6
  /**
6
7
  * Tool call states - track the lifecycle of a tool call
@@ -787,37 +788,13 @@ export interface RunStartedEvent extends AGUIRunStartedEvent {
787
788
  /** Model identifier for multi-model support */
788
789
  model?: string;
789
790
  }
791
+ export type { CompletionTokensDetails, PromptTokensDetails, ProviderUsageDetails, TokenUsage, UsageCostBreakdown, };
790
792
  /**
791
- * Provider-reported cost breakdown for a single request, normalized onto a
792
- * canonical shape so consumer code is portable across gateways. Each adapter's
793
- * extractor maps its provider-specific wire keys (e.g. OpenRouter's
794
- * `upstream_inference_prompt_cost`, `upstream_inference_input_cost`) onto these
795
- * fields at runtime.
793
+ * @deprecated Renamed to {@link TokenUsage}. Kept as an alias for backward
794
+ * compatibility with `@tanstack/ai@0.23` and earlier; will be removed in a
795
+ * future release.
796
796
  */
797
- export interface UsageCostBreakdown {
798
- /** Total cost the gateway paid the upstream provider. */
799
- upstreamCost?: number;
800
- /** Upstream cost for input (prompt) tokens. */
801
- upstreamInputCost?: number;
802
- /** Upstream cost for output (completion) tokens. */
803
- upstreamOutputCost?: number;
804
- }
805
- /**
806
- * Token usage totals for a run, optionally including provider-reported cost.
807
- *
808
- * `cost` and `costDetails` are populated only by adapters whose provider returns
809
- * authoritative per-request cost (e.g. OpenRouter). They are absent for adapters
810
- * that do not report cost, so consumers must treat them as optional.
811
- */
812
- export interface UsageTotals {
813
- promptTokens: number;
814
- completionTokens: number;
815
- totalTokens: number;
816
- /** Provider-reported cost for the request, when available. */
817
- cost?: number;
818
- /** Provider-reported cost breakdown, when available. */
819
- costDetails?: UsageCostBreakdown;
820
- }
797
+ export type UsageTotals = TokenUsage;
821
798
  /**
822
799
  * Emitted when a run completes successfully.
823
800
  *
@@ -829,8 +806,8 @@ export interface RunFinishedEvent extends AGUIRunFinishedEvent {
829
806
  model?: string;
830
807
  /** Why the generation stopped */
831
808
  finishReason?: 'stop' | 'length' | 'content_filter' | 'tool_calls' | null;
832
- /** Token usage statistics, optionally including provider-reported cost. */
833
- usage?: UsageTotals;
809
+ /** Token usage statistics with optional detailed breakdowns and provider-reported cost. */
810
+ usage?: TokenUsage;
834
811
  }
835
812
  /**
836
813
  * Emitted when an error occurs during a run.
@@ -1219,11 +1196,7 @@ export interface TextCompletionChunk {
1219
1196
  content: string;
1220
1197
  role?: 'assistant';
1221
1198
  finishReason?: 'stop' | 'length' | 'content_filter' | null;
1222
- usage?: {
1223
- promptTokens: number;
1224
- completionTokens: number;
1225
- totalTokens: number;
1226
- };
1199
+ usage?: TokenUsage;
1227
1200
  }
1228
1201
  export interface SummarizationOptions<TProviderOptions extends object = Record<string, unknown>> {
1229
1202
  model: string;
@@ -1243,11 +1216,7 @@ export interface SummarizationResult {
1243
1216
  id: string;
1244
1217
  model: string;
1245
1218
  summary: string;
1246
- usage: {
1247
- promptTokens: number;
1248
- completionTokens: number;
1249
- totalTokens: number;
1250
- };
1219
+ usage: TokenUsage;
1251
1220
  }
1252
1221
  /**
1253
1222
  * Options for image generation.
@@ -1303,11 +1272,7 @@ export interface ImageGenerationResult {
1303
1272
  /** Array of generated images */
1304
1273
  images: Array<GeneratedImage>;
1305
1274
  /** Token usage information (if available) */
1306
- usage?: {
1307
- inputTokens?: number;
1308
- outputTokens?: number;
1309
- totalTokens?: number;
1310
- };
1275
+ usage?: TokenUsage;
1311
1276
  }
1312
1277
  /**
1313
1278
  * Options for audio generation (music, sound effects, etc.).
@@ -1349,11 +1314,7 @@ export interface AudioGenerationResult {
1349
1314
  /** The generated audio */
1350
1315
  audio: GeneratedAudio;
1351
1316
  /** Token usage information (if available) */
1352
- usage?: {
1353
- inputTokens?: number;
1354
- outputTokens?: number;
1355
- totalTokens?: number;
1356
- };
1317
+ usage?: TokenUsage;
1357
1318
  }
1358
1319
  /**
1359
1320
  * Options for video generation.
@@ -1457,6 +1418,8 @@ export interface TTSResult {
1457
1418
  duration?: number;
1458
1419
  /** Content type of the audio (e.g., 'audio/mp3') */
1459
1420
  contentType?: string;
1421
+ /** Token usage information (if provided by the adapter) */
1422
+ usage?: TokenUsage;
1460
1423
  }
1461
1424
  /**
1462
1425
  * Options for audio transcription.
@@ -1528,6 +1491,8 @@ export interface TranscriptionResult {
1528
1491
  segments?: Array<TranscriptionSegment>;
1529
1492
  /** Word-level timestamps, if available */
1530
1493
  words?: Array<TranscriptionWord>;
1494
+ /** Token usage information (if provided by the adapter) */
1495
+ usage?: TokenUsage;
1531
1496
  }
1532
1497
  /**
1533
1498
  * Default metadata type for adapters that don't define custom metadata.
@@ -0,0 +1,31 @@
1
+ import { ProviderUsageDetails, TokenUsage } from '../types.js';
2
+ /**
3
+ * Input parameters for building base TokenUsage.
4
+ * Provider functions should extract these from their SDK's response.
5
+ */
6
+ export interface BaseUsageInput {
7
+ /** Total input/prompt tokens */
8
+ promptTokens: number;
9
+ /** Total output/completion tokens */
10
+ completionTokens: number;
11
+ /** Total tokens (prompt + completion) */
12
+ totalTokens: number;
13
+ }
14
+ /**
15
+ * Builds the base TokenUsage object with core fields.
16
+ * Provider-specific functions should use this and then add their own details.
17
+ *
18
+ * @param input - The base token counts
19
+ * @returns A TokenUsage object with promptTokens, completionTokens, totalTokens
20
+ *
21
+ * @example
22
+ * ```typescript
23
+ * const base = buildBaseUsage({
24
+ * promptTokens: 100,
25
+ * completionTokens: 50,
26
+ * totalTokens: 150
27
+ * });
28
+ * // Returns: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
29
+ * ```
30
+ */
31
+ export declare function buildBaseUsage<TProviderDetails = ProviderUsageDetails>(input: BaseUsageInput): TokenUsage<TProviderDetails>;
@@ -0,0 +1,11 @@
1
+ function buildBaseUsage(input) {
2
+ return {
3
+ promptTokens: input.promptTokens,
4
+ completionTokens: input.completionTokens,
5
+ totalTokens: input.totalTokens
6
+ };
7
+ }
8
+ export {
9
+ buildBaseUsage
10
+ };
11
+ //# sourceMappingURL=usage.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"usage.js","sources":["../../../src/utilities/usage.ts"],"sourcesContent":["import type { ProviderUsageDetails, TokenUsage } from '../types'\n\n/**\n * Input parameters for building base TokenUsage.\n * Provider functions should extract these from their SDK's response.\n */\nexport interface BaseUsageInput {\n /** Total input/prompt tokens */\n promptTokens: number\n /** Total output/completion tokens */\n completionTokens: number\n /** Total tokens (prompt + completion) */\n totalTokens: number\n}\n\n/**\n * Builds the base TokenUsage object with core fields.\n * Provider-specific functions should use this and then add their own details.\n *\n * @param input - The base token counts\n * @returns A TokenUsage object with promptTokens, completionTokens, totalTokens\n *\n * @example\n * ```typescript\n * const base = buildBaseUsage({\n * promptTokens: 100,\n * completionTokens: 50,\n * totalTokens: 150\n * });\n * // Returns: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }\n * ```\n */\nexport function buildBaseUsage<TProviderDetails = ProviderUsageDetails>(\n input: BaseUsageInput,\n): TokenUsage<TProviderDetails> {\n return {\n promptTokens: input.promptTokens,\n completionTokens: input.completionTokens,\n totalTokens: input.totalTokens,\n }\n}\n"],"names":[],"mappings":"AAgCO,SAAS,eACd,OAC8B;AAC9B,SAAO;AAAA,IACL,cAAc,MAAM;AAAA,IACpB,kBAAkB,MAAM;AAAA,IACxB,aAAa,MAAM;AAAA,EAAA;AAEvB;"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.24.0",
3
+ "version": "0.26.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -68,7 +68,7 @@
68
68
  "@ag-ui/core": "^0.0.52",
69
69
  "@standard-schema/spec": "^1.1.0",
70
70
  "partial-json": "^0.1.7",
71
- "@tanstack/ai-event-client": "0.4.3"
71
+ "@tanstack/ai-event-client": "0.5.1"
72
72
  },
73
73
  "peerDependencies": {
74
74
  "@opentelemetry/api": ">=1.9.0"
@@ -2,9 +2,11 @@
2
2
  name: ai-core/adapter-configuration
3
3
  description: >
4
4
  Provider adapter selection and configuration: openaiText, anthropicText,
5
- geminiText, ollamaText, grokText, groqText, openRouterText. Per-model
5
+ geminiText, ollamaText, grokText, groqText, openRouterText, openaiCompatible. Per-model
6
6
  type safety with modelOptions, reasoning/thinking configuration,
7
7
  runtime adapter switching, extendAdapter() for custom models, createModel().
8
+ Generic OpenAI-compatible providers (DeepSeek, Together, Fireworks, etc.) via
9
+ openaiCompatible({ baseURL, apiKey, models }) from @tanstack/ai-openai/compatible.
8
10
  API key env vars: OPENAI_API_KEY, ANTHROPIC_API_KEY, GOOGLE_API_KEY/GEMINI_API_KEY,
9
11
  XAI_API_KEY, GROQ_API_KEY, OPENROUTER_API_KEY, OLLAMA_HOST.
10
12
  type: sub-skill
@@ -60,15 +62,16 @@ into the factory, not into `chat()`.
60
62
  Each provider has a dedicated package with tree-shakeable adapter factories.
61
63
  The text adapter is the primary one for chat/completions:
62
64
 
63
- | Provider | Package | Factory | Env Var |
64
- | ---------- | ------------------------- | ---------------- | ------------------------------------------------- |
65
- | OpenAI | `@tanstack/ai-openai` | `openaiText` | `OPENAI_API_KEY` |
66
- | Anthropic | `@tanstack/ai-anthropic` | `anthropicText` | `ANTHROPIC_API_KEY` |
67
- | Gemini | `@tanstack/ai-gemini` | `geminiText` | `GOOGLE_API_KEY` or `GEMINI_API_KEY` |
68
- | Grok (xAI) | `@tanstack/ai-grok` | `grokText` | `XAI_API_KEY` |
69
- | Groq | `@tanstack/ai-groq` | `groqText` | `GROQ_API_KEY` |
70
- | OpenRouter | `@tanstack/ai-openrouter` | `openRouterText` | `OPENROUTER_API_KEY` |
71
- | Ollama | `@tanstack/ai-ollama` | `ollamaText` | `OLLAMA_HOST` (default: `http://localhost:11434`) |
65
+ | Provider | Package | Factory | Env Var |
66
+ | ----------------- | -------------------------------- | ------------------------------------------- | ------------------------------------------------- |
67
+ | OpenAI | `@tanstack/ai-openai` | `openaiText` | `OPENAI_API_KEY` |
68
+ | Anthropic | `@tanstack/ai-anthropic` | `anthropicText` | `ANTHROPIC_API_KEY` |
69
+ | Gemini | `@tanstack/ai-gemini` | `geminiText` | `GOOGLE_API_KEY` or `GEMINI_API_KEY` |
70
+ | Grok (xAI) | `@tanstack/ai-grok` | `grokText` | `XAI_API_KEY` |
71
+ | Groq | `@tanstack/ai-groq` | `groqText` | `GROQ_API_KEY` |
72
+ | OpenRouter | `@tanstack/ai-openrouter` | `openRouterText` | `OPENROUTER_API_KEY` |
73
+ | Ollama | `@tanstack/ai-ollama` | `ollamaText` | `OLLAMA_HOST` (default: `http://localhost:11434`) |
74
+ | OpenAI-compatible | `@tanstack/ai-openai/compatible` | `openaiCompatible` / `openaiCompatibleText` | provider-specific (passed via `apiKey`) |
72
75
 
73
76
  ```typescript
74
77
  // Each factory takes model as first arg, optional config as second
@@ -252,6 +255,63 @@ Subclasses can override to narrow the capability. When extending an
252
255
  adapter for a custom model that doesn't support the combination, return
253
256
  `false` explicitly.
254
257
 
258
+ ### 6. OpenAI-Compatible Providers
259
+
260
+ Any provider that implements the OpenAI **Chat Completions** API (DeepSeek,
261
+ Moonshot/Kimi, Together, Fireworks, Cerebras, Qwen/DashScope, Perplexity,
262
+ NVIDIA NIM, LM Studio, etc.) can be used through the generic
263
+ `openaiCompatible` factory from `@tanstack/ai-openai/compatible` — no
264
+ dedicated package required.
265
+
266
+ ```typescript
267
+ import { openaiCompatible } from '@tanstack/ai-openai/compatible'
268
+ import { createModel } from '@tanstack/ai'
269
+
270
+ // Provider-factory: configure baseURL + apiKey + models ONCE,
271
+ // then select a model per call (the model arg is a type-safe union).
272
+ const deepseek = openaiCompatible({
273
+ name: 'deepseek', // optional label for devtools/errors (default 'openai-compatible')
274
+ baseURL: 'https://api.deepseek.com/v1',
275
+ apiKey: process.env.DEEPSEEK_API_KEY!,
276
+ models: [
277
+ 'deepseek-chat', // bare string → optimistic defaults: text/image in, streaming, tools, structured output
278
+ createModel('deepseek-reasoner', {
279
+ // rich def → precise per-model capabilities
280
+ input: ['text'],
281
+ features: ['reasoning', 'structured_outputs'],
282
+ }),
283
+ ],
284
+ })
285
+
286
+ chat({ adapter: deepseek('deepseek-chat'), messages })
287
+ chat({ adapter: deepseek('deepseek-reasoner'), messages })
288
+ ```
289
+
290
+ `config` also accepts any OpenAI SDK `ClientOptions` (notably `defaultHeaders`
291
+ and `defaultQuery`) for providers that need extra auth headers or query params.
292
+
293
+ For a single model, use the one-shot helper:
294
+
295
+ ```typescript
296
+ import { openaiCompatibleText } from '@tanstack/ai-openai/compatible'
297
+
298
+ chat({
299
+ adapter: openaiCompatibleText('deepseek-chat', {
300
+ baseURL: 'https://api.deepseek.com/v1',
301
+ apiKey: process.env.DEEPSEEK_API_KEY!,
302
+ }),
303
+ messages,
304
+ })
305
+ ```
306
+
307
+ Pass `api: 'responses'` to target the OpenAI **Responses** API instead of Chat
308
+ Completions (only for the rare compatible provider that implements it, e.g.
309
+ Azure OpenAI); the default is `'chat-completions'`, which is what nearly all
310
+ compatible providers speak.
311
+
312
+ > Verify the provider's current `baseURL` and model ids against its live docs —
313
+ > they drift. See `docs/adapters/openai-compatible.md` for the full provider table.
314
+
255
315
  ## Common Mistakes
256
316
 
257
317
  ### a. HIGH: Confusing legacy monolithic with tree-shakeable adapter
@@ -4,6 +4,7 @@ import type {
4
4
  Modality,
5
5
  StreamChunk,
6
6
  TextOptions,
7
+ TokenUsage,
7
8
  } from '../../types'
8
9
 
9
10
  /**
@@ -40,6 +41,8 @@ export interface StructuredOutputResult<T = unknown> {
40
41
  data: T
41
42
  /** The raw text response from the model before parsing */
42
43
  rawText: string
44
+ /** Token usage information (if provided by the adapter) */
45
+ usage?: TokenUsage
43
46
  }
44
47
 
45
48
  /**
@@ -2,9 +2,9 @@ import type {
2
2
  JSONSchema,
3
3
  ModelMessage,
4
4
  StreamChunk,
5
+ TokenUsage,
5
6
  Tool,
6
7
  ToolCall,
7
- UsageTotals,
8
8
  } from '../../../types'
9
9
  import type { SystemPrompt } from '../../../system-prompts'
10
10
 
@@ -266,11 +266,11 @@ export interface ToolPhaseCompleteInfo {
266
266
  * Token usage statistics passed to the onUsage hook.
267
267
  * Extracted from the RUN_FINISHED chunk when usage data is present.
268
268
  *
269
- * Includes optional provider-reported `cost`/`costDetails` (see {@link UsageTotals}).
270
- * Kept as an interface extending `UsageTotals` to preserve declaration merging for
269
+ * Includes optional provider-reported `cost`/`costDetails` (see {@link TokenUsage}).
270
+ * Kept as an interface extending `TokenUsage` to preserve declaration merging for
271
271
  * this publicly exported type.
272
272
  */
273
- export interface UsageInfo extends UsageTotals {}
273
+ export interface UsageInfo extends TokenUsage {}
274
274
 
275
275
  // ===========================
276
276
  // Terminal Hook Info
@@ -287,7 +287,7 @@ export interface FinishInfo {
287
287
  /** Final accumulated text content */
288
288
  content: string
289
289
  /** Final usage totals, if available (optionally including provider-reported cost) */
290
- usage?: UsageTotals | undefined
290
+ usage?: TokenUsage | undefined
291
291
  }
292
292
 
293
293
  /**
@@ -174,6 +174,16 @@ async function runGenerateAudio<
174
174
  timestamp: Date.now(),
175
175
  })
176
176
 
177
+ if (result.usage) {
178
+ aiEventClient.emit('audio:usage', {
179
+ requestId,
180
+ model,
181
+ usage: result.usage,
182
+ modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
183
+ timestamp: Date.now(),
184
+ })
185
+ }
186
+
177
187
  logger.output(`activity=generateAudio provider=${providerName}`, {
178
188
  contentType: result.audio.contentType,
179
189
  audioDuration: result.audio.duration,
@@ -187,6 +187,16 @@ async function runGenerateSpeech<
187
187
  timestamp: Date.now(),
188
188
  })
189
189
 
190
+ if (result.usage) {
191
+ aiEventClient.emit('speech:usage', {
192
+ requestId,
193
+ model,
194
+ usage: result.usage,
195
+ modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
196
+ timestamp: Date.now(),
197
+ })
198
+ }
199
+
190
200
  logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {
191
201
  bytes: result.audio.length,
192
202
  contentType: result.contentType,
@@ -22,6 +22,8 @@ export interface ExtendedModelDef<
22
22
  TName extends string = string,
23
23
  TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>,
24
24
  TOptions = unknown,
25
+ TFeatures extends ReadonlyArray<string> = ReadonlyArray<string>,
26
+ TTools extends ReadonlyArray<string> = ReadonlyArray<string>,
25
27
  > {
26
28
  /** The model name identifier */
27
29
  name: TName
@@ -29,6 +31,23 @@ export interface ExtendedModelDef<
29
31
  input: TInput
30
32
  /** Type brand for provider options - use `{} as YourOptionsType` */
31
33
  modelOptions: TOptions
34
+ /** Optional declared features (e.g. 'reasoning', 'structured_outputs') */
35
+ features?: TFeatures
36
+ /** Optional declared provider tools (e.g. 'web_search') */
37
+ tools?: TTools
38
+ }
39
+
40
+ /** Capability bag accepted by the object form of `createModel`. */
41
+ export interface ModelCapabilities<
42
+ TInput extends ReadonlyArray<Modality> = ReadonlyArray<Modality>,
43
+ TFeatures extends ReadonlyArray<string> = ReadonlyArray<string>,
44
+ TTools extends ReadonlyArray<string> = ReadonlyArray<string>,
45
+ TOptions = unknown,
46
+ > {
47
+ input?: TInput
48
+ features?: TFeatures
49
+ tools?: TTools
50
+ modelOptions?: TOptions
32
51
  }
33
52
 
34
53
  /**
@@ -57,15 +76,57 @@ export interface ExtendedModelDef<
57
76
  *
58
77
  * const myOpenai = extendAdapter(openaiText, customModels)
59
78
  * ```
79
+ *
80
+ * @example
81
+ * ```typescript
82
+ * // Capabilities object form - declare features and provider tools
83
+ * const reasoner = createModel('reasoner', {
84
+ * input: ['text'],
85
+ * features: ['reasoning', 'structured_outputs'],
86
+ * tools: ['web_search'],
87
+ * })
88
+ * ```
60
89
  */
90
+ // Overload 1 — legacy positional input array (unchanged behavior)
61
91
  export function createModel<
62
92
  const TName extends string,
63
93
  const TInput extends ReadonlyArray<Modality>,
64
- >(name: TName, input: TInput): ExtendedModelDef<TName, TInput> {
94
+ >(name: TName, input: TInput): ExtendedModelDef<TName, TInput>
95
+ // Overload 2 — capabilities object
96
+ export function createModel<
97
+ const TName extends string,
98
+ const TCaps extends ModelCapabilities,
99
+ >(
100
+ name: TName,
101
+ capabilities: TCaps,
102
+ ): ExtendedModelDef<
103
+ TName,
104
+ TCaps['input'] extends ReadonlyArray<Modality>
105
+ ? TCaps['input']
106
+ : ReadonlyArray<Modality>,
107
+ TCaps['modelOptions'],
108
+ TCaps['features'] extends ReadonlyArray<string>
109
+ ? TCaps['features']
110
+ : ReadonlyArray<string>,
111
+ TCaps['tools'] extends ReadonlyArray<string>
112
+ ? TCaps['tools']
113
+ : ReadonlyArray<string>
114
+ >
115
+ // Implementation
116
+ export function createModel(
117
+ name: string,
118
+ second: ReadonlyArray<Modality> | ModelCapabilities,
119
+ ): ExtendedModelDef {
120
+ if (Array.isArray(second)) {
121
+ return { name, input: second, modelOptions: {} }
122
+ }
123
+ const caps = second as ModelCapabilities
65
124
  return {
66
125
  name,
67
- input,
68
- modelOptions: {},
126
+ input: caps.input ?? (['text'] as ReadonlyArray<Modality>),
127
+ modelOptions: caps.modelOptions ?? {},
128
+ features: caps.features,
129
+ tools: caps.tools,
69
130
  }
70
131
  }
71
132
 
package/src/index.ts CHANGED
@@ -111,6 +111,9 @@ export type {
111
111
  // All types
112
112
  export * from './types'
113
113
 
114
+ // Usage utilities
115
+ export { buildBaseUsage, type BaseUsageInput } from './utilities/usage'
116
+
114
117
  // System prompts (type + normaliser used by adapters)
115
118
  export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts'
116
119
  export { normalizeSystemPrompts } from './system-prompts'
@@ -197,7 +200,7 @@ export {
197
200
 
198
201
  // Adapter extension utilities
199
202
  export { createModel, extendAdapter } from './extend-adapter'
200
- export type { ExtendedModelDef } from './extend-adapter'
203
+ export type { ExtendedModelDef, ModelCapabilities } from './extend-adapter'
201
204
 
202
205
  // Logger
203
206
  export type {
package/src/types.ts CHANGED
@@ -4,6 +4,16 @@ import type {
4
4
  } from '@standard-schema/spec'
5
5
  import type { InternalLogger } from './logger/internal-logger'
6
6
  import type { SystemPrompt } from './system-prompts'
7
+ // The canonical usage types live in the leaf `@tanstack/ai-event-client`
8
+ // package (which `@tanstack/ai` already depends on) so there is a single source
9
+ // of truth without a dependency cycle. They are re-exported below.
10
+ import type {
11
+ CompletionTokensDetails,
12
+ PromptTokensDetails,
13
+ ProviderUsageDetails,
14
+ TokenUsage,
15
+ UsageCostBreakdown,
16
+ } from '@tanstack/ai-event-client'
7
17
  import type {
8
18
  BaseEvent as AGUIBaseEvent,
9
19
  CustomEvent as AGUICustomEvent,
@@ -978,38 +988,22 @@ export interface RunStartedEvent extends AGUIRunStartedEvent {
978
988
  model?: string
979
989
  }
980
990
 
981
- /**
982
- * Provider-reported cost breakdown for a single request, normalized onto a
983
- * canonical shape so consumer code is portable across gateways. Each adapter's
984
- * extractor maps its provider-specific wire keys (e.g. OpenRouter's
985
- * `upstream_inference_prompt_cost`, `upstream_inference_input_cost`) onto these
986
- * fields at runtime.
987
- */
988
- export interface UsageCostBreakdown {
989
- /** Total cost the gateway paid the upstream provider. */
990
- upstreamCost?: number
991
- /** Upstream cost for input (prompt) tokens. */
992
- upstreamInputCost?: number
993
- /** Upstream cost for output (completion) tokens. */
994
- upstreamOutputCost?: number
991
+ // Re-export the canonical usage types (defined in `@tanstack/ai-event-client`)
992
+ // so `@tanstack/ai` consumers keep importing them from here unchanged.
993
+ export type {
994
+ CompletionTokensDetails,
995
+ PromptTokensDetails,
996
+ ProviderUsageDetails,
997
+ TokenUsage,
998
+ UsageCostBreakdown,
995
999
  }
996
1000
 
997
1001
  /**
998
- * Token usage totals for a run, optionally including provider-reported cost.
999
- *
1000
- * `cost` and `costDetails` are populated only by adapters whose provider returns
1001
- * authoritative per-request cost (e.g. OpenRouter). They are absent for adapters
1002
- * that do not report cost, so consumers must treat them as optional.
1002
+ * @deprecated Renamed to {@link TokenUsage}. Kept as an alias for backward
1003
+ * compatibility with `@tanstack/ai@0.23` and earlier; will be removed in a
1004
+ * future release.
1003
1005
  */
1004
- export interface UsageTotals {
1005
- promptTokens: number
1006
- completionTokens: number
1007
- totalTokens: number
1008
- /** Provider-reported cost for the request, when available. */
1009
- cost?: number
1010
- /** Provider-reported cost breakdown, when available. */
1011
- costDetails?: UsageCostBreakdown
1012
- }
1006
+ export type UsageTotals = TokenUsage
1013
1007
 
1014
1008
  /**
1015
1009
  * Emitted when a run completes successfully.
@@ -1022,8 +1016,8 @@ export interface RunFinishedEvent extends AGUIRunFinishedEvent {
1022
1016
  model?: string
1023
1017
  /** Why the generation stopped */
1024
1018
  finishReason?: 'stop' | 'length' | 'content_filter' | 'tool_calls' | null
1025
- /** Token usage statistics, optionally including provider-reported cost. */
1026
- usage?: UsageTotals
1019
+ /** Token usage statistics with optional detailed breakdowns and provider-reported cost. */
1020
+ usage?: TokenUsage
1027
1021
  }
1028
1022
 
1029
1023
  /**
@@ -1473,11 +1467,7 @@ export interface TextCompletionChunk {
1473
1467
  content: string
1474
1468
  role?: 'assistant'
1475
1469
  finishReason?: 'stop' | 'length' | 'content_filter' | null
1476
- usage?: {
1477
- promptTokens: number
1478
- completionTokens: number
1479
- totalTokens: number
1480
- }
1470
+ usage?: TokenUsage
1481
1471
  }
1482
1472
 
1483
1473
  export interface SummarizationOptions<
@@ -1501,11 +1491,7 @@ export interface SummarizationResult {
1501
1491
  id: string
1502
1492
  model: string
1503
1493
  summary: string
1504
- usage: {
1505
- promptTokens: number
1506
- completionTokens: number
1507
- totalTokens: number
1508
- }
1494
+ usage: TokenUsage
1509
1495
  }
1510
1496
 
1511
1497
  // ============================================================================
@@ -1574,11 +1560,7 @@ export interface ImageGenerationResult {
1574
1560
  /** Array of generated images */
1575
1561
  images: Array<GeneratedImage>
1576
1562
  /** Token usage information (if available) */
1577
- usage?: {
1578
- inputTokens?: number
1579
- outputTokens?: number
1580
- totalTokens?: number
1581
- }
1563
+ usage?: TokenUsage
1582
1564
  }
1583
1565
 
1584
1566
  // ============================================================================
@@ -1629,11 +1611,7 @@ export interface AudioGenerationResult {
1629
1611
  /** The generated audio */
1630
1612
  audio: GeneratedAudio
1631
1613
  /** Token usage information (if available) */
1632
- usage?: {
1633
- inputTokens?: number
1634
- outputTokens?: number
1635
- totalTokens?: number
1636
- }
1614
+ usage?: TokenUsage
1637
1615
  }
1638
1616
 
1639
1617
  // ============================================================================
@@ -1754,6 +1732,8 @@ export interface TTSResult {
1754
1732
  duration?: number
1755
1733
  /** Content type of the audio (e.g., 'audio/mp3') */
1756
1734
  contentType?: string
1735
+ /** Token usage information (if provided by the adapter) */
1736
+ usage?: TokenUsage
1757
1737
  }
1758
1738
 
1759
1739
  // ============================================================================
@@ -1835,6 +1815,8 @@ export interface TranscriptionResult {
1835
1815
  segments?: Array<TranscriptionSegment>
1836
1816
  /** Word-level timestamps, if available */
1837
1817
  words?: Array<TranscriptionWord>
1818
+ /** Token usage information (if provided by the adapter) */
1819
+ usage?: TokenUsage
1838
1820
  }
1839
1821
 
1840
1822
  /**
@@ -0,0 +1,41 @@
1
+ import type { ProviderUsageDetails, TokenUsage } from '../types'
2
+
3
+ /**
4
+ * Input parameters for building base TokenUsage.
5
+ * Provider functions should extract these from their SDK's response.
6
+ */
7
+ export interface BaseUsageInput {
8
+ /** Total input/prompt tokens */
9
+ promptTokens: number
10
+ /** Total output/completion tokens */
11
+ completionTokens: number
12
+ /** Total tokens (prompt + completion) */
13
+ totalTokens: number
14
+ }
15
+
16
+ /**
17
+ * Builds the base TokenUsage object with core fields.
18
+ * Provider-specific functions should use this and then add their own details.
19
+ *
20
+ * @param input - The base token counts
21
+ * @returns A TokenUsage object with promptTokens, completionTokens, totalTokens
22
+ *
23
+ * @example
24
+ * ```typescript
25
+ * const base = buildBaseUsage({
26
+ * promptTokens: 100,
27
+ * completionTokens: 50,
28
+ * totalTokens: 150
29
+ * });
30
+ * // Returns: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
31
+ * ```
32
+ */
33
+ export function buildBaseUsage<TProviderDetails = ProviderUsageDetails>(
34
+ input: BaseUsageInput,
35
+ ): TokenUsage<TProviderDetails> {
36
+ return {
37
+ promptTokens: input.promptTokens,
38
+ completionTokens: input.completionTokens,
39
+ totalTokens: input.totalTokens,
40
+ }
41
+ }