@tanstack/ai 0.24.0 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +3 -1
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/middleware/types.d.ts +5 -5
- package/dist/esm/activities/generateAudio/index.js +9 -0
- package/dist/esm/activities/generateAudio/index.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.js +9 -0
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/index.d.ts +1 -0
- package/dist/esm/index.js +2 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/types.d.ts +16 -51
- package/dist/esm/utilities/usage.d.ts +31 -0
- package/dist/esm/utilities/usage.js +11 -0
- package/dist/esm/utilities/usage.js.map +1 -0
- package/package.json +2 -2
- package/src/activities/chat/adapter.ts +3 -0
- package/src/activities/chat/middleware/types.ts +5 -5
- package/src/activities/generateAudio/index.ts +10 -0
- package/src/activities/generateSpeech/index.ts +10 -0
- package/src/index.ts +3 -0
- package/src/types.ts +32 -50
- package/src/utilities/usage.ts +41 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { DefaultMessageMetadataByModality, JSONSchema, Modality, StreamChunk, TextOptions } from '../../types.js';
|
|
1
|
+
import { DefaultMessageMetadataByModality, JSONSchema, Modality, StreamChunk, TextOptions, TokenUsage } from '../../types.js';
|
|
2
2
|
/**
|
|
3
3
|
* Configuration for adapter instances
|
|
4
4
|
*/
|
|
@@ -31,6 +31,8 @@ export interface StructuredOutputResult<T = unknown> {
|
|
|
31
31
|
data: T;
|
|
32
32
|
/** The raw text response from the model before parsing */
|
|
33
33
|
rawText: string;
|
|
34
|
+
/** Token usage information (if provided by the adapter) */
|
|
35
|
+
usage?: TokenUsage;
|
|
34
36
|
}
|
|
35
37
|
/**
|
|
36
38
|
* Text adapter interface with pre-resolved generics.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Declares whether the adapter supports combining `tools` and a\n * schema-constrained final answer in a single streaming request.\n *\n * When `true`, the engine wires `outputSchema` into the regular\n * `chatStream()` call and skips the separate `runStructuredFinalization`\n * round-trip. The model's natural final turn carries the\n * schema-constrained JSON text and the engine harvests it from the agent\n * loop's accumulated content.\n *\n * When `false`, `undefined`, or the method is omitted, the engine runs\n * the agent loop without `outputSchema` and then issues a separate\n * `structuredOutput` / `structuredOutputStream` call against the JSON\n * schema for finalization (the legacy path).\n *\n * The method receives the per-call `modelOptions` so providers whose\n * support depends on the resolved upstream model (e.g. OpenRouter) can\n * answer per-request. Most adapters can return a constant.\n */\n supportsCombinedToolsAndSchema?: (\n modelOptions?: TProviderOptions | undefined,\n ) => boolean\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"adapter.js","sources":["../../../../src/activities/chat/adapter.ts"],"sourcesContent":["import type {\n DefaultMessageMetadataByModality,\n JSONSchema,\n Modality,\n StreamChunk,\n TextOptions,\n TokenUsage,\n} from '../../types'\n\n/**\n * Configuration for adapter instances\n */\nexport interface TextAdapterConfig {\n apiKey?: string\n baseUrl?: string\n timeout?: number\n maxRetries?: number\n headers?: Record<string, string>\n}\n\n/**\n * Options for structured output generation.\n *\n * The internal logger is threaded through `chatOptions.logger` (inherited from\n * `TextOptions`). Adapter implementations must call `logger.request()` before\n * SDK calls, `logger.provider()` for each chunk received, and `logger.errors()`\n * in catch blocks.\n */\nexport interface StructuredOutputOptions<TProviderOptions extends object> {\n /** Text options for the request */\n chatOptions: TextOptions<TProviderOptions>\n /** JSON Schema for structured output - already converted from Zod in the ai layer */\n outputSchema: JSONSchema\n}\n\n/**\n * Result from structured output generation\n */\nexport interface StructuredOutputResult<T = unknown> {\n /** The parsed data conforming to the schema */\n data: T\n /** The raw text response from the model before parsing */\n rawText: string\n /** Token usage information (if provided by the adapter) */\n usage?: TokenUsage\n}\n\n/**\n * Text adapter interface with pre-resolved generics.\n *\n * An adapter is created by a provider function: `provider('model')` → `adapter`\n * All type resolution happens at the provider call site, not in this interface.\n *\n * Generic parameters:\n * - TModel: The specific model name (e.g., 'gpt-4o')\n * - TProviderOptions: Provider-specific options for this model (already resolved)\n * - TInputModalities: Supported input modalities for this model (already resolved)\n * - TMessageMetadata: Metadata types for content parts (already resolved)\n * - TToolCapabilities: Tuple of tool-kind strings supported by this model, resolved from `supports.tools`\n * - TToolCallMetadata: Metadata type that round-trips with tool calls (e.g. Gemini's `thoughtSignature`)\n * - TSystemPromptMetadata: Provider-typed metadata accepted on each\n * `systemPrompts[i]` entry (e.g. Anthropic `cache_control`). Defaults to\n * `never` — adapters without per-prompt metadata reject the `metadata`\n * field at the call site.\n */\nexport interface TextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> {\n /** Discriminator for adapter kind */\n readonly kind: 'text'\n /** Provider name identifier (e.g., 'openai', 'anthropic') */\n readonly name: string\n /** The model this adapter is configured for */\n readonly model: TModel\n\n /**\n * @internal Type-only properties for inference. Not assigned at runtime.\n */\n '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n /**\n * Stream text completions from the model\n */\n chatStream: (\n options: TextOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * This method uses stream: false and sends the JSON schema to the provider\n * to ensure the response conforms to the expected structure.\n *\n * @param options - Structured output options containing chat options and JSON schema\n * @returns Promise with the raw data (validation is done in the chat function)\n */\n structuredOutput: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => Promise<StructuredOutputResult<unknown>>\n\n /**\n * Stream structured output using the provider's native streaming structured\n * output API (stream + response_format json_schema in a single request).\n *\n * Optional — adapters without native streaming JSON omit this method and the\n * activity layer synthesizes a stream around the non-streaming\n * `structuredOutput` call.\n *\n * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,\n * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final\n * `CUSTOM` event named `structured-output.complete` whose `value` is\n * `{ object, raw, reasoning? }`.\n */\n structuredOutputStream?: (\n options: StructuredOutputOptions<TProviderOptions>,\n ) => AsyncIterable<StreamChunk>\n\n /**\n * Declares whether the adapter supports combining `tools` and a\n * schema-constrained final answer in a single streaming request.\n *\n * When `true`, the engine wires `outputSchema` into the regular\n * `chatStream()` call and skips the separate `runStructuredFinalization`\n * round-trip. The model's natural final turn carries the\n * schema-constrained JSON text and the engine harvests it from the agent\n * loop's accumulated content.\n *\n * When `false`, `undefined`, or the method is omitted, the engine runs\n * the agent loop without `outputSchema` and then issues a separate\n * `structuredOutput` / `structuredOutputStream` call against the JSON\n * schema for finalization (the legacy path).\n *\n * The method receives the per-call `modelOptions` so providers whose\n * support depends on the resolved upstream model (e.g. OpenRouter) can\n * answer per-request. Most adapters can return a constant.\n */\n supportsCombinedToolsAndSchema?: (\n modelOptions?: TProviderOptions | undefined,\n ) => boolean\n}\n\n/**\n * A TextAdapter with any/unknown type parameters.\n * Useful as a constraint in generic functions and interfaces.\n */\nexport type AnyTextAdapter = TextAdapter<any, any, any, any, any, any, any>\n\n/**\n * Abstract base class for text adapters.\n * Extend this class to implement a text adapter for a specific provider.\n *\n * Generic parameters match TextAdapter - all pre-resolved by the provider function.\n */\nexport abstract class BaseTextAdapter<\n TModel extends string,\n TProviderOptions extends Record<string, any>,\n TInputModalities extends ReadonlyArray<Modality>,\n TMessageMetadataByModality extends DefaultMessageMetadataByModality,\n TToolCapabilities extends ReadonlyArray<string> = ReadonlyArray<string>,\n TToolCallMetadata = unknown,\n TSystemPromptMetadata = never,\n> implements TextAdapter<\n TModel,\n TProviderOptions,\n TInputModalities,\n TMessageMetadataByModality,\n TToolCapabilities,\n TToolCallMetadata,\n TSystemPromptMetadata\n> {\n readonly kind = 'text' as const\n abstract readonly name: string\n readonly model: TModel\n\n // Type-only property - never assigned at runtime\n declare '~types': {\n providerOptions: TProviderOptions\n inputModalities: TInputModalities\n messageMetadataByModality: TMessageMetadataByModality\n toolCapabilities: TToolCapabilities\n toolCallMetadata: TToolCallMetadata\n systemPromptMetadata: TSystemPromptMetadata\n }\n\n protected config: TextAdapterConfig\n\n constructor(config: TextAdapterConfig = {}, model: TModel) {\n this.config = config\n this.model = model\n }\n\n abstract chatStream(\n options: TextOptions<TProviderOptions>,\n ): AsyncIterable<StreamChunk>\n\n /**\n * Generate structured output using the provider's native structured output API.\n * Concrete implementations should override this to use provider-specific structured output.\n */\n abstract structuredOutput(\n options: StructuredOutputOptions<TProviderOptions>,\n ): Promise<StructuredOutputResult<unknown>>\n\n protected generateId(): string {\n return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`\n }\n}\n"],"names":[],"mappings":"AAqKO,MAAe,gBAgBpB;AAAA,EACS,OAAO;AAAA,EAEP;AAAA,EAYC;AAAA,EAEV,YAAY,SAA4B,CAAA,GAAI,OAAe;AACzD,SAAK,SAAS;AACd,SAAK,QAAQ;AAAA,EACf;AAAA,EAcU,aAAqB;AAC7B,WAAO,GAAG,KAAK,IAAI,IAAI,KAAK,KAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,UAAU,CAAC,CAAC;AAAA,EAC9E;AACF;"}
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { JSONSchema, ModelMessage, StreamChunk, Tool, ToolCall
|
|
1
|
+
import { JSONSchema, ModelMessage, StreamChunk, TokenUsage, Tool, ToolCall } from '../../../types.js';
|
|
2
2
|
import { SystemPrompt } from '../../../system-prompts.js';
|
|
3
3
|
/**
|
|
4
4
|
* Phase of the chat middleware lifecycle.
|
|
@@ -204,11 +204,11 @@ export interface ToolPhaseCompleteInfo {
|
|
|
204
204
|
* Token usage statistics passed to the onUsage hook.
|
|
205
205
|
* Extracted from the RUN_FINISHED chunk when usage data is present.
|
|
206
206
|
*
|
|
207
|
-
* Includes optional provider-reported `cost`/`costDetails` (see {@link
|
|
208
|
-
* Kept as an interface extending `
|
|
207
|
+
* Includes optional provider-reported `cost`/`costDetails` (see {@link TokenUsage}).
|
|
208
|
+
* Kept as an interface extending `TokenUsage` to preserve declaration merging for
|
|
209
209
|
* this publicly exported type.
|
|
210
210
|
*/
|
|
211
|
-
export interface UsageInfo extends
|
|
211
|
+
export interface UsageInfo extends TokenUsage {
|
|
212
212
|
}
|
|
213
213
|
/**
|
|
214
214
|
* Information passed to onFinish.
|
|
@@ -221,7 +221,7 @@ export interface FinishInfo {
|
|
|
221
221
|
/** Final accumulated text content */
|
|
222
222
|
content: string;
|
|
223
223
|
/** Final usage totals, if available (optionally including provider-reported cost) */
|
|
224
|
-
usage?:
|
|
224
|
+
usage?: TokenUsage | undefined;
|
|
225
225
|
}
|
|
226
226
|
/**
|
|
227
227
|
* Information passed to onAbort.
|
|
@@ -45,6 +45,15 @@ async function runGenerateAudio(options) {
|
|
|
45
45
|
modelOptions: rest.modelOptions,
|
|
46
46
|
timestamp: Date.now()
|
|
47
47
|
});
|
|
48
|
+
if (result.usage) {
|
|
49
|
+
aiEventClient.emit("audio:usage", {
|
|
50
|
+
requestId,
|
|
51
|
+
model,
|
|
52
|
+
usage: result.usage,
|
|
53
|
+
modelOptions: rest.modelOptions,
|
|
54
|
+
timestamp: Date.now()
|
|
55
|
+
});
|
|
56
|
+
}
|
|
48
57
|
logger.output(`activity=generateAudio provider=${providerName}`, {
|
|
49
58
|
contentType: result.audio.contentType,
|
|
50
59
|
audioDuration: result.audio.duration
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateAudio/index.ts"],"sourcesContent":["/**\n * Audio Generation Activity\n *\n * Generates audio (music, sound effects, etc.) from text prompts.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { AudioAdapter } from './adapter'\nimport type { AudioGenerationResult, StreamChunk } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'audio' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from an AudioAdapter via ~types.\n */\nexport type AudioProviderOptions<TAdapter> = TAdapter extends {\n '~types': { providerOptions: infer P extends object }\n}\n ? P\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the audio generation activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The audio adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface AudioActivityOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The audio adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** Text description of the desired audio */\n prompt: string\n /** Desired duration in seconds */\n duration?: number\n /** Provider-specific options for audio generation */\n modelOptions?: AudioProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<AudioGenerationResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the audio generation activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<AudioGenerationResult>\n */\nexport type AudioActivityResult<TStream extends boolean = false> =\n TStream extends true\n ? AsyncIterable<StreamChunk>\n : Promise<AudioGenerationResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * Audio generation activity - generates audio from text prompts.\n *\n * Uses AI models to create music, sound effects, and other audio content.\n *\n * @example Generate music from a prompt\n * ```ts\n * import { generateAudio } from '@tanstack/ai'\n * import { falAudio } from '@tanstack/ai-fal'\n *\n * const result = await generateAudio({\n * adapter: falAudio('fal-ai/diffrhythm'),\n * prompt: 'An upbeat electronic track with synths',\n * duration: 10\n * })\n *\n * console.log(result.audio.url) // URL to generated audio\n * ```\n */\nexport function generateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateAudio(options),\n ) as AudioActivityResult<TStream>\n }\n return runGenerateAudio(options) as AudioActivityResult<TStream>\n}\n\n/**\n * Run the core audio generation logic (non-streaming).\n */\nasync function runGenerateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n>(\n options: AudioActivityOptions<TAdapter, boolean>,\n): Promise<AudioGenerationResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('audio')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('audio:request:started', {\n requestId,\n provider: adapter.name,\n model,\n prompt: rest.prompt,\n duration: rest.duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateAudio provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateAudio({ ...rest, model, logger })\n const elapsedMs = Date.now() - startTime\n\n aiEventClient.emit('audio:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateAudio provider=${providerName}`, {\n contentType: result.audio.contentType,\n audioDuration: result.audio.duration,\n })\n\n return result\n } catch (error) {\n const elapsedMs = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('audio:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateAudio activity failed', {\n error,\n source: 'generateAudio',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateAudio() function without executing.\n */\nexport function createAudioOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type {\n AudioAdapter,\n AudioAdapterConfig,\n AnyAudioAdapter,\n} from './adapter'\nexport { BaseAudioAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAoEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAyBO,SAAS,cAId,SAC8B;AAC9B,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,iBAAiB,OAAO;AAAA,IAAA;AAAA,EAE5B;AACA,SAAO,iBAAiB,OAAO;AACjC;AAKA,eAAe,iBAGb,SACgC;AAChC,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,OAAO;AAClC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,yBAAyB;AAAA,IAC1C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,UAAU,KAAK;AAAA,IACf,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,mCAAmC,YAAY,IAAI;AAAA,IAChE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,cAAc,EAAE,GAAG,MAAM,OAAO,QAAQ;AACrE,UAAM,YAAY,KAAK,IAAA,IAAQ;AAE/B,kBAAc,KAAK,2BAA2B;AAAA,MAC5C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,mCAAmC,YAAY,IAAI;AAAA,MAC/D,aAAa,OAAO,MAAM;AAAA,MAC1B,eAAe,OAAO,MAAM;AAAA,IAAA,CAC7B;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,YAAY,KAAK,IAAA,IAAQ;AAC/B,UAAM,MAAM;AACZ,kBAAc,KAAK,uBAAuB;AAAA,MACxC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,iCAAiC;AAAA,MAC7C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,mBAId,SACyC;AACzC,SAAO;AACT;"}
|
|
1
|
+
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateAudio/index.ts"],"sourcesContent":["/**\n * Audio Generation Activity\n *\n * Generates audio (music, sound effects, etc.) from text prompts.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { AudioAdapter } from './adapter'\nimport type { AudioGenerationResult, StreamChunk } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'audio' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from an AudioAdapter via ~types.\n */\nexport type AudioProviderOptions<TAdapter> = TAdapter extends {\n '~types': { providerOptions: infer P extends object }\n}\n ? P\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the audio generation activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The audio adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface AudioActivityOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The audio adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** Text description of the desired audio */\n prompt: string\n /** Desired duration in seconds */\n duration?: number\n /** Provider-specific options for audio generation */\n modelOptions?: AudioProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<AudioGenerationResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the audio generation activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<AudioGenerationResult>\n */\nexport type AudioActivityResult<TStream extends boolean = false> =\n TStream extends true\n ? AsyncIterable<StreamChunk>\n : Promise<AudioGenerationResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * Audio generation activity - generates audio from text prompts.\n *\n * Uses AI models to create music, sound effects, and other audio content.\n *\n * @example Generate music from a prompt\n * ```ts\n * import { generateAudio } from '@tanstack/ai'\n * import { falAudio } from '@tanstack/ai-fal'\n *\n * const result = await generateAudio({\n * adapter: falAudio('fal-ai/diffrhythm'),\n * prompt: 'An upbeat electronic track with synths',\n * duration: 10\n * })\n *\n * console.log(result.audio.url) // URL to generated audio\n * ```\n */\nexport function generateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateAudio(options),\n ) as AudioActivityResult<TStream>\n }\n return runGenerateAudio(options) as AudioActivityResult<TStream>\n}\n\n/**\n * Run the core audio generation logic (non-streaming).\n */\nasync function runGenerateAudio<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n>(\n options: AudioActivityOptions<TAdapter, boolean>,\n): Promise<AudioGenerationResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('audio')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('audio:request:started', {\n requestId,\n provider: adapter.name,\n model,\n prompt: rest.prompt,\n duration: rest.duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateAudio provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateAudio({ ...rest, model, logger })\n const elapsedMs = Date.now() - startTime\n\n aiEventClient.emit('audio:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n if (result.usage) {\n aiEventClient.emit('audio:usage', {\n requestId,\n model,\n usage: result.usage,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n }\n\n logger.output(`activity=generateAudio provider=${providerName}`, {\n contentType: result.audio.contentType,\n audioDuration: result.audio.duration,\n })\n\n return result\n } catch (error) {\n const elapsedMs = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('audio:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration: elapsedMs,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateAudio activity failed', {\n error,\n source: 'generateAudio',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateAudio() function without executing.\n */\nexport function createAudioOptions<\n TAdapter extends AudioAdapter<string, AudioProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: AudioActivityOptions<TAdapter, TStream>,\n): AudioActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type {\n AudioAdapter,\n AudioAdapterConfig,\n AnyAudioAdapter,\n} from './adapter'\nexport { BaseAudioAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAoEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAyBO,SAAS,cAId,SAC8B;AAC9B,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,iBAAiB,OAAO;AAAA,IAAA;AAAA,EAE5B;AACA,SAAO,iBAAiB,OAAO;AACjC;AAKA,eAAe,iBAGb,SACgC;AAChC,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,OAAO;AAClC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,yBAAyB;AAAA,IAC1C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,QAAQ,KAAK;AAAA,IACb,UAAU,KAAK;AAAA,IACf,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,mCAAmC,YAAY,IAAI;AAAA,IAChE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,cAAc,EAAE,GAAG,MAAM,OAAO,QAAQ;AACrE,UAAM,YAAY,KAAK,IAAA,IAAQ;AAE/B,kBAAc,KAAK,2BAA2B;AAAA,MAC5C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,QAAI,OAAO,OAAO;AAChB,oBAAc,KAAK,eAAe;AAAA,QAChC;AAAA,QACA;AAAA,QACA,OAAO,OAAO;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,WAAW,KAAK,IAAA;AAAA,MAAI,CACrB;AAAA,IACH;AAEA,WAAO,OAAO,mCAAmC,YAAY,IAAI;AAAA,MAC/D,aAAa,OAAO,MAAM;AAAA,MAC1B,eAAe,OAAO,MAAM;AAAA,IAAA,CAC7B;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,YAAY,KAAK,IAAA,IAAQ;AAC/B,UAAM,MAAM;AACZ,kBAAc,KAAK,uBAAuB;AAAA,MACxC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC,UAAU;AAAA,MACV,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,iCAAiC;AAAA,MAC7C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,mBAId,SACyC;AACzC,SAAO;AACT;"}
|
|
@@ -50,6 +50,15 @@ async function runGenerateSpeech(options) {
|
|
|
50
50
|
modelOptions: rest.modelOptions,
|
|
51
51
|
timestamp: Date.now()
|
|
52
52
|
});
|
|
53
|
+
if (result.usage) {
|
|
54
|
+
aiEventClient.emit("speech:usage", {
|
|
55
|
+
requestId,
|
|
56
|
+
model,
|
|
57
|
+
usage: result.usage,
|
|
58
|
+
modelOptions: rest.modelOptions,
|
|
59
|
+
timestamp: Date.now()
|
|
60
|
+
});
|
|
61
|
+
}
|
|
53
62
|
logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {
|
|
54
63
|
bytes: result.audio.length,
|
|
55
64
|
contentType: result.contentType
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiSpeech } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
|
|
1
|
+
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiSpeech } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n if (result.usage) {\n aiEventClient.emit('speech:usage', {\n requestId,\n model,\n usage: result.usage,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n }\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,QAAI,OAAO,OAAO;AAChB,oBAAc,KAAK,gBAAgB;AAAA,QACjC;AAAA,QACA;AAAA,QACA,OAAO,OAAO;AAAA,QACd,cAAc,KAAK;AAAA,QACnB,WAAW,KAAK,IAAA;AAAA,MAAI,CACrB;AAAA,IACH;AAEA,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
|
package/dist/esm/index.d.ts
CHANGED
|
@@ -17,6 +17,7 @@ export { maxIterations, untilFinishReason, combineStrategies, } from './activiti
|
|
|
17
17
|
export { createToolRegistry, createFrozenRegistry, type ToolRegistry, } from './tool-registry.js';
|
|
18
18
|
export type { ChatMiddleware, ChatMiddlewareContext, ChatMiddlewarePhase, ChatMiddlewareConfig, StructuredOutputMiddlewareConfig, ToolCallHookContext, BeforeToolCallDecision, AfterToolCallInfo, IterationInfo, ToolPhaseCompleteInfo, UsageInfo, FinishInfo, AbortInfo, ErrorInfo, } from './activities/chat/middleware/index.js';
|
|
19
19
|
export * from './types.js';
|
|
20
|
+
export { buildBaseUsage, type BaseUsageInput } from './utilities/usage.js';
|
|
20
21
|
export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts.js';
|
|
21
22
|
export { normalizeSystemPrompts } from './system-prompts.js';
|
|
22
23
|
export { detectImageMimeType } from './utils.js';
|
package/dist/esm/index.js
CHANGED
|
@@ -12,6 +12,7 @@ import { ToolCallManager } from "./activities/chat/tools/tool-calls.js";
|
|
|
12
12
|
import { brandProviderTool } from "./tools/provider-tool.js";
|
|
13
13
|
import { combineStrategies, maxIterations, untilFinishReason } from "./activities/chat/agent-loop-strategies.js";
|
|
14
14
|
import { createFrozenRegistry, createToolRegistry } from "./tool-registry.js";
|
|
15
|
+
import { buildBaseUsage } from "./utilities/usage.js";
|
|
15
16
|
import { normalizeSystemPrompts } from "./system-prompts.js";
|
|
16
17
|
import { detectImageMimeType } from "./utils.js";
|
|
17
18
|
import { realtimeToken } from "./realtime/index.js";
|
|
@@ -38,6 +39,7 @@ export {
|
|
|
38
39
|
ToolCallManager,
|
|
39
40
|
WordBoundaryStrategy,
|
|
40
41
|
brandProviderTool,
|
|
42
|
+
buildBaseUsage,
|
|
41
43
|
chat,
|
|
42
44
|
chatParamsFromRequest,
|
|
43
45
|
chatParamsFromRequestBody,
|
package/dist/esm/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;"}
|
package/dist/esm/types.d.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { StandardJSONSchemaV1, StandardSchemaV1 } from '@standard-schema/spec';
|
|
2
2
|
import { InternalLogger } from './logger/internal-logger.js';
|
|
3
3
|
import { SystemPrompt } from './system-prompts.js';
|
|
4
|
+
import { CompletionTokensDetails, PromptTokensDetails, ProviderUsageDetails, TokenUsage, UsageCostBreakdown } from '@tanstack/ai-event-client';
|
|
4
5
|
import { BaseEvent as AGUIBaseEvent, CustomEvent as AGUICustomEvent, MessagesSnapshotEvent as AGUIMessagesSnapshotEvent, ReasoningEncryptedValueEvent as AGUIReasoningEncryptedValueEvent, ReasoningEndEvent as AGUIReasoningEndEvent, ReasoningMessageContentEvent as AGUIReasoningMessageContentEvent, ReasoningMessageEndEvent as AGUIReasoningMessageEndEvent, ReasoningMessageStartEvent as AGUIReasoningMessageStartEvent, ReasoningStartEvent as AGUIReasoningStartEvent, RunErrorEvent as AGUIRunErrorEvent, RunFinishedEvent as AGUIRunFinishedEvent, RunStartedEvent as AGUIRunStartedEvent, StateDeltaEvent as AGUIStateDeltaEvent, StateSnapshotEvent as AGUIStateSnapshotEvent, StepFinishedEvent as AGUIStepFinishedEvent, StepStartedEvent as AGUIStepStartedEvent, TextMessageContentEvent as AGUITextMessageContentEvent, TextMessageEndEvent as AGUITextMessageEndEvent, TextMessageStartEvent as AGUITextMessageStartEvent, ToolCallArgsEvent as AGUIToolCallArgsEvent, ToolCallEndEvent as AGUIToolCallEndEvent, ToolCallResultEvent as AGUIToolCallResultEvent, ToolCallStartEvent as AGUIToolCallStartEvent, EventType } from '@ag-ui/core';
|
|
5
6
|
/**
|
|
6
7
|
* Tool call states - track the lifecycle of a tool call
|
|
@@ -787,37 +788,13 @@ export interface RunStartedEvent extends AGUIRunStartedEvent {
|
|
|
787
788
|
/** Model identifier for multi-model support */
|
|
788
789
|
model?: string;
|
|
789
790
|
}
|
|
791
|
+
export type { CompletionTokensDetails, PromptTokensDetails, ProviderUsageDetails, TokenUsage, UsageCostBreakdown, };
|
|
790
792
|
/**
|
|
791
|
-
*
|
|
792
|
-
*
|
|
793
|
-
*
|
|
794
|
-
* `upstream_inference_prompt_cost`, `upstream_inference_input_cost`) onto these
|
|
795
|
-
* fields at runtime.
|
|
793
|
+
* @deprecated Renamed to {@link TokenUsage}. Kept as an alias for backward
|
|
794
|
+
* compatibility with `@tanstack/ai@0.23` and earlier; will be removed in a
|
|
795
|
+
* future release.
|
|
796
796
|
*/
|
|
797
|
-
export
|
|
798
|
-
/** Total cost the gateway paid the upstream provider. */
|
|
799
|
-
upstreamCost?: number;
|
|
800
|
-
/** Upstream cost for input (prompt) tokens. */
|
|
801
|
-
upstreamInputCost?: number;
|
|
802
|
-
/** Upstream cost for output (completion) tokens. */
|
|
803
|
-
upstreamOutputCost?: number;
|
|
804
|
-
}
|
|
805
|
-
/**
|
|
806
|
-
* Token usage totals for a run, optionally including provider-reported cost.
|
|
807
|
-
*
|
|
808
|
-
* `cost` and `costDetails` are populated only by adapters whose provider returns
|
|
809
|
-
* authoritative per-request cost (e.g. OpenRouter). They are absent for adapters
|
|
810
|
-
* that do not report cost, so consumers must treat them as optional.
|
|
811
|
-
*/
|
|
812
|
-
export interface UsageTotals {
|
|
813
|
-
promptTokens: number;
|
|
814
|
-
completionTokens: number;
|
|
815
|
-
totalTokens: number;
|
|
816
|
-
/** Provider-reported cost for the request, when available. */
|
|
817
|
-
cost?: number;
|
|
818
|
-
/** Provider-reported cost breakdown, when available. */
|
|
819
|
-
costDetails?: UsageCostBreakdown;
|
|
820
|
-
}
|
|
797
|
+
export type UsageTotals = TokenUsage;
|
|
821
798
|
/**
|
|
822
799
|
* Emitted when a run completes successfully.
|
|
823
800
|
*
|
|
@@ -829,8 +806,8 @@ export interface RunFinishedEvent extends AGUIRunFinishedEvent {
|
|
|
829
806
|
model?: string;
|
|
830
807
|
/** Why the generation stopped */
|
|
831
808
|
finishReason?: 'stop' | 'length' | 'content_filter' | 'tool_calls' | null;
|
|
832
|
-
/** Token usage statistics
|
|
833
|
-
usage?:
|
|
809
|
+
/** Token usage statistics with optional detailed breakdowns and provider-reported cost. */
|
|
810
|
+
usage?: TokenUsage;
|
|
834
811
|
}
|
|
835
812
|
/**
|
|
836
813
|
* Emitted when an error occurs during a run.
|
|
@@ -1219,11 +1196,7 @@ export interface TextCompletionChunk {
|
|
|
1219
1196
|
content: string;
|
|
1220
1197
|
role?: 'assistant';
|
|
1221
1198
|
finishReason?: 'stop' | 'length' | 'content_filter' | null;
|
|
1222
|
-
usage?:
|
|
1223
|
-
promptTokens: number;
|
|
1224
|
-
completionTokens: number;
|
|
1225
|
-
totalTokens: number;
|
|
1226
|
-
};
|
|
1199
|
+
usage?: TokenUsage;
|
|
1227
1200
|
}
|
|
1228
1201
|
export interface SummarizationOptions<TProviderOptions extends object = Record<string, unknown>> {
|
|
1229
1202
|
model: string;
|
|
@@ -1243,11 +1216,7 @@ export interface SummarizationResult {
|
|
|
1243
1216
|
id: string;
|
|
1244
1217
|
model: string;
|
|
1245
1218
|
summary: string;
|
|
1246
|
-
usage:
|
|
1247
|
-
promptTokens: number;
|
|
1248
|
-
completionTokens: number;
|
|
1249
|
-
totalTokens: number;
|
|
1250
|
-
};
|
|
1219
|
+
usage: TokenUsage;
|
|
1251
1220
|
}
|
|
1252
1221
|
/**
|
|
1253
1222
|
* Options for image generation.
|
|
@@ -1303,11 +1272,7 @@ export interface ImageGenerationResult {
|
|
|
1303
1272
|
/** Array of generated images */
|
|
1304
1273
|
images: Array<GeneratedImage>;
|
|
1305
1274
|
/** Token usage information (if available) */
|
|
1306
|
-
usage?:
|
|
1307
|
-
inputTokens?: number;
|
|
1308
|
-
outputTokens?: number;
|
|
1309
|
-
totalTokens?: number;
|
|
1310
|
-
};
|
|
1275
|
+
usage?: TokenUsage;
|
|
1311
1276
|
}
|
|
1312
1277
|
/**
|
|
1313
1278
|
* Options for audio generation (music, sound effects, etc.).
|
|
@@ -1349,11 +1314,7 @@ export interface AudioGenerationResult {
|
|
|
1349
1314
|
/** The generated audio */
|
|
1350
1315
|
audio: GeneratedAudio;
|
|
1351
1316
|
/** Token usage information (if available) */
|
|
1352
|
-
usage?:
|
|
1353
|
-
inputTokens?: number;
|
|
1354
|
-
outputTokens?: number;
|
|
1355
|
-
totalTokens?: number;
|
|
1356
|
-
};
|
|
1317
|
+
usage?: TokenUsage;
|
|
1357
1318
|
}
|
|
1358
1319
|
/**
|
|
1359
1320
|
* Options for video generation.
|
|
@@ -1457,6 +1418,8 @@ export interface TTSResult {
|
|
|
1457
1418
|
duration?: number;
|
|
1458
1419
|
/** Content type of the audio (e.g., 'audio/mp3') */
|
|
1459
1420
|
contentType?: string;
|
|
1421
|
+
/** Token usage information (if provided by the adapter) */
|
|
1422
|
+
usage?: TokenUsage;
|
|
1460
1423
|
}
|
|
1461
1424
|
/**
|
|
1462
1425
|
* Options for audio transcription.
|
|
@@ -1528,6 +1491,8 @@ export interface TranscriptionResult {
|
|
|
1528
1491
|
segments?: Array<TranscriptionSegment>;
|
|
1529
1492
|
/** Word-level timestamps, if available */
|
|
1530
1493
|
words?: Array<TranscriptionWord>;
|
|
1494
|
+
/** Token usage information (if provided by the adapter) */
|
|
1495
|
+
usage?: TokenUsage;
|
|
1531
1496
|
}
|
|
1532
1497
|
/**
|
|
1533
1498
|
* Default metadata type for adapters that don't define custom metadata.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { ProviderUsageDetails, TokenUsage } from '../types.js';
|
|
2
|
+
/**
|
|
3
|
+
* Input parameters for building base TokenUsage.
|
|
4
|
+
* Provider functions should extract these from their SDK's response.
|
|
5
|
+
*/
|
|
6
|
+
export interface BaseUsageInput {
|
|
7
|
+
/** Total input/prompt tokens */
|
|
8
|
+
promptTokens: number;
|
|
9
|
+
/** Total output/completion tokens */
|
|
10
|
+
completionTokens: number;
|
|
11
|
+
/** Total tokens (prompt + completion) */
|
|
12
|
+
totalTokens: number;
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* Builds the base TokenUsage object with core fields.
|
|
16
|
+
* Provider-specific functions should use this and then add their own details.
|
|
17
|
+
*
|
|
18
|
+
* @param input - The base token counts
|
|
19
|
+
* @returns A TokenUsage object with promptTokens, completionTokens, totalTokens
|
|
20
|
+
*
|
|
21
|
+
* @example
|
|
22
|
+
* ```typescript
|
|
23
|
+
* const base = buildBaseUsage({
|
|
24
|
+
* promptTokens: 100,
|
|
25
|
+
* completionTokens: 50,
|
|
26
|
+
* totalTokens: 150
|
|
27
|
+
* });
|
|
28
|
+
* // Returns: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
|
|
29
|
+
* ```
|
|
30
|
+
*/
|
|
31
|
+
export declare function buildBaseUsage<TProviderDetails = ProviderUsageDetails>(input: BaseUsageInput): TokenUsage<TProviderDetails>;
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"usage.js","sources":["../../../src/utilities/usage.ts"],"sourcesContent":["import type { ProviderUsageDetails, TokenUsage } from '../types'\n\n/**\n * Input parameters for building base TokenUsage.\n * Provider functions should extract these from their SDK's response.\n */\nexport interface BaseUsageInput {\n /** Total input/prompt tokens */\n promptTokens: number\n /** Total output/completion tokens */\n completionTokens: number\n /** Total tokens (prompt + completion) */\n totalTokens: number\n}\n\n/**\n * Builds the base TokenUsage object with core fields.\n * Provider-specific functions should use this and then add their own details.\n *\n * @param input - The base token counts\n * @returns A TokenUsage object with promptTokens, completionTokens, totalTokens\n *\n * @example\n * ```typescript\n * const base = buildBaseUsage({\n * promptTokens: 100,\n * completionTokens: 50,\n * totalTokens: 150\n * });\n * // Returns: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }\n * ```\n */\nexport function buildBaseUsage<TProviderDetails = ProviderUsageDetails>(\n input: BaseUsageInput,\n): TokenUsage<TProviderDetails> {\n return {\n promptTokens: input.promptTokens,\n completionTokens: input.completionTokens,\n totalTokens: input.totalTokens,\n }\n}\n"],"names":[],"mappings":"AAgCO,SAAS,eACd,OAC8B;AAC9B,SAAO;AAAA,IACL,cAAc,MAAM;AAAA,IACpB,kBAAkB,MAAM;AAAA,IACxB,aAAa,MAAM;AAAA,EAAA;AAEvB;"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.25.0",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -68,7 +68,7 @@
|
|
|
68
68
|
"@ag-ui/core": "^0.0.52",
|
|
69
69
|
"@standard-schema/spec": "^1.1.0",
|
|
70
70
|
"partial-json": "^0.1.7",
|
|
71
|
-
"@tanstack/ai-event-client": "0.
|
|
71
|
+
"@tanstack/ai-event-client": "0.5.0"
|
|
72
72
|
},
|
|
73
73
|
"peerDependencies": {
|
|
74
74
|
"@opentelemetry/api": ">=1.9.0"
|
|
@@ -4,6 +4,7 @@ import type {
|
|
|
4
4
|
Modality,
|
|
5
5
|
StreamChunk,
|
|
6
6
|
TextOptions,
|
|
7
|
+
TokenUsage,
|
|
7
8
|
} from '../../types'
|
|
8
9
|
|
|
9
10
|
/**
|
|
@@ -40,6 +41,8 @@ export interface StructuredOutputResult<T = unknown> {
|
|
|
40
41
|
data: T
|
|
41
42
|
/** The raw text response from the model before parsing */
|
|
42
43
|
rawText: string
|
|
44
|
+
/** Token usage information (if provided by the adapter) */
|
|
45
|
+
usage?: TokenUsage
|
|
43
46
|
}
|
|
44
47
|
|
|
45
48
|
/**
|
|
@@ -2,9 +2,9 @@ import type {
|
|
|
2
2
|
JSONSchema,
|
|
3
3
|
ModelMessage,
|
|
4
4
|
StreamChunk,
|
|
5
|
+
TokenUsage,
|
|
5
6
|
Tool,
|
|
6
7
|
ToolCall,
|
|
7
|
-
UsageTotals,
|
|
8
8
|
} from '../../../types'
|
|
9
9
|
import type { SystemPrompt } from '../../../system-prompts'
|
|
10
10
|
|
|
@@ -266,11 +266,11 @@ export interface ToolPhaseCompleteInfo {
|
|
|
266
266
|
* Token usage statistics passed to the onUsage hook.
|
|
267
267
|
* Extracted from the RUN_FINISHED chunk when usage data is present.
|
|
268
268
|
*
|
|
269
|
-
* Includes optional provider-reported `cost`/`costDetails` (see {@link
|
|
270
|
-
* Kept as an interface extending `
|
|
269
|
+
* Includes optional provider-reported `cost`/`costDetails` (see {@link TokenUsage}).
|
|
270
|
+
* Kept as an interface extending `TokenUsage` to preserve declaration merging for
|
|
271
271
|
* this publicly exported type.
|
|
272
272
|
*/
|
|
273
|
-
export interface UsageInfo extends
|
|
273
|
+
export interface UsageInfo extends TokenUsage {}
|
|
274
274
|
|
|
275
275
|
// ===========================
|
|
276
276
|
// Terminal Hook Info
|
|
@@ -287,7 +287,7 @@ export interface FinishInfo {
|
|
|
287
287
|
/** Final accumulated text content */
|
|
288
288
|
content: string
|
|
289
289
|
/** Final usage totals, if available (optionally including provider-reported cost) */
|
|
290
|
-
usage?:
|
|
290
|
+
usage?: TokenUsage | undefined
|
|
291
291
|
}
|
|
292
292
|
|
|
293
293
|
/**
|
|
@@ -174,6 +174,16 @@ async function runGenerateAudio<
|
|
|
174
174
|
timestamp: Date.now(),
|
|
175
175
|
})
|
|
176
176
|
|
|
177
|
+
if (result.usage) {
|
|
178
|
+
aiEventClient.emit('audio:usage', {
|
|
179
|
+
requestId,
|
|
180
|
+
model,
|
|
181
|
+
usage: result.usage,
|
|
182
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
183
|
+
timestamp: Date.now(),
|
|
184
|
+
})
|
|
185
|
+
}
|
|
186
|
+
|
|
177
187
|
logger.output(`activity=generateAudio provider=${providerName}`, {
|
|
178
188
|
contentType: result.audio.contentType,
|
|
179
189
|
audioDuration: result.audio.duration,
|
|
@@ -187,6 +187,16 @@ async function runGenerateSpeech<
|
|
|
187
187
|
timestamp: Date.now(),
|
|
188
188
|
})
|
|
189
189
|
|
|
190
|
+
if (result.usage) {
|
|
191
|
+
aiEventClient.emit('speech:usage', {
|
|
192
|
+
requestId,
|
|
193
|
+
model,
|
|
194
|
+
usage: result.usage,
|
|
195
|
+
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
196
|
+
timestamp: Date.now(),
|
|
197
|
+
})
|
|
198
|
+
}
|
|
199
|
+
|
|
190
200
|
logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {
|
|
191
201
|
bytes: result.audio.length,
|
|
192
202
|
contentType: result.contentType,
|
package/src/index.ts
CHANGED
|
@@ -111,6 +111,9 @@ export type {
|
|
|
111
111
|
// All types
|
|
112
112
|
export * from './types'
|
|
113
113
|
|
|
114
|
+
// Usage utilities
|
|
115
|
+
export { buildBaseUsage, type BaseUsageInput } from './utilities/usage'
|
|
116
|
+
|
|
114
117
|
// System prompts (type + normaliser used by adapters)
|
|
115
118
|
export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts'
|
|
116
119
|
export { normalizeSystemPrompts } from './system-prompts'
|
package/src/types.ts
CHANGED
|
@@ -4,6 +4,16 @@ import type {
|
|
|
4
4
|
} from '@standard-schema/spec'
|
|
5
5
|
import type { InternalLogger } from './logger/internal-logger'
|
|
6
6
|
import type { SystemPrompt } from './system-prompts'
|
|
7
|
+
// The canonical usage types live in the leaf `@tanstack/ai-event-client`
|
|
8
|
+
// package (which `@tanstack/ai` already depends on) so there is a single source
|
|
9
|
+
// of truth without a dependency cycle. They are re-exported below.
|
|
10
|
+
import type {
|
|
11
|
+
CompletionTokensDetails,
|
|
12
|
+
PromptTokensDetails,
|
|
13
|
+
ProviderUsageDetails,
|
|
14
|
+
TokenUsage,
|
|
15
|
+
UsageCostBreakdown,
|
|
16
|
+
} from '@tanstack/ai-event-client'
|
|
7
17
|
import type {
|
|
8
18
|
BaseEvent as AGUIBaseEvent,
|
|
9
19
|
CustomEvent as AGUICustomEvent,
|
|
@@ -978,38 +988,22 @@ export interface RunStartedEvent extends AGUIRunStartedEvent {
|
|
|
978
988
|
model?: string
|
|
979
989
|
}
|
|
980
990
|
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
/** Total cost the gateway paid the upstream provider. */
|
|
990
|
-
upstreamCost?: number
|
|
991
|
-
/** Upstream cost for input (prompt) tokens. */
|
|
992
|
-
upstreamInputCost?: number
|
|
993
|
-
/** Upstream cost for output (completion) tokens. */
|
|
994
|
-
upstreamOutputCost?: number
|
|
991
|
+
// Re-export the canonical usage types (defined in `@tanstack/ai-event-client`)
|
|
992
|
+
// so `@tanstack/ai` consumers keep importing them from here unchanged.
|
|
993
|
+
export type {
|
|
994
|
+
CompletionTokensDetails,
|
|
995
|
+
PromptTokensDetails,
|
|
996
|
+
ProviderUsageDetails,
|
|
997
|
+
TokenUsage,
|
|
998
|
+
UsageCostBreakdown,
|
|
995
999
|
}
|
|
996
1000
|
|
|
997
1001
|
/**
|
|
998
|
-
*
|
|
999
|
-
*
|
|
1000
|
-
*
|
|
1001
|
-
* authoritative per-request cost (e.g. OpenRouter). They are absent for adapters
|
|
1002
|
-
* that do not report cost, so consumers must treat them as optional.
|
|
1002
|
+
* @deprecated Renamed to {@link TokenUsage}. Kept as an alias for backward
|
|
1003
|
+
* compatibility with `@tanstack/ai@0.23` and earlier; will be removed in a
|
|
1004
|
+
* future release.
|
|
1003
1005
|
*/
|
|
1004
|
-
export
|
|
1005
|
-
promptTokens: number
|
|
1006
|
-
completionTokens: number
|
|
1007
|
-
totalTokens: number
|
|
1008
|
-
/** Provider-reported cost for the request, when available. */
|
|
1009
|
-
cost?: number
|
|
1010
|
-
/** Provider-reported cost breakdown, when available. */
|
|
1011
|
-
costDetails?: UsageCostBreakdown
|
|
1012
|
-
}
|
|
1006
|
+
export type UsageTotals = TokenUsage
|
|
1013
1007
|
|
|
1014
1008
|
/**
|
|
1015
1009
|
* Emitted when a run completes successfully.
|
|
@@ -1022,8 +1016,8 @@ export interface RunFinishedEvent extends AGUIRunFinishedEvent {
|
|
|
1022
1016
|
model?: string
|
|
1023
1017
|
/** Why the generation stopped */
|
|
1024
1018
|
finishReason?: 'stop' | 'length' | 'content_filter' | 'tool_calls' | null
|
|
1025
|
-
/** Token usage statistics
|
|
1026
|
-
usage?:
|
|
1019
|
+
/** Token usage statistics with optional detailed breakdowns and provider-reported cost. */
|
|
1020
|
+
usage?: TokenUsage
|
|
1027
1021
|
}
|
|
1028
1022
|
|
|
1029
1023
|
/**
|
|
@@ -1473,11 +1467,7 @@ export interface TextCompletionChunk {
|
|
|
1473
1467
|
content: string
|
|
1474
1468
|
role?: 'assistant'
|
|
1475
1469
|
finishReason?: 'stop' | 'length' | 'content_filter' | null
|
|
1476
|
-
usage?:
|
|
1477
|
-
promptTokens: number
|
|
1478
|
-
completionTokens: number
|
|
1479
|
-
totalTokens: number
|
|
1480
|
-
}
|
|
1470
|
+
usage?: TokenUsage
|
|
1481
1471
|
}
|
|
1482
1472
|
|
|
1483
1473
|
export interface SummarizationOptions<
|
|
@@ -1501,11 +1491,7 @@ export interface SummarizationResult {
|
|
|
1501
1491
|
id: string
|
|
1502
1492
|
model: string
|
|
1503
1493
|
summary: string
|
|
1504
|
-
usage:
|
|
1505
|
-
promptTokens: number
|
|
1506
|
-
completionTokens: number
|
|
1507
|
-
totalTokens: number
|
|
1508
|
-
}
|
|
1494
|
+
usage: TokenUsage
|
|
1509
1495
|
}
|
|
1510
1496
|
|
|
1511
1497
|
// ============================================================================
|
|
@@ -1574,11 +1560,7 @@ export interface ImageGenerationResult {
|
|
|
1574
1560
|
/** Array of generated images */
|
|
1575
1561
|
images: Array<GeneratedImage>
|
|
1576
1562
|
/** Token usage information (if available) */
|
|
1577
|
-
usage?:
|
|
1578
|
-
inputTokens?: number
|
|
1579
|
-
outputTokens?: number
|
|
1580
|
-
totalTokens?: number
|
|
1581
|
-
}
|
|
1563
|
+
usage?: TokenUsage
|
|
1582
1564
|
}
|
|
1583
1565
|
|
|
1584
1566
|
// ============================================================================
|
|
@@ -1629,11 +1611,7 @@ export interface AudioGenerationResult {
|
|
|
1629
1611
|
/** The generated audio */
|
|
1630
1612
|
audio: GeneratedAudio
|
|
1631
1613
|
/** Token usage information (if available) */
|
|
1632
|
-
usage?:
|
|
1633
|
-
inputTokens?: number
|
|
1634
|
-
outputTokens?: number
|
|
1635
|
-
totalTokens?: number
|
|
1636
|
-
}
|
|
1614
|
+
usage?: TokenUsage
|
|
1637
1615
|
}
|
|
1638
1616
|
|
|
1639
1617
|
// ============================================================================
|
|
@@ -1754,6 +1732,8 @@ export interface TTSResult {
|
|
|
1754
1732
|
duration?: number
|
|
1755
1733
|
/** Content type of the audio (e.g., 'audio/mp3') */
|
|
1756
1734
|
contentType?: string
|
|
1735
|
+
/** Token usage information (if provided by the adapter) */
|
|
1736
|
+
usage?: TokenUsage
|
|
1757
1737
|
}
|
|
1758
1738
|
|
|
1759
1739
|
// ============================================================================
|
|
@@ -1835,6 +1815,8 @@ export interface TranscriptionResult {
|
|
|
1835
1815
|
segments?: Array<TranscriptionSegment>
|
|
1836
1816
|
/** Word-level timestamps, if available */
|
|
1837
1817
|
words?: Array<TranscriptionWord>
|
|
1818
|
+
/** Token usage information (if provided by the adapter) */
|
|
1819
|
+
usage?: TokenUsage
|
|
1838
1820
|
}
|
|
1839
1821
|
|
|
1840
1822
|
/**
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import type { ProviderUsageDetails, TokenUsage } from '../types'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Input parameters for building base TokenUsage.
|
|
5
|
+
* Provider functions should extract these from their SDK's response.
|
|
6
|
+
*/
|
|
7
|
+
export interface BaseUsageInput {
|
|
8
|
+
/** Total input/prompt tokens */
|
|
9
|
+
promptTokens: number
|
|
10
|
+
/** Total output/completion tokens */
|
|
11
|
+
completionTokens: number
|
|
12
|
+
/** Total tokens (prompt + completion) */
|
|
13
|
+
totalTokens: number
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Builds the base TokenUsage object with core fields.
|
|
18
|
+
* Provider-specific functions should use this and then add their own details.
|
|
19
|
+
*
|
|
20
|
+
* @param input - The base token counts
|
|
21
|
+
* @returns A TokenUsage object with promptTokens, completionTokens, totalTokens
|
|
22
|
+
*
|
|
23
|
+
* @example
|
|
24
|
+
* ```typescript
|
|
25
|
+
* const base = buildBaseUsage({
|
|
26
|
+
* promptTokens: 100,
|
|
27
|
+
* completionTokens: 50,
|
|
28
|
+
* totalTokens: 150
|
|
29
|
+
* });
|
|
30
|
+
* // Returns: { promptTokens: 100, completionTokens: 50, totalTokens: 150 }
|
|
31
|
+
* ```
|
|
32
|
+
*/
|
|
33
|
+
export function buildBaseUsage<TProviderDetails = ProviderUsageDetails>(
|
|
34
|
+
input: BaseUsageInput,
|
|
35
|
+
): TokenUsage<TProviderDetails> {
|
|
36
|
+
return {
|
|
37
|
+
promptTokens: input.promptTokens,
|
|
38
|
+
completionTokens: input.completionTokens,
|
|
39
|
+
totalTokens: input.totalTokens,
|
|
40
|
+
}
|
|
41
|
+
}
|