@tanstack/ai 0.29.0 → 0.32.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/dist/esm/activities/chat/adapter.d.ts +10 -0
  2. package/dist/esm/activities/chat/adapter.js +1 -0
  3. package/dist/esm/activities/chat/adapter.js.map +1 -1
  4. package/dist/esm/activities/chat/index.d.ts +3 -2
  5. package/dist/esm/activities/chat/index.js +13 -1
  6. package/dist/esm/activities/chat/index.js.map +1 -1
  7. package/dist/esm/activities/chat/middleware/builder.d.ts +46 -0
  8. package/dist/esm/activities/chat/middleware/builder.js +17 -0
  9. package/dist/esm/activities/chat/middleware/builder.js.map +1 -0
  10. package/dist/esm/activities/chat/middleware/capabilities.d.ts +93 -0
  11. package/dist/esm/activities/chat/middleware/capabilities.js +45 -0
  12. package/dist/esm/activities/chat/middleware/capabilities.js.map +1 -0
  13. package/dist/esm/activities/chat/middleware/compose.d.ts +10 -0
  14. package/dist/esm/activities/chat/middleware/compose.js +47 -0
  15. package/dist/esm/activities/chat/middleware/compose.js.map +1 -1
  16. package/dist/esm/activities/chat/middleware/define.d.ts +20 -0
  17. package/dist/esm/activities/chat/middleware/define.js +7 -0
  18. package/dist/esm/activities/chat/middleware/define.js.map +1 -0
  19. package/dist/esm/activities/chat/middleware/index.d.ts +8 -0
  20. package/dist/esm/activities/chat/middleware/types.d.ts +51 -0
  21. package/dist/esm/activities/chat/middleware/validate.d.ts +19 -0
  22. package/dist/esm/activities/chat/middleware/validate.js +30 -0
  23. package/dist/esm/activities/chat/middleware/validate.js.map +1 -0
  24. package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
  25. package/dist/esm/activities/generateImage/adapter.js.map +1 -1
  26. package/dist/esm/activities/generateImage/index.d.ts +19 -3
  27. package/dist/esm/activities/generateImage/index.js +12 -1
  28. package/dist/esm/activities/generateImage/index.js.map +1 -1
  29. package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
  30. package/dist/esm/activities/generateVideo/adapter.js +14 -0
  31. package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
  32. package/dist/esm/activities/generateVideo/index.d.ts +31 -5
  33. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  34. package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
  35. package/dist/esm/activities/generateVideo/snap.js +54 -0
  36. package/dist/esm/activities/generateVideo/snap.js.map +1 -0
  37. package/dist/esm/activities/index.d.ts +3 -2
  38. package/dist/esm/activities/index.js +2 -0
  39. package/dist/esm/activities/index.js.map +1 -1
  40. package/dist/esm/client.d.ts +1 -1
  41. package/dist/esm/client.js.map +1 -1
  42. package/dist/esm/index.d.ts +4 -0
  43. package/dist/esm/index.js +8 -0
  44. package/dist/esm/index.js.map +1 -1
  45. package/dist/esm/middlewares/otel.js +40 -12
  46. package/dist/esm/middlewares/otel.js.map +1 -1
  47. package/dist/esm/types.d.ts +96 -7
  48. package/dist/esm/utilities/media-prompt.d.ts +35 -0
  49. package/dist/esm/utilities/media-prompt.js +43 -0
  50. package/dist/esm/utilities/media-prompt.js.map +1 -0
  51. package/package.json +10 -2
  52. package/skills/ai-core/media-generation/SKILL.md +173 -3
  53. package/src/activities/chat/adapter.ts +11 -0
  54. package/src/activities/chat/index.ts +20 -1
  55. package/src/activities/chat/middleware/builder.ts +109 -0
  56. package/src/activities/chat/middleware/capabilities.ts +162 -0
  57. package/src/activities/chat/middleware/compose.ts +51 -0
  58. package/src/activities/chat/middleware/define.ts +34 -0
  59. package/src/activities/chat/middleware/index.ts +20 -0
  60. package/src/activities/chat/middleware/types.ts +60 -0
  61. package/src/activities/chat/middleware/validate.ts +55 -0
  62. package/src/activities/generateImage/adapter.ts +16 -3
  63. package/src/activities/generateImage/index.ts +48 -4
  64. package/src/activities/generateVideo/adapter.ts +80 -4
  65. package/src/activities/generateVideo/index.ts +53 -4
  66. package/src/activities/generateVideo/snap.ts +100 -0
  67. package/src/activities/index.ts +4 -0
  68. package/src/client.ts +4 -0
  69. package/src/index.ts +18 -0
  70. package/src/middlewares/otel.ts +57 -12
  71. package/src/types.ts +119 -6
  72. package/src/utilities/media-prompt.ts +86 -0
@@ -1,10 +1,30 @@
1
1
  import type {
2
+ ModelInputModalitiesByName,
2
3
  VideoGenerationOptions,
3
4
  VideoJobResult,
4
5
  VideoStatusResult,
5
6
  VideoUrlResult,
6
7
  } from '../../types'
7
8
 
9
+ /**
10
+ * Structured description of the durations a video model accepts.
11
+ *
12
+ * Tagged union so the same shape can express discrete enums (OpenAI Sora,
13
+ * Veo), continuous ranges, mixed shapes, and models with no duration field.
14
+ * Consumed by `VideoAdapter.availableDurations()`.
15
+ *
16
+ * @experimental Video generation is an experimental feature and may change.
17
+ */
18
+ export type DurationOptions<T extends string | number | undefined> =
19
+ | { kind: 'discrete'; values: ReadonlyArray<NonNullable<T>> }
20
+ | { kind: 'range'; min: number; max: number; step?: number; unit: 'seconds' }
21
+ | {
22
+ kind: 'mixed'
23
+ values: ReadonlyArray<NonNullable<T>>
24
+ range?: { min: number; max: number; step?: number }
25
+ }
26
+ | { kind: 'none' }
27
+
8
28
  /**
9
29
  * Configuration for video adapter instances
10
30
  *
@@ -31,6 +51,11 @@ export interface VideoAdapterConfig {
31
51
  * - TProviderOptions: Provider-specific options (already resolved)
32
52
  * - TModelProviderOptionsByName: Map from model name to its specific provider options
33
53
  * - TModelSizeByName: Map from model name to its supported sizes
54
+ * - TModelInputModalitiesByName: Map from model name to the non-text prompt
55
+ * modalities it accepts (constrains the `prompt` part types at compile time)
56
+ * - TModelDurationByName: Map from model name to its supported duration
57
+ * union. Defaults to `Record<string, number>` so adapters that haven't
58
+ * declared a map keep today's `duration?: number` typing.
34
59
  */
35
60
  export interface VideoAdapter<
36
61
  TModel extends string = string,
@@ -40,6 +65,10 @@ export interface VideoAdapter<
40
65
  string,
41
66
  string
42
67
  >,
68
+ TModelInputModalitiesByName extends ModelInputModalitiesByName =
69
+ ModelInputModalitiesByName,
70
+ TModelDurationByName extends Record<string, string | number | undefined> =
71
+ Record<string, number>,
43
72
  > {
44
73
  /** Discriminator for adapter kind - used to determine API shape */
45
74
  readonly kind: 'video'
@@ -55,6 +84,8 @@ export interface VideoAdapter<
55
84
  providerOptions: TProviderOptions
56
85
  modelProviderOptionsByName: TModelProviderOptionsByName
57
86
  modelSizeByName: TModelSizeByName
87
+ modelInputModalitiesByName: TModelInputModalitiesByName
88
+ modelDurationByName: TModelDurationByName
58
89
  }
59
90
 
60
91
  /**
@@ -62,7 +93,11 @@ export interface VideoAdapter<
62
93
  * Returns a job ID that can be used to poll for status and retrieve the video.
63
94
  */
64
95
  createVideoJob: (
65
- options: VideoGenerationOptions<TProviderOptions, TModelSizeByName[TModel]>,
96
+ options: VideoGenerationOptions<
97
+ TProviderOptions,
98
+ TModelSizeByName[TModel],
99
+ TModelDurationByName[TModel]
100
+ >,
66
101
  ) => Promise<VideoJobResult>
67
102
 
68
103
  /**
@@ -75,13 +110,26 @@ export interface VideoAdapter<
75
110
  * Should only be called after status is 'completed'.
76
111
  */
77
112
  getVideoUrl: (jobId: string) => Promise<VideoUrlResult>
113
+
114
+ /**
115
+ * Describe the durations this adapter's model accepts. Returns a tagged
116
+ * union so consumers can render UI / coerce input without provider-specific
117
+ * knowledge.
118
+ */
119
+ availableDurations: () => DurationOptions<TModelDurationByName[TModel]>
120
+
121
+ /**
122
+ * Coerce a raw seconds value to the closest valid duration for this model.
123
+ * Returns `undefined` for models with no duration field.
124
+ */
125
+ snapDuration: (seconds: number) => TModelDurationByName[TModel] | undefined
78
126
  }
79
127
 
80
128
  /**
81
129
  * A VideoAdapter with any/unknown type parameters.
82
130
  * Useful as a constraint in generic functions and interfaces.
83
131
  */
84
- export type AnyVideoAdapter = VideoAdapter<any, any, any, any>
132
+ export type AnyVideoAdapter = VideoAdapter<any, any, any, any, any, any>
85
133
 
86
134
  /**
87
135
  * Abstract base class for video generation adapters.
@@ -99,11 +147,17 @@ export abstract class BaseVideoAdapter<
99
147
  string,
100
148
  string
101
149
  >,
150
+ TModelInputModalitiesByName extends ModelInputModalitiesByName =
151
+ ModelInputModalitiesByName,
152
+ TModelDurationByName extends Record<string, string | number | undefined> =
153
+ Record<string, number>,
102
154
  > implements VideoAdapter<
103
155
  TModel,
104
156
  TProviderOptions,
105
157
  TModelProviderOptionsByName,
106
- TModelSizeByName
158
+ TModelSizeByName,
159
+ TModelInputModalitiesByName,
160
+ TModelDurationByName
107
161
  > {
108
162
  readonly kind = 'video' as const
109
163
  abstract readonly name: string
@@ -114,6 +168,8 @@ export abstract class BaseVideoAdapter<
114
168
  providerOptions: TProviderOptions
115
169
  modelProviderOptionsByName: TModelProviderOptionsByName
116
170
  modelSizeByName: TModelSizeByName
171
+ modelInputModalitiesByName: TModelInputModalitiesByName
172
+ modelDurationByName: TModelDurationByName
117
173
  }
118
174
 
119
175
  protected config: VideoAdapterConfig
@@ -124,13 +180,33 @@ export abstract class BaseVideoAdapter<
124
180
  }
125
181
 
126
182
  abstract createVideoJob(
127
- options: VideoGenerationOptions<TProviderOptions, TModelSizeByName[TModel]>,
183
+ options: VideoGenerationOptions<
184
+ TProviderOptions,
185
+ TModelSizeByName[TModel],
186
+ TModelDurationByName[TModel]
187
+ >,
128
188
  ): Promise<VideoJobResult>
129
189
 
130
190
  abstract getVideoStatus(jobId: string): Promise<VideoStatusResult>
131
191
 
132
192
  abstract getVideoUrl(jobId: string): Promise<VideoUrlResult>
133
193
 
194
+ /**
195
+ * Default implementation returns `{ kind: 'none' }`. Adapters that have
196
+ * declared their per-model duration map should override this.
197
+ */
198
+ availableDurations(): DurationOptions<TModelDurationByName[TModel]> {
199
+ return { kind: 'none' }
200
+ }
201
+
202
+ /**
203
+ * Default implementation returns `undefined`. Adapters that have declared
204
+ * their per-model duration map should override.
205
+ */
206
+ snapDuration(_seconds: number): TModelDurationByName[TModel] | undefined {
207
+ return undefined
208
+ }
209
+
134
210
  protected generateId(): string {
135
211
  return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`
136
212
  }
@@ -14,6 +14,8 @@ import type { InternalLogger } from '../../logger/internal-logger'
14
14
  import type { DebugOption } from '../../logger/types'
15
15
  import type { VideoAdapter } from './adapter'
16
16
  import type {
17
+ MediaPrompt,
18
+ MediaPromptFor,
17
19
  StreamChunk,
18
20
  TokenUsage,
19
21
  VideoJobResult,
@@ -50,6 +52,40 @@ export type VideoSizeForAdapter<TAdapter> =
50
52
  : string
51
53
  : string
52
54
 
55
+ /**
56
+ * Extract the prompt type a model accepts from a VideoAdapter via ~types.
57
+ * Mirrors `ImagePromptForModel`: models in the adapter's input-modality map
58
+ * get a `prompt` narrowed to text + their supported part types; adapters
59
+ * without a map fall back to the full MediaPrompt.
60
+ */
61
+ export type VideoPromptForAdapter<TAdapter> =
62
+ TAdapter extends VideoAdapter<infer TModel, any, any, any, infer ModsByName>
63
+ ? string extends keyof ModsByName
64
+ ? MediaPrompt
65
+ : TModel extends keyof ModsByName
66
+ ? MediaPromptFor<ModsByName[TModel][number]>
67
+ : MediaPrompt
68
+ : MediaPrompt
69
+
70
+ /**
71
+ * Extract the duration type for a VideoAdapter's model via ~types.
72
+ * Mirrors `VideoSizeForAdapter`. Falls back to `number` for adapters that
73
+ * haven't declared per-model duration constraints.
74
+ */
75
+ export type VideoDurationForAdapter<TAdapter> =
76
+ TAdapter extends VideoAdapter<
77
+ infer TModel,
78
+ any,
79
+ any,
80
+ any,
81
+ any,
82
+ infer TDurationMap
83
+ >
84
+ ? TModel extends keyof TDurationMap
85
+ ? TDurationMap[TModel]
86
+ : number
87
+ : number
88
+
53
89
  // ===========================
54
90
  // Activity Options Types
55
91
 
@@ -84,12 +120,25 @@ export type VideoCreateOptions<
84
120
  > = VideoActivityBaseOptions<TAdapter> & {
85
121
  /** Request type - create a new job (default if not specified) */
86
122
  request?: 'create'
87
- /** Text description of the desired video */
88
- prompt: string
123
+ /**
124
+ * Description of the desired video. Either a plain string, or — for models
125
+ * that support image-conditioned generation — an ordered array of content
126
+ * parts interleaving text with image inputs. Image parts may carry
127
+ * `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
128
+ * 'character'`) to disambiguate intent; positional fallback otherwise. The
129
+ * accepted part types are narrowed per model via the adapter's
130
+ * input-modality map.
131
+ */
132
+ prompt: VideoPromptForAdapter<TAdapter>
89
133
  /** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
90
134
  size?: VideoSizeForAdapter<TAdapter>
91
- /** Video duration in seconds */
92
- duration?: number
135
+ /**
136
+ * Video duration in seconds. Adapters that declare a per-model duration
137
+ * map narrow this to the model's valid union (e.g. `4 | 6 | 8` for Veo 3).
138
+ * Pass `adapter.snapDuration(seconds)` to coerce raw seconds to a valid
139
+ * value.
140
+ */
141
+ duration?: VideoDurationForAdapter<TAdapter>
93
142
  /**
94
143
  * Whether to stream the video generation lifecycle.
95
144
  * When true, returns an AsyncIterable<StreamChunk> that handles the full
@@ -0,0 +1,100 @@
1
+ import type { DurationOptions } from './adapter'
2
+
3
+ /**
4
+ * Extract a numeric seconds value from a `DurationOptions` entry. Returns
5
+ * `null` for entries that don't parse as a number — e.g. `'auto'`.
6
+ *
7
+ * Handles the keyword-with-unit form FAL uses for Luma/Veo (`'8s'`, `'9s'`)
8
+ * by stripping a trailing `s`. Pure-numeric strings (`'5'`, `'10'`) parse via
9
+ * Number(). Numbers pass through.
10
+ */
11
+ function entryToSeconds(entry: string | number): number | null {
12
+ if (typeof entry === 'number') {
13
+ return Number.isFinite(entry) ? entry : null
14
+ }
15
+ const stripped = entry.endsWith('s') ? entry.slice(0, -1) : entry
16
+ const parsed = Number(stripped)
17
+ return Number.isFinite(parsed) ? parsed : null
18
+ }
19
+
20
+ /**
21
+ * Snap a raw seconds value to the closest valid duration for a model's
22
+ * `DurationOptions`.
23
+ *
24
+ * - `none` → `undefined`
25
+ * - `discrete` → closest numeric-parseable entry; if none parse,
26
+ * returns `values[0]` (keyword-only models like 'auto')
27
+ * - `range` → clamped to [min, max] and rounded to `step` (default 1)
28
+ * - `mixed` → closest of (discrete numerics ∪ range values)
29
+ *
30
+ * @experimental Video generation is an experimental feature and may change.
31
+ */
32
+ export function snapToDurationOption<T extends string | number | undefined>(
33
+ seconds: number,
34
+ options: DurationOptions<T>,
35
+ ): T | undefined {
36
+ switch (options.kind) {
37
+ case 'none':
38
+ return undefined
39
+
40
+ case 'discrete': {
41
+ return pickClosestDiscrete(seconds, options.values)
42
+ }
43
+
44
+ case 'range': {
45
+ const step = options.step ?? 1
46
+ const clamped = Math.min(options.max, Math.max(options.min, seconds))
47
+ const snapped =
48
+ Math.round((clamped - options.min) / step) * step + options.min
49
+ return Math.min(options.max, Math.max(options.min, snapped)) as T
50
+ }
51
+
52
+ case 'mixed': {
53
+ const discreteCandidate = pickClosestDiscrete(seconds, options.values)
54
+ if (!options.range) return discreteCandidate
55
+
56
+ const { min, max, step = 1 } = options.range
57
+ const clamped = Math.min(max, Math.max(min, seconds))
58
+ const rangeValue = Math.min(
59
+ max,
60
+ Math.max(min, Math.round((clamped - min) / step) * step + min),
61
+ )
62
+
63
+ // Compare distance; range value is numeric, discrete may have non-numeric
64
+ // first-entry fallback (return distance Infinity for non-numerics).
65
+ const discreteSeconds =
66
+ typeof discreteCandidate === 'number'
67
+ ? discreteCandidate
68
+ : discreteCandidate !== undefined
69
+ ? (entryToSeconds(discreteCandidate) ?? Infinity)
70
+ : Infinity
71
+
72
+ return Math.abs(discreteSeconds - seconds) <=
73
+ Math.abs(rangeValue - seconds)
74
+ ? discreteCandidate
75
+ : (rangeValue as T)
76
+ }
77
+ }
78
+ }
79
+
80
+ function pickClosestDiscrete<T extends string | number>(
81
+ seconds: number,
82
+ values: ReadonlyArray<T>,
83
+ ): T | undefined {
84
+ if (values.length === 0) return undefined
85
+
86
+ let best: T | undefined
87
+ let bestDistance = Infinity
88
+ for (const value of values) {
89
+ const v = entryToSeconds(value)
90
+ if (v === null) continue
91
+ const distance = Math.abs(v - seconds)
92
+ if (distance < bestDistance) {
93
+ bestDistance = distance
94
+ best = value
95
+ }
96
+ }
97
+
98
+ // Keyword-only set (no numeric-parseable entries) — fall back to first entry.
99
+ return best ?? values[0]
100
+ }
@@ -119,6 +119,7 @@ export {
119
119
  type VideoCreateOptions,
120
120
  type VideoStatusOptions,
121
121
  type VideoUrlOptions,
122
+ type VideoDurationForAdapter,
122
123
  } from './generateVideo/index'
123
124
 
124
125
  export {
@@ -126,8 +127,11 @@ export {
126
127
  type VideoAdapter,
127
128
  type VideoAdapterConfig,
128
129
  type AnyVideoAdapter,
130
+ type DurationOptions,
129
131
  } from './generateVideo/adapter'
130
132
 
133
+ export { snapToDurationOption } from './generateVideo/snap'
134
+
131
135
  // ===========================
132
136
  // TTS Activity
133
137
  // ===========================
package/src/client.ts CHANGED
@@ -97,6 +97,10 @@ export type {
97
97
  CustomEvent,
98
98
  DocumentPart,
99
99
  ImagePart,
100
+ MediaInputMetadata,
101
+ MediaInputRole,
102
+ MediaPrompt,
103
+ MediaPromptPart,
100
104
  MessagePart,
101
105
  ModelMessage,
102
106
  RunErrorEvent,
package/src/index.ts CHANGED
@@ -118,12 +118,30 @@ export type {
118
118
  ErrorInfo,
119
119
  } from './activities/chat/middleware/index'
120
120
 
121
+ // Capability primitives + middleware builder
122
+ export {
123
+ createCapability,
124
+ defineChatMiddleware,
125
+ createChatMiddleware,
126
+ } from './activities/chat/middleware/index'
127
+ export type {
128
+ Capability,
129
+ CapabilityHandle,
130
+ CapabilityContext,
131
+ CapabilityGetter,
132
+ CapabilityProvider,
133
+ } from './activities/chat/middleware/index'
134
+
121
135
  // All types
122
136
  export * from './types'
123
137
 
124
138
  // Usage utilities
125
139
  export { buildBaseUsage, type BaseUsageInput } from './utilities/usage'
126
140
 
141
+ // Media-generation prompt resolution (used by image / video adapters)
142
+ export { resolveMediaPrompt } from './utilities/media-prompt'
143
+ export type { ResolvedMediaPrompt } from './utilities/media-prompt'
144
+
127
145
  // System prompts (type + normaliser used by adapters)
128
146
  export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts'
129
147
  export { normalizeSystemPrompts } from './system-prompts'
@@ -20,6 +20,7 @@ import type {
20
20
  ChatMiddleware,
21
21
  ChatMiddlewareContext,
22
22
  } from '../activities/chat/middleware/types'
23
+ import type { TokenUsage } from '../types'
23
24
 
24
25
  /**
25
26
  * Scope (role) of an OTel span emitted by this middleware.
@@ -179,6 +180,59 @@ function firstNumber(...candidates: Array<unknown>): number | undefined {
179
180
  return undefined
180
181
  }
181
182
 
183
+ /**
184
+ * Build the full set of `gen_ai.usage.*` span attributes from a `TokenUsage`.
185
+ *
186
+ * Beyond input/output tokens, this emits provider-reported cost, total tokens,
187
+ * cache and reasoning breakdowns, and duration-based billing — every field is
188
+ * guarded so spans stay clean when a provider doesn't report it. Cache and
189
+ * reasoning use the official GenAI semconv names; `gen_ai.usage.cost` and
190
+ * `gen_ai.usage.total_tokens` are de-facto extensions consumed by backends
191
+ * like PostHog (which otherwise re-derive cost from their own price tables,
192
+ * losing cache discounts and gateway markup). Fields with no semconv or
193
+ * de-facto convention (`costDetails`, `durationSeconds`) are
194
+ * TanStack-namespaced. Deliberately not emitted: `unitsBilled`,
195
+ * `providerUsageDetails`, and the per-modality token breakdowns — those are
196
+ * media-oriented; media-activity observability is tracked in #720.
197
+ */
198
+ function usageAttributes(usage: TokenUsage): Record<string, AttributeValue> {
199
+ const attrs: Record<string, AttributeValue> = {
200
+ 'gen_ai.usage.input_tokens': usage.promptTokens,
201
+ 'gen_ai.usage.output_tokens': usage.completionTokens,
202
+ }
203
+ const optional: Array<[key: string, value: unknown]> = [
204
+ ['gen_ai.usage.total_tokens', usage.totalTokens],
205
+ ['gen_ai.usage.cost', usage.cost],
206
+ [
207
+ 'gen_ai.usage.cache_read.input_tokens',
208
+ usage.promptTokensDetails?.cachedTokens,
209
+ ],
210
+ [
211
+ 'gen_ai.usage.cache_creation.input_tokens',
212
+ usage.promptTokensDetails?.cacheWriteTokens,
213
+ ],
214
+ [
215
+ 'gen_ai.usage.reasoning.output_tokens',
216
+ usage.completionTokensDetails?.reasoningTokens,
217
+ ],
218
+ ['tanstack.ai.usage.duration_seconds', usage.durationSeconds],
219
+ ['tanstack.ai.usage.upstream_cost', usage.costDetails?.upstreamCost],
220
+ [
221
+ 'tanstack.ai.usage.upstream_input_cost',
222
+ usage.costDetails?.upstreamInputCost,
223
+ ],
224
+ [
225
+ 'tanstack.ai.usage.upstream_output_cost',
226
+ usage.costDetails?.upstreamOutputCost,
227
+ ],
228
+ ]
229
+ for (const [key, value] of optional) {
230
+ const num = firstNumber(value)
231
+ if (num !== undefined) attrs[key] = num
232
+ }
233
+ return attrs
234
+ }
235
+
182
236
  function errorMessage(err: unknown): string | undefined {
183
237
  if (err instanceof Error) return err.message
184
238
  if (typeof err === 'string') return err
@@ -524,10 +578,7 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
524
578
  // `runOnUsage` when `chunk.usage` is present, and `onUsage` is the
525
579
  // canonical place for the metric. Recording in both would double-count.
526
580
  if (chunk.usage) {
527
- span.setAttributes({
528
- 'gen_ai.usage.input_tokens': chunk.usage.promptTokens,
529
- 'gen_ai.usage.output_tokens': chunk.usage.completionTokens,
530
- })
581
+ span.setAttributes(usageAttributes(chunk.usage))
531
582
  }
532
583
 
533
584
  if (captureContent && state.assistantTextBuffer.length > 0) {
@@ -584,10 +635,7 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
584
635
  }
585
636
 
586
637
  const span = state.currentIterationSpan ?? state.rootSpan
587
- span.setAttributes({
588
- 'gen_ai.usage.input_tokens': usage.promptTokens,
589
- 'gen_ai.usage.output_tokens': usage.completionTokens,
590
- })
638
+ span.setAttributes(usageAttributes(usage))
591
639
  })
592
640
  },
593
641
 
@@ -905,10 +953,7 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
905
953
  }
906
954
 
907
955
  if (info.usage) {
908
- state.rootSpan.setAttributes({
909
- 'gen_ai.usage.input_tokens': info.usage.promptTokens,
910
- 'gen_ai.usage.output_tokens': info.usage.completionTokens,
911
- })
956
+ state.rootSpan.setAttributes(usageAttributes(info.usage))
912
957
  }
913
958
  if (info.finishReason) {
914
959
  state.rootSpan.setAttribute('gen_ai.response.finish_reasons', [
package/src/types.ts CHANGED
@@ -1470,6 +1470,99 @@ export interface SummarizationResult {
1470
1470
  // Image Generation Types
1471
1471
  // ============================================================================
1472
1472
 
1473
+ /**
1474
+ * Optional role hint on a media input part (image / video / audio). Adapters
1475
+ * read `metadata.role` to route the part to the provider-specific request
1476
+ * field — e.g. `'mask'` → OpenAI `mask` / fal `mask_url`, `'end_frame'` → fal
1477
+ * `end_image_url`, `'reference'` → fal `reference_image_urls`. When omitted
1478
+ * the adapter falls back to positional routing.
1479
+ */
1480
+ export type MediaInputRole =
1481
+ | 'reference'
1482
+ | 'mask'
1483
+ | 'control'
1484
+ | 'start_frame'
1485
+ | 'end_frame'
1486
+ | 'character'
1487
+
1488
+ /**
1489
+ * Metadata convention for image / video / audio inputs to media generation.
1490
+ * Carried on `ImagePart.metadata` / `VideoPart.metadata` / `AudioPart.metadata`
1491
+ * when used as conditioning inputs to `generateImage()` or `generateVideo()`.
1492
+ */
1493
+ export interface MediaInputMetadata {
1494
+ /** Optional role hint disambiguating the part's intent for the adapter */
1495
+ role?: MediaInputRole
1496
+ /**
1497
+ * Optional user-defined label for this input (e.g. `'woman-in-red-dress'`).
1498
+ * **Informational only** — adapters never read it and the SDK never
1499
+ * rewrites prompt text based on it. Use it to correlate parts with the
1500
+ * references you write in your prompt using the provider's own syntax
1501
+ * (fal's `@Image1`, OpenAI's "image 1", etc.), or for your own
1502
+ * bookkeeping/logging.
1503
+ */
1504
+ tag?: string
1505
+ }
1506
+
1507
+ /**
1508
+ * A single part of a multimodal media-generation prompt. Reuses the chat
1509
+ * content-part shapes: text parts carry the instruction, image / video /
1510
+ * audio parts carry conditioning inputs (with an optional
1511
+ * `metadata.role` hint — see {@link MediaInputRole}).
1512
+ */
1513
+ export type MediaPromptPart =
1514
+ | TextPart
1515
+ | ImagePart<MediaInputMetadata>
1516
+ | VideoPart<MediaInputMetadata>
1517
+ | AudioPart<MediaInputMetadata>
1518
+
1519
+ /**
1520
+ * Prompt accepted by `generateImage()` / `generateVideo()`: a plain string,
1521
+ * or an ordered array of content parts for image-conditioned generation
1522
+ * ("not like this *(image)*, more like this *(image)*"). Part order is
1523
+ * meaningful — adapters with native multimodal prompts (Gemini, OpenRouter)
1524
+ * preserve the interleaving; named-field providers (fal, OpenAI, xAI)
1525
+ * extract the media parts and flatten the text. Text is always sent
1526
+ * verbatim: to reference inputs from the prompt, write the provider's own
1527
+ * syntax yourself (e.g. fal's `@Image1`, OpenAI's "image 1"). An array may
1528
+ * be media-only (e.g. upscalers or pure img2img endpoints that take no
1529
+ * instruction text).
1530
+ */
1531
+ export type MediaPrompt = string | Array<MediaPromptPart>
1532
+
1533
+ /**
1534
+ * Non-text modalities a media-generation model can accept in its prompt.
1535
+ */
1536
+ export type MediaPromptModality = 'image' | 'video' | 'audio'
1537
+
1538
+ /** Maps a prompt modality to its content-part type. @internal */
1539
+ interface MediaPartByModality {
1540
+ image: ImagePart<MediaInputMetadata>
1541
+ video: VideoPart<MediaInputMetadata>
1542
+ audio: AudioPart<MediaInputMetadata>
1543
+ }
1544
+
1545
+ /**
1546
+ * Prompt type narrowed to the modalities a specific model supports.
1547
+ * `MediaPromptFor<never>` (a text-only model) is `string | Array<TextPart>`;
1548
+ * `MediaPromptFor<'image'>` additionally admits image parts, etc. Used by
1549
+ * the activity option types together with the adapter's per-model input
1550
+ * modality map so unsupported parts fail at compile time.
1551
+ */
1552
+ export type MediaPromptFor<TModalities extends MediaPromptModality = never> =
1553
+ | string
1554
+ | Array<TextPart | MediaPartByModality[TModalities]>
1555
+
1556
+ /**
1557
+ * Per-model map from model name to the prompt modalities it accepts, used as
1558
+ * an adapter type parameter (`TModelInputModalitiesByName`). Models absent
1559
+ * from the map fall back to the unconstrained {@link MediaPrompt}.
1560
+ */
1561
+ export type ModelInputModalitiesByName = Record<
1562
+ string,
1563
+ ReadonlyArray<MediaPromptModality>
1564
+ >
1565
+
1473
1566
  /**
1474
1567
  * Options for image generation.
1475
1568
  * These are the common options supported across providers.
@@ -1480,8 +1573,16 @@ export interface ImageGenerationOptions<
1480
1573
  > {
1481
1574
  /** The model to use for image generation */
1482
1575
  model: string
1483
- /** Text description of the desired image(s) */
1484
- prompt: string
1576
+ /**
1577
+ * Description of the desired image(s): a plain string, or an ordered array
1578
+ * of content parts for image-conditioned generation (image-to-image,
1579
+ * reference-guided, edit, multi-reference). Media parts may carry
1580
+ * `metadata.role` to disambiguate intent (mask, control, reference, …).
1581
+ * Adapters map parts onto the provider-native request — e.g. Gemini
1582
+ * multimodal `contents`, OpenAI `images.edit()`, fal `image_url` /
1583
+ * `mask_url` — and throw a clear runtime error for unsupported modalities.
1584
+ */
1585
+ prompt: MediaPrompt
1485
1586
  /** Number of images to generate (default: 1) */
1486
1587
  numberOfImages?: number
1487
1588
  /** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
@@ -1599,15 +1700,27 @@ export interface AudioGenerationResult {
1599
1700
  export interface VideoGenerationOptions<
1600
1701
  TProviderOptions extends object = object,
1601
1702
  TSize extends string | undefined = string,
1703
+ TDuration extends string | number | undefined = number,
1602
1704
  > {
1603
1705
  /** The model to use for video generation */
1604
1706
  model: string
1605
- /** Text description of the desired video */
1606
- prompt: string
1707
+ /**
1708
+ * Description of the desired video: a plain string, or an ordered array of
1709
+ * content parts for image-conditioned generation. Image parts may carry
1710
+ * `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
1711
+ * 'character'`) to disambiguate intent; adapters route them onto the
1712
+ * provider-native request (e.g. OpenAI Sora `input_reference`, fal
1713
+ * `image_url` / `end_image_url`) and throw at runtime if unsupported.
1714
+ */
1715
+ prompt: MediaPrompt
1607
1716
  /** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
1608
1717
  size?: TSize
1609
- /** Video duration in seconds */
1610
- duration?: number
1718
+ /**
1719
+ * Video duration in seconds. Adapters that declare a per-model duration
1720
+ * map narrow this to the model's valid union; use
1721
+ * `adapter.snapDuration(seconds)` to coerce raw seconds to a valid value.
1722
+ */
1723
+ duration?: TDuration
1611
1724
  /** Model-specific options for video generation */
1612
1725
  modelOptions?: TProviderOptions
1613
1726
  /**