@tanstack/ai 0.29.0 → 0.32.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +10 -0
- package/dist/esm/activities/chat/adapter.js +1 -0
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/index.d.ts +3 -2
- package/dist/esm/activities/chat/index.js +13 -1
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/middleware/builder.d.ts +46 -0
- package/dist/esm/activities/chat/middleware/builder.js +17 -0
- package/dist/esm/activities/chat/middleware/builder.js.map +1 -0
- package/dist/esm/activities/chat/middleware/capabilities.d.ts +93 -0
- package/dist/esm/activities/chat/middleware/capabilities.js +45 -0
- package/dist/esm/activities/chat/middleware/capabilities.js.map +1 -0
- package/dist/esm/activities/chat/middleware/compose.d.ts +10 -0
- package/dist/esm/activities/chat/middleware/compose.js +47 -0
- package/dist/esm/activities/chat/middleware/compose.js.map +1 -1
- package/dist/esm/activities/chat/middleware/define.d.ts +20 -0
- package/dist/esm/activities/chat/middleware/define.js +7 -0
- package/dist/esm/activities/chat/middleware/define.js.map +1 -0
- package/dist/esm/activities/chat/middleware/index.d.ts +8 -0
- package/dist/esm/activities/chat/middleware/types.d.ts +51 -0
- package/dist/esm/activities/chat/middleware/validate.d.ts +19 -0
- package/dist/esm/activities/chat/middleware/validate.js +30 -0
- package/dist/esm/activities/chat/middleware/validate.js.map +1 -0
- package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
- package/dist/esm/activities/generateImage/adapter.js.map +1 -1
- package/dist/esm/activities/generateImage/index.d.ts +19 -3
- package/dist/esm/activities/generateImage/index.js +12 -1
- package/dist/esm/activities/generateImage/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
- package/dist/esm/activities/generateVideo/adapter.js +14 -0
- package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
- package/dist/esm/activities/generateVideo/index.d.ts +31 -5
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
- package/dist/esm/activities/generateVideo/snap.js +54 -0
- package/dist/esm/activities/generateVideo/snap.js.map +1 -0
- package/dist/esm/activities/index.d.ts +3 -2
- package/dist/esm/activities/index.js +2 -0
- package/dist/esm/activities/index.js.map +1 -1
- package/dist/esm/client.d.ts +1 -1
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -0
- package/dist/esm/index.js +8 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/middlewares/otel.js +40 -12
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/types.d.ts +96 -7
- package/dist/esm/utilities/media-prompt.d.ts +35 -0
- package/dist/esm/utilities/media-prompt.js +43 -0
- package/dist/esm/utilities/media-prompt.js.map +1 -0
- package/package.json +10 -2
- package/skills/ai-core/media-generation/SKILL.md +173 -3
- package/src/activities/chat/adapter.ts +11 -0
- package/src/activities/chat/index.ts +20 -1
- package/src/activities/chat/middleware/builder.ts +109 -0
- package/src/activities/chat/middleware/capabilities.ts +162 -0
- package/src/activities/chat/middleware/compose.ts +51 -0
- package/src/activities/chat/middleware/define.ts +34 -0
- package/src/activities/chat/middleware/index.ts +20 -0
- package/src/activities/chat/middleware/types.ts +60 -0
- package/src/activities/chat/middleware/validate.ts +55 -0
- package/src/activities/generateImage/adapter.ts +16 -3
- package/src/activities/generateImage/index.ts +48 -4
- package/src/activities/generateVideo/adapter.ts +80 -4
- package/src/activities/generateVideo/index.ts +53 -4
- package/src/activities/generateVideo/snap.ts +100 -0
- package/src/activities/index.ts +4 -0
- package/src/client.ts +4 -0
- package/src/index.ts +18 -0
- package/src/middlewares/otel.ts +57 -12
- package/src/types.ts +119 -6
- package/src/utilities/media-prompt.ts +86 -0
|
@@ -1,10 +1,30 @@
|
|
|
1
1
|
import type {
|
|
2
|
+
ModelInputModalitiesByName,
|
|
2
3
|
VideoGenerationOptions,
|
|
3
4
|
VideoJobResult,
|
|
4
5
|
VideoStatusResult,
|
|
5
6
|
VideoUrlResult,
|
|
6
7
|
} from '../../types'
|
|
7
8
|
|
|
9
|
+
/**
|
|
10
|
+
* Structured description of the durations a video model accepts.
|
|
11
|
+
*
|
|
12
|
+
* Tagged union so the same shape can express discrete enums (OpenAI Sora,
|
|
13
|
+
* Veo), continuous ranges, mixed shapes, and models with no duration field.
|
|
14
|
+
* Consumed by `VideoAdapter.availableDurations()`.
|
|
15
|
+
*
|
|
16
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
17
|
+
*/
|
|
18
|
+
export type DurationOptions<T extends string | number | undefined> =
|
|
19
|
+
| { kind: 'discrete'; values: ReadonlyArray<NonNullable<T>> }
|
|
20
|
+
| { kind: 'range'; min: number; max: number; step?: number; unit: 'seconds' }
|
|
21
|
+
| {
|
|
22
|
+
kind: 'mixed'
|
|
23
|
+
values: ReadonlyArray<NonNullable<T>>
|
|
24
|
+
range?: { min: number; max: number; step?: number }
|
|
25
|
+
}
|
|
26
|
+
| { kind: 'none' }
|
|
27
|
+
|
|
8
28
|
/**
|
|
9
29
|
* Configuration for video adapter instances
|
|
10
30
|
*
|
|
@@ -31,6 +51,11 @@ export interface VideoAdapterConfig {
|
|
|
31
51
|
* - TProviderOptions: Provider-specific options (already resolved)
|
|
32
52
|
* - TModelProviderOptionsByName: Map from model name to its specific provider options
|
|
33
53
|
* - TModelSizeByName: Map from model name to its supported sizes
|
|
54
|
+
* - TModelInputModalitiesByName: Map from model name to the non-text prompt
|
|
55
|
+
* modalities it accepts (constrains the `prompt` part types at compile time)
|
|
56
|
+
* - TModelDurationByName: Map from model name to its supported duration
|
|
57
|
+
* union. Defaults to `Record<string, number>` so adapters that haven't
|
|
58
|
+
* declared a map keep today's `duration?: number` typing.
|
|
34
59
|
*/
|
|
35
60
|
export interface VideoAdapter<
|
|
36
61
|
TModel extends string = string,
|
|
@@ -40,6 +65,10 @@ export interface VideoAdapter<
|
|
|
40
65
|
string,
|
|
41
66
|
string
|
|
42
67
|
>,
|
|
68
|
+
TModelInputModalitiesByName extends ModelInputModalitiesByName =
|
|
69
|
+
ModelInputModalitiesByName,
|
|
70
|
+
TModelDurationByName extends Record<string, string | number | undefined> =
|
|
71
|
+
Record<string, number>,
|
|
43
72
|
> {
|
|
44
73
|
/** Discriminator for adapter kind - used to determine API shape */
|
|
45
74
|
readonly kind: 'video'
|
|
@@ -55,6 +84,8 @@ export interface VideoAdapter<
|
|
|
55
84
|
providerOptions: TProviderOptions
|
|
56
85
|
modelProviderOptionsByName: TModelProviderOptionsByName
|
|
57
86
|
modelSizeByName: TModelSizeByName
|
|
87
|
+
modelInputModalitiesByName: TModelInputModalitiesByName
|
|
88
|
+
modelDurationByName: TModelDurationByName
|
|
58
89
|
}
|
|
59
90
|
|
|
60
91
|
/**
|
|
@@ -62,7 +93,11 @@ export interface VideoAdapter<
|
|
|
62
93
|
* Returns a job ID that can be used to poll for status and retrieve the video.
|
|
63
94
|
*/
|
|
64
95
|
createVideoJob: (
|
|
65
|
-
options: VideoGenerationOptions<
|
|
96
|
+
options: VideoGenerationOptions<
|
|
97
|
+
TProviderOptions,
|
|
98
|
+
TModelSizeByName[TModel],
|
|
99
|
+
TModelDurationByName[TModel]
|
|
100
|
+
>,
|
|
66
101
|
) => Promise<VideoJobResult>
|
|
67
102
|
|
|
68
103
|
/**
|
|
@@ -75,13 +110,26 @@ export interface VideoAdapter<
|
|
|
75
110
|
* Should only be called after status is 'completed'.
|
|
76
111
|
*/
|
|
77
112
|
getVideoUrl: (jobId: string) => Promise<VideoUrlResult>
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Describe the durations this adapter's model accepts. Returns a tagged
|
|
116
|
+
* union so consumers can render UI / coerce input without provider-specific
|
|
117
|
+
* knowledge.
|
|
118
|
+
*/
|
|
119
|
+
availableDurations: () => DurationOptions<TModelDurationByName[TModel]>
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Coerce a raw seconds value to the closest valid duration for this model.
|
|
123
|
+
* Returns `undefined` for models with no duration field.
|
|
124
|
+
*/
|
|
125
|
+
snapDuration: (seconds: number) => TModelDurationByName[TModel] | undefined
|
|
78
126
|
}
|
|
79
127
|
|
|
80
128
|
/**
|
|
81
129
|
* A VideoAdapter with any/unknown type parameters.
|
|
82
130
|
* Useful as a constraint in generic functions and interfaces.
|
|
83
131
|
*/
|
|
84
|
-
export type AnyVideoAdapter = VideoAdapter<any, any, any, any>
|
|
132
|
+
export type AnyVideoAdapter = VideoAdapter<any, any, any, any, any, any>
|
|
85
133
|
|
|
86
134
|
/**
|
|
87
135
|
* Abstract base class for video generation adapters.
|
|
@@ -99,11 +147,17 @@ export abstract class BaseVideoAdapter<
|
|
|
99
147
|
string,
|
|
100
148
|
string
|
|
101
149
|
>,
|
|
150
|
+
TModelInputModalitiesByName extends ModelInputModalitiesByName =
|
|
151
|
+
ModelInputModalitiesByName,
|
|
152
|
+
TModelDurationByName extends Record<string, string | number | undefined> =
|
|
153
|
+
Record<string, number>,
|
|
102
154
|
> implements VideoAdapter<
|
|
103
155
|
TModel,
|
|
104
156
|
TProviderOptions,
|
|
105
157
|
TModelProviderOptionsByName,
|
|
106
|
-
TModelSizeByName
|
|
158
|
+
TModelSizeByName,
|
|
159
|
+
TModelInputModalitiesByName,
|
|
160
|
+
TModelDurationByName
|
|
107
161
|
> {
|
|
108
162
|
readonly kind = 'video' as const
|
|
109
163
|
abstract readonly name: string
|
|
@@ -114,6 +168,8 @@ export abstract class BaseVideoAdapter<
|
|
|
114
168
|
providerOptions: TProviderOptions
|
|
115
169
|
modelProviderOptionsByName: TModelProviderOptionsByName
|
|
116
170
|
modelSizeByName: TModelSizeByName
|
|
171
|
+
modelInputModalitiesByName: TModelInputModalitiesByName
|
|
172
|
+
modelDurationByName: TModelDurationByName
|
|
117
173
|
}
|
|
118
174
|
|
|
119
175
|
protected config: VideoAdapterConfig
|
|
@@ -124,13 +180,33 @@ export abstract class BaseVideoAdapter<
|
|
|
124
180
|
}
|
|
125
181
|
|
|
126
182
|
abstract createVideoJob(
|
|
127
|
-
options: VideoGenerationOptions<
|
|
183
|
+
options: VideoGenerationOptions<
|
|
184
|
+
TProviderOptions,
|
|
185
|
+
TModelSizeByName[TModel],
|
|
186
|
+
TModelDurationByName[TModel]
|
|
187
|
+
>,
|
|
128
188
|
): Promise<VideoJobResult>
|
|
129
189
|
|
|
130
190
|
abstract getVideoStatus(jobId: string): Promise<VideoStatusResult>
|
|
131
191
|
|
|
132
192
|
abstract getVideoUrl(jobId: string): Promise<VideoUrlResult>
|
|
133
193
|
|
|
194
|
+
/**
|
|
195
|
+
* Default implementation returns `{ kind: 'none' }`. Adapters that have
|
|
196
|
+
* declared their per-model duration map should override this.
|
|
197
|
+
*/
|
|
198
|
+
availableDurations(): DurationOptions<TModelDurationByName[TModel]> {
|
|
199
|
+
return { kind: 'none' }
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Default implementation returns `undefined`. Adapters that have declared
|
|
204
|
+
* their per-model duration map should override.
|
|
205
|
+
*/
|
|
206
|
+
snapDuration(_seconds: number): TModelDurationByName[TModel] | undefined {
|
|
207
|
+
return undefined
|
|
208
|
+
}
|
|
209
|
+
|
|
134
210
|
protected generateId(): string {
|
|
135
211
|
return `${this.name}-${Date.now()}-${Math.random().toString(36).substring(7)}`
|
|
136
212
|
}
|
|
@@ -14,6 +14,8 @@ import type { InternalLogger } from '../../logger/internal-logger'
|
|
|
14
14
|
import type { DebugOption } from '../../logger/types'
|
|
15
15
|
import type { VideoAdapter } from './adapter'
|
|
16
16
|
import type {
|
|
17
|
+
MediaPrompt,
|
|
18
|
+
MediaPromptFor,
|
|
17
19
|
StreamChunk,
|
|
18
20
|
TokenUsage,
|
|
19
21
|
VideoJobResult,
|
|
@@ -50,6 +52,40 @@ export type VideoSizeForAdapter<TAdapter> =
|
|
|
50
52
|
: string
|
|
51
53
|
: string
|
|
52
54
|
|
|
55
|
+
/**
|
|
56
|
+
* Extract the prompt type a model accepts from a VideoAdapter via ~types.
|
|
57
|
+
* Mirrors `ImagePromptForModel`: models in the adapter's input-modality map
|
|
58
|
+
* get a `prompt` narrowed to text + their supported part types; adapters
|
|
59
|
+
* without a map fall back to the full MediaPrompt.
|
|
60
|
+
*/
|
|
61
|
+
export type VideoPromptForAdapter<TAdapter> =
|
|
62
|
+
TAdapter extends VideoAdapter<infer TModel, any, any, any, infer ModsByName>
|
|
63
|
+
? string extends keyof ModsByName
|
|
64
|
+
? MediaPrompt
|
|
65
|
+
: TModel extends keyof ModsByName
|
|
66
|
+
? MediaPromptFor<ModsByName[TModel][number]>
|
|
67
|
+
: MediaPrompt
|
|
68
|
+
: MediaPrompt
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Extract the duration type for a VideoAdapter's model via ~types.
|
|
72
|
+
* Mirrors `VideoSizeForAdapter`. Falls back to `number` for adapters that
|
|
73
|
+
* haven't declared per-model duration constraints.
|
|
74
|
+
*/
|
|
75
|
+
export type VideoDurationForAdapter<TAdapter> =
|
|
76
|
+
TAdapter extends VideoAdapter<
|
|
77
|
+
infer TModel,
|
|
78
|
+
any,
|
|
79
|
+
any,
|
|
80
|
+
any,
|
|
81
|
+
any,
|
|
82
|
+
infer TDurationMap
|
|
83
|
+
>
|
|
84
|
+
? TModel extends keyof TDurationMap
|
|
85
|
+
? TDurationMap[TModel]
|
|
86
|
+
: number
|
|
87
|
+
: number
|
|
88
|
+
|
|
53
89
|
// ===========================
|
|
54
90
|
// Activity Options Types
|
|
55
91
|
|
|
@@ -84,12 +120,25 @@ export type VideoCreateOptions<
|
|
|
84
120
|
> = VideoActivityBaseOptions<TAdapter> & {
|
|
85
121
|
/** Request type - create a new job (default if not specified) */
|
|
86
122
|
request?: 'create'
|
|
87
|
-
/**
|
|
88
|
-
|
|
123
|
+
/**
|
|
124
|
+
* Description of the desired video. Either a plain string, or — for models
|
|
125
|
+
* that support image-conditioned generation — an ordered array of content
|
|
126
|
+
* parts interleaving text with image inputs. Image parts may carry
|
|
127
|
+
* `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
|
|
128
|
+
* 'character'`) to disambiguate intent; positional fallback otherwise. The
|
|
129
|
+
* accepted part types are narrowed per model via the adapter's
|
|
130
|
+
* input-modality map.
|
|
131
|
+
*/
|
|
132
|
+
prompt: VideoPromptForAdapter<TAdapter>
|
|
89
133
|
/** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
|
|
90
134
|
size?: VideoSizeForAdapter<TAdapter>
|
|
91
|
-
/**
|
|
92
|
-
|
|
135
|
+
/**
|
|
136
|
+
* Video duration in seconds. Adapters that declare a per-model duration
|
|
137
|
+
* map narrow this to the model's valid union (e.g. `4 | 6 | 8` for Veo 3).
|
|
138
|
+
* Pass `adapter.snapDuration(seconds)` to coerce raw seconds to a valid
|
|
139
|
+
* value.
|
|
140
|
+
*/
|
|
141
|
+
duration?: VideoDurationForAdapter<TAdapter>
|
|
93
142
|
/**
|
|
94
143
|
* Whether to stream the video generation lifecycle.
|
|
95
144
|
* When true, returns an AsyncIterable<StreamChunk> that handles the full
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
import type { DurationOptions } from './adapter'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Extract a numeric seconds value from a `DurationOptions` entry. Returns
|
|
5
|
+
* `null` for entries that don't parse as a number — e.g. `'auto'`.
|
|
6
|
+
*
|
|
7
|
+
* Handles the keyword-with-unit form FAL uses for Luma/Veo (`'8s'`, `'9s'`)
|
|
8
|
+
* by stripping a trailing `s`. Pure-numeric strings (`'5'`, `'10'`) parse via
|
|
9
|
+
* Number(). Numbers pass through.
|
|
10
|
+
*/
|
|
11
|
+
function entryToSeconds(entry: string | number): number | null {
|
|
12
|
+
if (typeof entry === 'number') {
|
|
13
|
+
return Number.isFinite(entry) ? entry : null
|
|
14
|
+
}
|
|
15
|
+
const stripped = entry.endsWith('s') ? entry.slice(0, -1) : entry
|
|
16
|
+
const parsed = Number(stripped)
|
|
17
|
+
return Number.isFinite(parsed) ? parsed : null
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Snap a raw seconds value to the closest valid duration for a model's
|
|
22
|
+
* `DurationOptions`.
|
|
23
|
+
*
|
|
24
|
+
* - `none` → `undefined`
|
|
25
|
+
* - `discrete` → closest numeric-parseable entry; if none parse,
|
|
26
|
+
* returns `values[0]` (keyword-only models like 'auto')
|
|
27
|
+
* - `range` → clamped to [min, max] and rounded to `step` (default 1)
|
|
28
|
+
* - `mixed` → closest of (discrete numerics ∪ range values)
|
|
29
|
+
*
|
|
30
|
+
* @experimental Video generation is an experimental feature and may change.
|
|
31
|
+
*/
|
|
32
|
+
export function snapToDurationOption<T extends string | number | undefined>(
|
|
33
|
+
seconds: number,
|
|
34
|
+
options: DurationOptions<T>,
|
|
35
|
+
): T | undefined {
|
|
36
|
+
switch (options.kind) {
|
|
37
|
+
case 'none':
|
|
38
|
+
return undefined
|
|
39
|
+
|
|
40
|
+
case 'discrete': {
|
|
41
|
+
return pickClosestDiscrete(seconds, options.values)
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
case 'range': {
|
|
45
|
+
const step = options.step ?? 1
|
|
46
|
+
const clamped = Math.min(options.max, Math.max(options.min, seconds))
|
|
47
|
+
const snapped =
|
|
48
|
+
Math.round((clamped - options.min) / step) * step + options.min
|
|
49
|
+
return Math.min(options.max, Math.max(options.min, snapped)) as T
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
case 'mixed': {
|
|
53
|
+
const discreteCandidate = pickClosestDiscrete(seconds, options.values)
|
|
54
|
+
if (!options.range) return discreteCandidate
|
|
55
|
+
|
|
56
|
+
const { min, max, step = 1 } = options.range
|
|
57
|
+
const clamped = Math.min(max, Math.max(min, seconds))
|
|
58
|
+
const rangeValue = Math.min(
|
|
59
|
+
max,
|
|
60
|
+
Math.max(min, Math.round((clamped - min) / step) * step + min),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
// Compare distance; range value is numeric, discrete may have non-numeric
|
|
64
|
+
// first-entry fallback (return distance Infinity for non-numerics).
|
|
65
|
+
const discreteSeconds =
|
|
66
|
+
typeof discreteCandidate === 'number'
|
|
67
|
+
? discreteCandidate
|
|
68
|
+
: discreteCandidate !== undefined
|
|
69
|
+
? (entryToSeconds(discreteCandidate) ?? Infinity)
|
|
70
|
+
: Infinity
|
|
71
|
+
|
|
72
|
+
return Math.abs(discreteSeconds - seconds) <=
|
|
73
|
+
Math.abs(rangeValue - seconds)
|
|
74
|
+
? discreteCandidate
|
|
75
|
+
: (rangeValue as T)
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function pickClosestDiscrete<T extends string | number>(
|
|
81
|
+
seconds: number,
|
|
82
|
+
values: ReadonlyArray<T>,
|
|
83
|
+
): T | undefined {
|
|
84
|
+
if (values.length === 0) return undefined
|
|
85
|
+
|
|
86
|
+
let best: T | undefined
|
|
87
|
+
let bestDistance = Infinity
|
|
88
|
+
for (const value of values) {
|
|
89
|
+
const v = entryToSeconds(value)
|
|
90
|
+
if (v === null) continue
|
|
91
|
+
const distance = Math.abs(v - seconds)
|
|
92
|
+
if (distance < bestDistance) {
|
|
93
|
+
bestDistance = distance
|
|
94
|
+
best = value
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// Keyword-only set (no numeric-parseable entries) — fall back to first entry.
|
|
99
|
+
return best ?? values[0]
|
|
100
|
+
}
|
package/src/activities/index.ts
CHANGED
|
@@ -119,6 +119,7 @@ export {
|
|
|
119
119
|
type VideoCreateOptions,
|
|
120
120
|
type VideoStatusOptions,
|
|
121
121
|
type VideoUrlOptions,
|
|
122
|
+
type VideoDurationForAdapter,
|
|
122
123
|
} from './generateVideo/index'
|
|
123
124
|
|
|
124
125
|
export {
|
|
@@ -126,8 +127,11 @@ export {
|
|
|
126
127
|
type VideoAdapter,
|
|
127
128
|
type VideoAdapterConfig,
|
|
128
129
|
type AnyVideoAdapter,
|
|
130
|
+
type DurationOptions,
|
|
129
131
|
} from './generateVideo/adapter'
|
|
130
132
|
|
|
133
|
+
export { snapToDurationOption } from './generateVideo/snap'
|
|
134
|
+
|
|
131
135
|
// ===========================
|
|
132
136
|
// TTS Activity
|
|
133
137
|
// ===========================
|
package/src/client.ts
CHANGED
package/src/index.ts
CHANGED
|
@@ -118,12 +118,30 @@ export type {
|
|
|
118
118
|
ErrorInfo,
|
|
119
119
|
} from './activities/chat/middleware/index'
|
|
120
120
|
|
|
121
|
+
// Capability primitives + middleware builder
|
|
122
|
+
export {
|
|
123
|
+
createCapability,
|
|
124
|
+
defineChatMiddleware,
|
|
125
|
+
createChatMiddleware,
|
|
126
|
+
} from './activities/chat/middleware/index'
|
|
127
|
+
export type {
|
|
128
|
+
Capability,
|
|
129
|
+
CapabilityHandle,
|
|
130
|
+
CapabilityContext,
|
|
131
|
+
CapabilityGetter,
|
|
132
|
+
CapabilityProvider,
|
|
133
|
+
} from './activities/chat/middleware/index'
|
|
134
|
+
|
|
121
135
|
// All types
|
|
122
136
|
export * from './types'
|
|
123
137
|
|
|
124
138
|
// Usage utilities
|
|
125
139
|
export { buildBaseUsage, type BaseUsageInput } from './utilities/usage'
|
|
126
140
|
|
|
141
|
+
// Media-generation prompt resolution (used by image / video adapters)
|
|
142
|
+
export { resolveMediaPrompt } from './utilities/media-prompt'
|
|
143
|
+
export type { ResolvedMediaPrompt } from './utilities/media-prompt'
|
|
144
|
+
|
|
127
145
|
// System prompts (type + normaliser used by adapters)
|
|
128
146
|
export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts'
|
|
129
147
|
export { normalizeSystemPrompts } from './system-prompts'
|
package/src/middlewares/otel.ts
CHANGED
|
@@ -20,6 +20,7 @@ import type {
|
|
|
20
20
|
ChatMiddleware,
|
|
21
21
|
ChatMiddlewareContext,
|
|
22
22
|
} from '../activities/chat/middleware/types'
|
|
23
|
+
import type { TokenUsage } from '../types'
|
|
23
24
|
|
|
24
25
|
/**
|
|
25
26
|
* Scope (role) of an OTel span emitted by this middleware.
|
|
@@ -179,6 +180,59 @@ function firstNumber(...candidates: Array<unknown>): number | undefined {
|
|
|
179
180
|
return undefined
|
|
180
181
|
}
|
|
181
182
|
|
|
183
|
+
/**
|
|
184
|
+
* Build the full set of `gen_ai.usage.*` span attributes from a `TokenUsage`.
|
|
185
|
+
*
|
|
186
|
+
* Beyond input/output tokens, this emits provider-reported cost, total tokens,
|
|
187
|
+
* cache and reasoning breakdowns, and duration-based billing — every field is
|
|
188
|
+
* guarded so spans stay clean when a provider doesn't report it. Cache and
|
|
189
|
+
* reasoning use the official GenAI semconv names; `gen_ai.usage.cost` and
|
|
190
|
+
* `gen_ai.usage.total_tokens` are de-facto extensions consumed by backends
|
|
191
|
+
* like PostHog (which otherwise re-derive cost from their own price tables,
|
|
192
|
+
* losing cache discounts and gateway markup). Fields with no semconv or
|
|
193
|
+
* de-facto convention (`costDetails`, `durationSeconds`) are
|
|
194
|
+
* TanStack-namespaced. Deliberately not emitted: `unitsBilled`,
|
|
195
|
+
* `providerUsageDetails`, and the per-modality token breakdowns — those are
|
|
196
|
+
* media-oriented; media-activity observability is tracked in #720.
|
|
197
|
+
*/
|
|
198
|
+
function usageAttributes(usage: TokenUsage): Record<string, AttributeValue> {
|
|
199
|
+
const attrs: Record<string, AttributeValue> = {
|
|
200
|
+
'gen_ai.usage.input_tokens': usage.promptTokens,
|
|
201
|
+
'gen_ai.usage.output_tokens': usage.completionTokens,
|
|
202
|
+
}
|
|
203
|
+
const optional: Array<[key: string, value: unknown]> = [
|
|
204
|
+
['gen_ai.usage.total_tokens', usage.totalTokens],
|
|
205
|
+
['gen_ai.usage.cost', usage.cost],
|
|
206
|
+
[
|
|
207
|
+
'gen_ai.usage.cache_read.input_tokens',
|
|
208
|
+
usage.promptTokensDetails?.cachedTokens,
|
|
209
|
+
],
|
|
210
|
+
[
|
|
211
|
+
'gen_ai.usage.cache_creation.input_tokens',
|
|
212
|
+
usage.promptTokensDetails?.cacheWriteTokens,
|
|
213
|
+
],
|
|
214
|
+
[
|
|
215
|
+
'gen_ai.usage.reasoning.output_tokens',
|
|
216
|
+
usage.completionTokensDetails?.reasoningTokens,
|
|
217
|
+
],
|
|
218
|
+
['tanstack.ai.usage.duration_seconds', usage.durationSeconds],
|
|
219
|
+
['tanstack.ai.usage.upstream_cost', usage.costDetails?.upstreamCost],
|
|
220
|
+
[
|
|
221
|
+
'tanstack.ai.usage.upstream_input_cost',
|
|
222
|
+
usage.costDetails?.upstreamInputCost,
|
|
223
|
+
],
|
|
224
|
+
[
|
|
225
|
+
'tanstack.ai.usage.upstream_output_cost',
|
|
226
|
+
usage.costDetails?.upstreamOutputCost,
|
|
227
|
+
],
|
|
228
|
+
]
|
|
229
|
+
for (const [key, value] of optional) {
|
|
230
|
+
const num = firstNumber(value)
|
|
231
|
+
if (num !== undefined) attrs[key] = num
|
|
232
|
+
}
|
|
233
|
+
return attrs
|
|
234
|
+
}
|
|
235
|
+
|
|
182
236
|
function errorMessage(err: unknown): string | undefined {
|
|
183
237
|
if (err instanceof Error) return err.message
|
|
184
238
|
if (typeof err === 'string') return err
|
|
@@ -524,10 +578,7 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
|
|
|
524
578
|
// `runOnUsage` when `chunk.usage` is present, and `onUsage` is the
|
|
525
579
|
// canonical place for the metric. Recording in both would double-count.
|
|
526
580
|
if (chunk.usage) {
|
|
527
|
-
span.setAttributes(
|
|
528
|
-
'gen_ai.usage.input_tokens': chunk.usage.promptTokens,
|
|
529
|
-
'gen_ai.usage.output_tokens': chunk.usage.completionTokens,
|
|
530
|
-
})
|
|
581
|
+
span.setAttributes(usageAttributes(chunk.usage))
|
|
531
582
|
}
|
|
532
583
|
|
|
533
584
|
if (captureContent && state.assistantTextBuffer.length > 0) {
|
|
@@ -584,10 +635,7 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
|
|
|
584
635
|
}
|
|
585
636
|
|
|
586
637
|
const span = state.currentIterationSpan ?? state.rootSpan
|
|
587
|
-
span.setAttributes(
|
|
588
|
-
'gen_ai.usage.input_tokens': usage.promptTokens,
|
|
589
|
-
'gen_ai.usage.output_tokens': usage.completionTokens,
|
|
590
|
-
})
|
|
638
|
+
span.setAttributes(usageAttributes(usage))
|
|
591
639
|
})
|
|
592
640
|
},
|
|
593
641
|
|
|
@@ -905,10 +953,7 @@ export function otelMiddleware(options: OtelMiddlewareOptions): ChatMiddleware {
|
|
|
905
953
|
}
|
|
906
954
|
|
|
907
955
|
if (info.usage) {
|
|
908
|
-
state.rootSpan.setAttributes(
|
|
909
|
-
'gen_ai.usage.input_tokens': info.usage.promptTokens,
|
|
910
|
-
'gen_ai.usage.output_tokens': info.usage.completionTokens,
|
|
911
|
-
})
|
|
956
|
+
state.rootSpan.setAttributes(usageAttributes(info.usage))
|
|
912
957
|
}
|
|
913
958
|
if (info.finishReason) {
|
|
914
959
|
state.rootSpan.setAttribute('gen_ai.response.finish_reasons', [
|
package/src/types.ts
CHANGED
|
@@ -1470,6 +1470,99 @@ export interface SummarizationResult {
|
|
|
1470
1470
|
// Image Generation Types
|
|
1471
1471
|
// ============================================================================
|
|
1472
1472
|
|
|
1473
|
+
/**
|
|
1474
|
+
* Optional role hint on a media input part (image / video / audio). Adapters
|
|
1475
|
+
* read `metadata.role` to route the part to the provider-specific request
|
|
1476
|
+
* field — e.g. `'mask'` → OpenAI `mask` / fal `mask_url`, `'end_frame'` → fal
|
|
1477
|
+
* `end_image_url`, `'reference'` → fal `reference_image_urls`. When omitted
|
|
1478
|
+
* the adapter falls back to positional routing.
|
|
1479
|
+
*/
|
|
1480
|
+
export type MediaInputRole =
|
|
1481
|
+
| 'reference'
|
|
1482
|
+
| 'mask'
|
|
1483
|
+
| 'control'
|
|
1484
|
+
| 'start_frame'
|
|
1485
|
+
| 'end_frame'
|
|
1486
|
+
| 'character'
|
|
1487
|
+
|
|
1488
|
+
/**
|
|
1489
|
+
* Metadata convention for image / video / audio inputs to media generation.
|
|
1490
|
+
* Carried on `ImagePart.metadata` / `VideoPart.metadata` / `AudioPart.metadata`
|
|
1491
|
+
* when used as conditioning inputs to `generateImage()` or `generateVideo()`.
|
|
1492
|
+
*/
|
|
1493
|
+
export interface MediaInputMetadata {
|
|
1494
|
+
/** Optional role hint disambiguating the part's intent for the adapter */
|
|
1495
|
+
role?: MediaInputRole
|
|
1496
|
+
/**
|
|
1497
|
+
* Optional user-defined label for this input (e.g. `'woman-in-red-dress'`).
|
|
1498
|
+
* **Informational only** — adapters never read it and the SDK never
|
|
1499
|
+
* rewrites prompt text based on it. Use it to correlate parts with the
|
|
1500
|
+
* references you write in your prompt using the provider's own syntax
|
|
1501
|
+
* (fal's `@Image1`, OpenAI's "image 1", etc.), or for your own
|
|
1502
|
+
* bookkeeping/logging.
|
|
1503
|
+
*/
|
|
1504
|
+
tag?: string
|
|
1505
|
+
}
|
|
1506
|
+
|
|
1507
|
+
/**
|
|
1508
|
+
* A single part of a multimodal media-generation prompt. Reuses the chat
|
|
1509
|
+
* content-part shapes: text parts carry the instruction, image / video /
|
|
1510
|
+
* audio parts carry conditioning inputs (with an optional
|
|
1511
|
+
* `metadata.role` hint — see {@link MediaInputRole}).
|
|
1512
|
+
*/
|
|
1513
|
+
export type MediaPromptPart =
|
|
1514
|
+
| TextPart
|
|
1515
|
+
| ImagePart<MediaInputMetadata>
|
|
1516
|
+
| VideoPart<MediaInputMetadata>
|
|
1517
|
+
| AudioPart<MediaInputMetadata>
|
|
1518
|
+
|
|
1519
|
+
/**
|
|
1520
|
+
* Prompt accepted by `generateImage()` / `generateVideo()`: a plain string,
|
|
1521
|
+
* or an ordered array of content parts for image-conditioned generation
|
|
1522
|
+
* ("not like this *(image)*, more like this *(image)*"). Part order is
|
|
1523
|
+
* meaningful — adapters with native multimodal prompts (Gemini, OpenRouter)
|
|
1524
|
+
* preserve the interleaving; named-field providers (fal, OpenAI, xAI)
|
|
1525
|
+
* extract the media parts and flatten the text. Text is always sent
|
|
1526
|
+
* verbatim: to reference inputs from the prompt, write the provider's own
|
|
1527
|
+
* syntax yourself (e.g. fal's `@Image1`, OpenAI's "image 1"). An array may
|
|
1528
|
+
* be media-only (e.g. upscalers or pure img2img endpoints that take no
|
|
1529
|
+
* instruction text).
|
|
1530
|
+
*/
|
|
1531
|
+
export type MediaPrompt = string | Array<MediaPromptPart>
|
|
1532
|
+
|
|
1533
|
+
/**
|
|
1534
|
+
* Non-text modalities a media-generation model can accept in its prompt.
|
|
1535
|
+
*/
|
|
1536
|
+
export type MediaPromptModality = 'image' | 'video' | 'audio'
|
|
1537
|
+
|
|
1538
|
+
/** Maps a prompt modality to its content-part type. @internal */
|
|
1539
|
+
interface MediaPartByModality {
|
|
1540
|
+
image: ImagePart<MediaInputMetadata>
|
|
1541
|
+
video: VideoPart<MediaInputMetadata>
|
|
1542
|
+
audio: AudioPart<MediaInputMetadata>
|
|
1543
|
+
}
|
|
1544
|
+
|
|
1545
|
+
/**
|
|
1546
|
+
* Prompt type narrowed to the modalities a specific model supports.
|
|
1547
|
+
* `MediaPromptFor<never>` (a text-only model) is `string | Array<TextPart>`;
|
|
1548
|
+
* `MediaPromptFor<'image'>` additionally admits image parts, etc. Used by
|
|
1549
|
+
* the activity option types together with the adapter's per-model input
|
|
1550
|
+
* modality map so unsupported parts fail at compile time.
|
|
1551
|
+
*/
|
|
1552
|
+
export type MediaPromptFor<TModalities extends MediaPromptModality = never> =
|
|
1553
|
+
| string
|
|
1554
|
+
| Array<TextPart | MediaPartByModality[TModalities]>
|
|
1555
|
+
|
|
1556
|
+
/**
|
|
1557
|
+
* Per-model map from model name to the prompt modalities it accepts, used as
|
|
1558
|
+
* an adapter type parameter (`TModelInputModalitiesByName`). Models absent
|
|
1559
|
+
* from the map fall back to the unconstrained {@link MediaPrompt}.
|
|
1560
|
+
*/
|
|
1561
|
+
export type ModelInputModalitiesByName = Record<
|
|
1562
|
+
string,
|
|
1563
|
+
ReadonlyArray<MediaPromptModality>
|
|
1564
|
+
>
|
|
1565
|
+
|
|
1473
1566
|
/**
|
|
1474
1567
|
* Options for image generation.
|
|
1475
1568
|
* These are the common options supported across providers.
|
|
@@ -1480,8 +1573,16 @@ export interface ImageGenerationOptions<
|
|
|
1480
1573
|
> {
|
|
1481
1574
|
/** The model to use for image generation */
|
|
1482
1575
|
model: string
|
|
1483
|
-
/**
|
|
1484
|
-
|
|
1576
|
+
/**
|
|
1577
|
+
* Description of the desired image(s): a plain string, or an ordered array
|
|
1578
|
+
* of content parts for image-conditioned generation (image-to-image,
|
|
1579
|
+
* reference-guided, edit, multi-reference). Media parts may carry
|
|
1580
|
+
* `metadata.role` to disambiguate intent (mask, control, reference, …).
|
|
1581
|
+
* Adapters map parts onto the provider-native request — e.g. Gemini
|
|
1582
|
+
* multimodal `contents`, OpenAI `images.edit()`, fal `image_url` /
|
|
1583
|
+
* `mask_url` — and throw a clear runtime error for unsupported modalities.
|
|
1584
|
+
*/
|
|
1585
|
+
prompt: MediaPrompt
|
|
1485
1586
|
/** Number of images to generate (default: 1) */
|
|
1486
1587
|
numberOfImages?: number
|
|
1487
1588
|
/** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
|
|
@@ -1599,15 +1700,27 @@ export interface AudioGenerationResult {
|
|
|
1599
1700
|
export interface VideoGenerationOptions<
|
|
1600
1701
|
TProviderOptions extends object = object,
|
|
1601
1702
|
TSize extends string | undefined = string,
|
|
1703
|
+
TDuration extends string | number | undefined = number,
|
|
1602
1704
|
> {
|
|
1603
1705
|
/** The model to use for video generation */
|
|
1604
1706
|
model: string
|
|
1605
|
-
/**
|
|
1606
|
-
|
|
1707
|
+
/**
|
|
1708
|
+
* Description of the desired video: a plain string, or an ordered array of
|
|
1709
|
+
* content parts for image-conditioned generation. Image parts may carry
|
|
1710
|
+
* `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
|
|
1711
|
+
* 'character'`) to disambiguate intent; adapters route them onto the
|
|
1712
|
+
* provider-native request (e.g. OpenAI Sora `input_reference`, fal
|
|
1713
|
+
* `image_url` / `end_image_url`) and throw at runtime if unsupported.
|
|
1714
|
+
*/
|
|
1715
|
+
prompt: MediaPrompt
|
|
1607
1716
|
/** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
|
|
1608
1717
|
size?: TSize
|
|
1609
|
-
/**
|
|
1610
|
-
|
|
1718
|
+
/**
|
|
1719
|
+
* Video duration in seconds. Adapters that declare a per-model duration
|
|
1720
|
+
* map narrow this to the model's valid union; use
|
|
1721
|
+
* `adapter.snapDuration(seconds)` to coerce raw seconds to a valid value.
|
|
1722
|
+
*/
|
|
1723
|
+
duration?: TDuration
|
|
1611
1724
|
/** Model-specific options for video generation */
|
|
1612
1725
|
modelOptions?: TProviderOptions
|
|
1613
1726
|
/**
|