@tanstack/ai 0.31.0 → 0.33.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/index.js +24 -3
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/middleware/types.d.ts +7 -0
- package/dist/esm/activities/chat/tools/lazy-tool-manager.d.ts +25 -1
- package/dist/esm/activities/chat/tools/lazy-tool-manager.js +26 -2
- package/dist/esm/activities/chat/tools/lazy-tool-manager.js.map +1 -1
- package/dist/esm/activities/generateAudio/index.d.ts +7 -0
- package/dist/esm/activities/generateAudio/index.js +26 -1
- package/dist/esm/activities/generateAudio/index.js.map +1 -1
- package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
- package/dist/esm/activities/generateImage/adapter.js.map +1 -1
- package/dist/esm/activities/generateImage/index.d.ts +26 -3
- package/dist/esm/activities/generateImage/index.js +38 -2
- package/dist/esm/activities/generateImage/index.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +7 -0
- package/dist/esm/activities/generateSpeech/index.js +26 -1
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/activities/generateTranscription/index.d.ts +7 -0
- package/dist/esm/activities/generateTranscription/index.js +26 -1
- package/dist/esm/activities/generateTranscription/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
- package/dist/esm/activities/generateVideo/adapter.js +14 -0
- package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
- package/dist/esm/activities/generateVideo/index.d.ts +40 -5
- package/dist/esm/activities/generateVideo/index.js +52 -2
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
- package/dist/esm/activities/generateVideo/snap.js +54 -0
- package/dist/esm/activities/generateVideo/snap.js.map +1 -0
- package/dist/esm/activities/index.d.ts +3 -2
- package/dist/esm/activities/index.js +2 -0
- package/dist/esm/activities/index.js.map +1 -1
- package/dist/esm/activities/middleware/index.d.ts +2 -0
- package/dist/esm/activities/middleware/run.d.ts +20 -0
- package/dist/esm/activities/middleware/run.js +42 -0
- package/dist/esm/activities/middleware/run.js.map +1 -0
- package/dist/esm/activities/middleware/types.d.ts +118 -0
- package/dist/esm/client.d.ts +1 -1
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -0
- package/dist/esm/index.js +4 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/middlewares/otel.d.ts +8 -2
- package/dist/esm/middlewares/otel.js +145 -95
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/middlewares/usage-attributes.d.ts +24 -0
- package/dist/esm/middlewares/usage-attributes.js +43 -0
- package/dist/esm/middlewares/usage-attributes.js.map +1 -0
- package/dist/esm/types.d.ts +103 -14
- package/dist/esm/utilities/errors.d.ts +13 -0
- package/dist/esm/utilities/errors.js +22 -0
- package/dist/esm/utilities/errors.js.map +1 -0
- package/dist/esm/utilities/media-prompt.d.ts +35 -0
- package/dist/esm/utilities/media-prompt.js +43 -0
- package/dist/esm/utilities/media-prompt.js.map +1 -0
- package/dist/esm/utilities/numbers.d.ts +8 -0
- package/dist/esm/utilities/numbers.js +12 -0
- package/dist/esm/utilities/numbers.js.map +1 -0
- package/package.json +2 -2
- package/skills/ai-core/media-generation/SKILL.md +173 -3
- package/src/activities/chat/index.ts +32 -4
- package/src/activities/chat/middleware/types.ts +7 -0
- package/src/activities/chat/tools/lazy-tool-manager.ts +46 -4
- package/src/activities/generateAudio/index.ts +42 -1
- package/src/activities/generateImage/adapter.ts +16 -3
- package/src/activities/generateImage/index.ts +90 -5
- package/src/activities/generateSpeech/index.ts +42 -1
- package/src/activities/generateTranscription/index.ts +42 -1
- package/src/activities/generateVideo/adapter.ts +80 -4
- package/src/activities/generateVideo/index.ts +141 -6
- package/src/activities/generateVideo/snap.ts +100 -0
- package/src/activities/index.ts +4 -0
- package/src/activities/middleware/index.ts +20 -0
- package/src/activities/middleware/run.ts +88 -0
- package/src/activities/middleware/types.ts +173 -0
- package/src/client.ts +4 -0
- package/src/index.ts +23 -0
- package/src/middlewares/otel.ts +195 -120
- package/src/middlewares/usage-attributes.ts +65 -0
- package/src/types.ts +126 -13
- package/src/utilities/errors.ts +29 -0
- package/src/utilities/media-prompt.ts +86 -0
- package/src/utilities/numbers.ts +15 -0
package/src/types.ts
CHANGED
|
@@ -820,14 +820,14 @@ export interface TextOptions<
|
|
|
820
820
|
systemPrompts?: Array<SystemPrompt>
|
|
821
821
|
agentLoopStrategy?: AgentLoopStrategy
|
|
822
822
|
/**
|
|
823
|
-
*
|
|
824
|
-
*
|
|
825
|
-
*
|
|
823
|
+
* Observability metadata attached to this call. Surfaced to middleware,
|
|
824
|
+
* devtools, and the event client; values may be arbitrarily structured
|
|
825
|
+
* (objects, arrays). Adapters never forward this field onto the provider
|
|
826
|
+
* wire request.
|
|
826
827
|
*
|
|
827
|
-
*
|
|
828
|
-
*
|
|
829
|
-
*
|
|
830
|
-
* - Gemini: Not directly available in TextProviderOptions
|
|
828
|
+
* To send provider-side request metadata, use the provider's
|
|
829
|
+
* `modelOptions` field instead, where the provider supports one (e.g.
|
|
830
|
+
* OpenAI's and OpenRouter's `metadata` are both Record<string, string>).
|
|
831
831
|
*/
|
|
832
832
|
metadata?: Record<string, any> | undefined
|
|
833
833
|
modelOptions?: TProviderOptionsForModel
|
|
@@ -1470,6 +1470,99 @@ export interface SummarizationResult {
|
|
|
1470
1470
|
// Image Generation Types
|
|
1471
1471
|
// ============================================================================
|
|
1472
1472
|
|
|
1473
|
+
/**
|
|
1474
|
+
* Optional role hint on a media input part (image / video / audio). Adapters
|
|
1475
|
+
* read `metadata.role` to route the part to the provider-specific request
|
|
1476
|
+
* field — e.g. `'mask'` → OpenAI `mask` / fal `mask_url`, `'end_frame'` → fal
|
|
1477
|
+
* `end_image_url`, `'reference'` → fal `reference_image_urls`. When omitted
|
|
1478
|
+
* the adapter falls back to positional routing.
|
|
1479
|
+
*/
|
|
1480
|
+
export type MediaInputRole =
|
|
1481
|
+
| 'reference'
|
|
1482
|
+
| 'mask'
|
|
1483
|
+
| 'control'
|
|
1484
|
+
| 'start_frame'
|
|
1485
|
+
| 'end_frame'
|
|
1486
|
+
| 'character'
|
|
1487
|
+
|
|
1488
|
+
/**
|
|
1489
|
+
* Metadata convention for image / video / audio inputs to media generation.
|
|
1490
|
+
* Carried on `ImagePart.metadata` / `VideoPart.metadata` / `AudioPart.metadata`
|
|
1491
|
+
* when used as conditioning inputs to `generateImage()` or `generateVideo()`.
|
|
1492
|
+
*/
|
|
1493
|
+
export interface MediaInputMetadata {
|
|
1494
|
+
/** Optional role hint disambiguating the part's intent for the adapter */
|
|
1495
|
+
role?: MediaInputRole
|
|
1496
|
+
/**
|
|
1497
|
+
* Optional user-defined label for this input (e.g. `'woman-in-red-dress'`).
|
|
1498
|
+
* **Informational only** — adapters never read it and the SDK never
|
|
1499
|
+
* rewrites prompt text based on it. Use it to correlate parts with the
|
|
1500
|
+
* references you write in your prompt using the provider's own syntax
|
|
1501
|
+
* (fal's `@Image1`, OpenAI's "image 1", etc.), or for your own
|
|
1502
|
+
* bookkeeping/logging.
|
|
1503
|
+
*/
|
|
1504
|
+
tag?: string
|
|
1505
|
+
}
|
|
1506
|
+
|
|
1507
|
+
/**
|
|
1508
|
+
* A single part of a multimodal media-generation prompt. Reuses the chat
|
|
1509
|
+
* content-part shapes: text parts carry the instruction, image / video /
|
|
1510
|
+
* audio parts carry conditioning inputs (with an optional
|
|
1511
|
+
* `metadata.role` hint — see {@link MediaInputRole}).
|
|
1512
|
+
*/
|
|
1513
|
+
export type MediaPromptPart =
|
|
1514
|
+
| TextPart
|
|
1515
|
+
| ImagePart<MediaInputMetadata>
|
|
1516
|
+
| VideoPart<MediaInputMetadata>
|
|
1517
|
+
| AudioPart<MediaInputMetadata>
|
|
1518
|
+
|
|
1519
|
+
/**
|
|
1520
|
+
* Prompt accepted by `generateImage()` / `generateVideo()`: a plain string,
|
|
1521
|
+
* or an ordered array of content parts for image-conditioned generation
|
|
1522
|
+
* ("not like this *(image)*, more like this *(image)*"). Part order is
|
|
1523
|
+
* meaningful — adapters with native multimodal prompts (Gemini, OpenRouter)
|
|
1524
|
+
* preserve the interleaving; named-field providers (fal, OpenAI, xAI)
|
|
1525
|
+
* extract the media parts and flatten the text. Text is always sent
|
|
1526
|
+
* verbatim: to reference inputs from the prompt, write the provider's own
|
|
1527
|
+
* syntax yourself (e.g. fal's `@Image1`, OpenAI's "image 1"). An array may
|
|
1528
|
+
* be media-only (e.g. upscalers or pure img2img endpoints that take no
|
|
1529
|
+
* instruction text).
|
|
1530
|
+
*/
|
|
1531
|
+
export type MediaPrompt = string | Array<MediaPromptPart>
|
|
1532
|
+
|
|
1533
|
+
/**
|
|
1534
|
+
* Non-text modalities a media-generation model can accept in its prompt.
|
|
1535
|
+
*/
|
|
1536
|
+
export type MediaPromptModality = 'image' | 'video' | 'audio'
|
|
1537
|
+
|
|
1538
|
+
/** Maps a prompt modality to its content-part type. @internal */
|
|
1539
|
+
interface MediaPartByModality {
|
|
1540
|
+
image: ImagePart<MediaInputMetadata>
|
|
1541
|
+
video: VideoPart<MediaInputMetadata>
|
|
1542
|
+
audio: AudioPart<MediaInputMetadata>
|
|
1543
|
+
}
|
|
1544
|
+
|
|
1545
|
+
/**
|
|
1546
|
+
* Prompt type narrowed to the modalities a specific model supports.
|
|
1547
|
+
* `MediaPromptFor<never>` (a text-only model) is `string | Array<TextPart>`;
|
|
1548
|
+
* `MediaPromptFor<'image'>` additionally admits image parts, etc. Used by
|
|
1549
|
+
* the activity option types together with the adapter's per-model input
|
|
1550
|
+
* modality map so unsupported parts fail at compile time.
|
|
1551
|
+
*/
|
|
1552
|
+
export type MediaPromptFor<TModalities extends MediaPromptModality = never> =
|
|
1553
|
+
| string
|
|
1554
|
+
| Array<TextPart | MediaPartByModality[TModalities]>
|
|
1555
|
+
|
|
1556
|
+
/**
|
|
1557
|
+
* Per-model map from model name to the prompt modalities it accepts, used as
|
|
1558
|
+
* an adapter type parameter (`TModelInputModalitiesByName`). Models absent
|
|
1559
|
+
* from the map fall back to the unconstrained {@link MediaPrompt}.
|
|
1560
|
+
*/
|
|
1561
|
+
export type ModelInputModalitiesByName = Record<
|
|
1562
|
+
string,
|
|
1563
|
+
ReadonlyArray<MediaPromptModality>
|
|
1564
|
+
>
|
|
1565
|
+
|
|
1473
1566
|
/**
|
|
1474
1567
|
* Options for image generation.
|
|
1475
1568
|
* These are the common options supported across providers.
|
|
@@ -1480,8 +1573,16 @@ export interface ImageGenerationOptions<
|
|
|
1480
1573
|
> {
|
|
1481
1574
|
/** The model to use for image generation */
|
|
1482
1575
|
model: string
|
|
1483
|
-
/**
|
|
1484
|
-
|
|
1576
|
+
/**
|
|
1577
|
+
* Description of the desired image(s): a plain string, or an ordered array
|
|
1578
|
+
* of content parts for image-conditioned generation (image-to-image,
|
|
1579
|
+
* reference-guided, edit, multi-reference). Media parts may carry
|
|
1580
|
+
* `metadata.role` to disambiguate intent (mask, control, reference, …).
|
|
1581
|
+
* Adapters map parts onto the provider-native request — e.g. Gemini
|
|
1582
|
+
* multimodal `contents`, OpenAI `images.edit()`, fal `image_url` /
|
|
1583
|
+
* `mask_url` — and throw a clear runtime error for unsupported modalities.
|
|
1584
|
+
*/
|
|
1585
|
+
prompt: MediaPrompt
|
|
1485
1586
|
/** Number of images to generate (default: 1) */
|
|
1486
1587
|
numberOfImages?: number
|
|
1487
1588
|
/** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
|
|
@@ -1599,15 +1700,27 @@ export interface AudioGenerationResult {
|
|
|
1599
1700
|
export interface VideoGenerationOptions<
|
|
1600
1701
|
TProviderOptions extends object = object,
|
|
1601
1702
|
TSize extends string | undefined = string,
|
|
1703
|
+
TDuration extends string | number | undefined = number,
|
|
1602
1704
|
> {
|
|
1603
1705
|
/** The model to use for video generation */
|
|
1604
1706
|
model: string
|
|
1605
|
-
/**
|
|
1606
|
-
|
|
1707
|
+
/**
|
|
1708
|
+
* Description of the desired video: a plain string, or an ordered array of
|
|
1709
|
+
* content parts for image-conditioned generation. Image parts may carry
|
|
1710
|
+
* `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
|
|
1711
|
+
* 'character'`) to disambiguate intent; adapters route them onto the
|
|
1712
|
+
* provider-native request (e.g. OpenAI Sora `input_reference`, fal
|
|
1713
|
+
* `image_url` / `end_image_url`) and throw at runtime if unsupported.
|
|
1714
|
+
*/
|
|
1715
|
+
prompt: MediaPrompt
|
|
1607
1716
|
/** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
|
|
1608
1717
|
size?: TSize
|
|
1609
|
-
/**
|
|
1610
|
-
|
|
1718
|
+
/**
|
|
1719
|
+
* Video duration in seconds. Adapters that declare a per-model duration
|
|
1720
|
+
* map narrow this to the model's valid union; use
|
|
1721
|
+
* `adapter.snapDuration(seconds)` to coerce raw seconds to a valid value.
|
|
1722
|
+
*/
|
|
1723
|
+
duration?: TDuration
|
|
1611
1724
|
/** Model-specific options for video generation */
|
|
1612
1725
|
modelOptions?: TProviderOptions
|
|
1613
1726
|
/**
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Best-effort extraction of a human-readable message from an unknown thrown
|
|
3
|
+
* value, returning `undefined` when none can be found.
|
|
4
|
+
*
|
|
5
|
+
* Used by `otelMiddleware` so error reporting stays identical across chat and
|
|
6
|
+
* media spans.
|
|
7
|
+
*/
|
|
8
|
+
export function errorMessage(err: unknown): string | undefined {
|
|
9
|
+
if (err instanceof Error) return err.message
|
|
10
|
+
if (typeof err === 'string') return err
|
|
11
|
+
if (err && typeof err === 'object' && 'message' in err) {
|
|
12
|
+
const m = (err as { message?: unknown }).message
|
|
13
|
+
if (typeof m === 'string') return m
|
|
14
|
+
}
|
|
15
|
+
return undefined
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Best-effort extraction of an error's type name (used for the `error.type`
|
|
20
|
+
* metric attribute), falling back to `'Error'` when no name is available.
|
|
21
|
+
*/
|
|
22
|
+
export function errorTypeName(err: unknown): string {
|
|
23
|
+
if (err instanceof Error) return err.name || 'Error'
|
|
24
|
+
if (err && typeof err === 'object' && 'name' in err) {
|
|
25
|
+
const n = (err as { name?: unknown }).name
|
|
26
|
+
if (typeof n === 'string') return n
|
|
27
|
+
}
|
|
28
|
+
return 'Error'
|
|
29
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
AudioPart,
|
|
3
|
+
ImagePart,
|
|
4
|
+
MediaInputMetadata,
|
|
5
|
+
MediaPrompt,
|
|
6
|
+
MediaPromptPart,
|
|
7
|
+
TextPart,
|
|
8
|
+
VideoPart,
|
|
9
|
+
} from '../types'
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* A {@link MediaPrompt} decomposed into the views adapters consume.
|
|
13
|
+
*
|
|
14
|
+
* Adapters with native multimodal prompts (Gemini `contents`, OpenRouter
|
|
15
|
+
* chat content parts) consume `parts` to preserve interleaving; named-field
|
|
16
|
+
* providers (fal, OpenAI) consume `text` plus the typed media buckets.
|
|
17
|
+
*
|
|
18
|
+
* Prompt text is **never rewritten**: text parts are concatenated verbatim.
|
|
19
|
+
* Providers that support referencing inputs from the prompt (e.g. fal's
|
|
20
|
+
* `@Image1`, OpenAI's "image 1" prose) expect the user to write that syntax
|
|
21
|
+
* themselves — the SDK does not inject or substitute markers.
|
|
22
|
+
*/
|
|
23
|
+
export interface ResolvedMediaPrompt {
|
|
24
|
+
/**
|
|
25
|
+
* Text parts concatenated verbatim (paragraph-separated). Empty string
|
|
26
|
+
* for media-only prompts.
|
|
27
|
+
*/
|
|
28
|
+
text: string
|
|
29
|
+
/** The prompt as ordered parts; a string prompt becomes one text part. */
|
|
30
|
+
parts: Array<MediaPromptPart>
|
|
31
|
+
/** Image parts in prompt order. */
|
|
32
|
+
images: Array<ImagePart<MediaInputMetadata>>
|
|
33
|
+
/** Video parts in prompt order. */
|
|
34
|
+
videos: Array<VideoPart<MediaInputMetadata>>
|
|
35
|
+
/** Audio parts in prompt order. */
|
|
36
|
+
audios: Array<AudioPart<MediaInputMetadata>>
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Decompose a {@link MediaPrompt} into flattened text and per-modality part
|
|
41
|
+
* buckets, preserving prompt order everywhere. This is the single downrev
|
|
42
|
+
* point from the canonical interleaved prompt shape to the named-field
|
|
43
|
+
* request shapes most providers expose.
|
|
44
|
+
*/
|
|
45
|
+
export function resolveMediaPrompt(prompt: MediaPrompt): ResolvedMediaPrompt {
|
|
46
|
+
if (typeof prompt === 'string') {
|
|
47
|
+
const textPart: TextPart = { type: 'text', content: prompt }
|
|
48
|
+
return {
|
|
49
|
+
text: prompt,
|
|
50
|
+
parts: [textPart],
|
|
51
|
+
images: [],
|
|
52
|
+
videos: [],
|
|
53
|
+
audios: [],
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const images: Array<ImagePart<MediaInputMetadata>> = []
|
|
58
|
+
const videos: Array<VideoPart<MediaInputMetadata>> = []
|
|
59
|
+
const audios: Array<AudioPart<MediaInputMetadata>> = []
|
|
60
|
+
const textSegments: Array<string> = []
|
|
61
|
+
|
|
62
|
+
for (const part of prompt) {
|
|
63
|
+
switch (part.type) {
|
|
64
|
+
case 'text':
|
|
65
|
+
if (part.content) textSegments.push(part.content)
|
|
66
|
+
break
|
|
67
|
+
case 'image':
|
|
68
|
+
images.push(part)
|
|
69
|
+
break
|
|
70
|
+
case 'video':
|
|
71
|
+
videos.push(part)
|
|
72
|
+
break
|
|
73
|
+
case 'audio':
|
|
74
|
+
audios.push(part)
|
|
75
|
+
break
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
return {
|
|
80
|
+
text: textSegments.join('\n\n'),
|
|
81
|
+
parts: prompt,
|
|
82
|
+
images,
|
|
83
|
+
videos,
|
|
84
|
+
audios,
|
|
85
|
+
}
|
|
86
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Return the first candidate that is a finite `number`, or `undefined`.
|
|
3
|
+
*
|
|
4
|
+
* Handy for picking a value from among several possible spellings/sources where
|
|
5
|
+
* only some are populated — e.g. the provider-native sampling option names read
|
|
6
|
+
* by the OTel middleware, or the optional numeric fields on `TokenUsage`.
|
|
7
|
+
*/
|
|
8
|
+
export function firstNumber(...candidates: Array<unknown>): number | undefined {
|
|
9
|
+
for (const candidate of candidates) {
|
|
10
|
+
if (typeof candidate === 'number' && Number.isFinite(candidate)) {
|
|
11
|
+
return candidate
|
|
12
|
+
}
|
|
13
|
+
}
|
|
14
|
+
return undefined
|
|
15
|
+
}
|