@tanstack/ai 0.31.0 → 0.33.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/dist/esm/activities/chat/index.js +24 -3
  2. package/dist/esm/activities/chat/index.js.map +1 -1
  3. package/dist/esm/activities/chat/middleware/types.d.ts +7 -0
  4. package/dist/esm/activities/chat/tools/lazy-tool-manager.d.ts +25 -1
  5. package/dist/esm/activities/chat/tools/lazy-tool-manager.js +26 -2
  6. package/dist/esm/activities/chat/tools/lazy-tool-manager.js.map +1 -1
  7. package/dist/esm/activities/generateAudio/index.d.ts +7 -0
  8. package/dist/esm/activities/generateAudio/index.js +26 -1
  9. package/dist/esm/activities/generateAudio/index.js.map +1 -1
  10. package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
  11. package/dist/esm/activities/generateImage/adapter.js.map +1 -1
  12. package/dist/esm/activities/generateImage/index.d.ts +26 -3
  13. package/dist/esm/activities/generateImage/index.js +38 -2
  14. package/dist/esm/activities/generateImage/index.js.map +1 -1
  15. package/dist/esm/activities/generateSpeech/index.d.ts +7 -0
  16. package/dist/esm/activities/generateSpeech/index.js +26 -1
  17. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  18. package/dist/esm/activities/generateTranscription/index.d.ts +7 -0
  19. package/dist/esm/activities/generateTranscription/index.js +26 -1
  20. package/dist/esm/activities/generateTranscription/index.js.map +1 -1
  21. package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
  22. package/dist/esm/activities/generateVideo/adapter.js +14 -0
  23. package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
  24. package/dist/esm/activities/generateVideo/index.d.ts +40 -5
  25. package/dist/esm/activities/generateVideo/index.js +52 -2
  26. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  27. package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
  28. package/dist/esm/activities/generateVideo/snap.js +54 -0
  29. package/dist/esm/activities/generateVideo/snap.js.map +1 -0
  30. package/dist/esm/activities/index.d.ts +3 -2
  31. package/dist/esm/activities/index.js +2 -0
  32. package/dist/esm/activities/index.js.map +1 -1
  33. package/dist/esm/activities/middleware/index.d.ts +2 -0
  34. package/dist/esm/activities/middleware/run.d.ts +20 -0
  35. package/dist/esm/activities/middleware/run.js +42 -0
  36. package/dist/esm/activities/middleware/run.js.map +1 -0
  37. package/dist/esm/activities/middleware/types.d.ts +118 -0
  38. package/dist/esm/client.d.ts +1 -1
  39. package/dist/esm/client.js.map +1 -1
  40. package/dist/esm/index.d.ts +4 -0
  41. package/dist/esm/index.js +4 -0
  42. package/dist/esm/index.js.map +1 -1
  43. package/dist/esm/middlewares/otel.d.ts +8 -2
  44. package/dist/esm/middlewares/otel.js +145 -95
  45. package/dist/esm/middlewares/otel.js.map +1 -1
  46. package/dist/esm/middlewares/usage-attributes.d.ts +24 -0
  47. package/dist/esm/middlewares/usage-attributes.js +43 -0
  48. package/dist/esm/middlewares/usage-attributes.js.map +1 -0
  49. package/dist/esm/types.d.ts +103 -14
  50. package/dist/esm/utilities/errors.d.ts +13 -0
  51. package/dist/esm/utilities/errors.js +22 -0
  52. package/dist/esm/utilities/errors.js.map +1 -0
  53. package/dist/esm/utilities/media-prompt.d.ts +35 -0
  54. package/dist/esm/utilities/media-prompt.js +43 -0
  55. package/dist/esm/utilities/media-prompt.js.map +1 -0
  56. package/dist/esm/utilities/numbers.d.ts +8 -0
  57. package/dist/esm/utilities/numbers.js +12 -0
  58. package/dist/esm/utilities/numbers.js.map +1 -0
  59. package/package.json +2 -2
  60. package/skills/ai-core/media-generation/SKILL.md +173 -3
  61. package/src/activities/chat/index.ts +32 -4
  62. package/src/activities/chat/middleware/types.ts +7 -0
  63. package/src/activities/chat/tools/lazy-tool-manager.ts +46 -4
  64. package/src/activities/generateAudio/index.ts +42 -1
  65. package/src/activities/generateImage/adapter.ts +16 -3
  66. package/src/activities/generateImage/index.ts +90 -5
  67. package/src/activities/generateSpeech/index.ts +42 -1
  68. package/src/activities/generateTranscription/index.ts +42 -1
  69. package/src/activities/generateVideo/adapter.ts +80 -4
  70. package/src/activities/generateVideo/index.ts +141 -6
  71. package/src/activities/generateVideo/snap.ts +100 -0
  72. package/src/activities/index.ts +4 -0
  73. package/src/activities/middleware/index.ts +20 -0
  74. package/src/activities/middleware/run.ts +88 -0
  75. package/src/activities/middleware/types.ts +173 -0
  76. package/src/client.ts +4 -0
  77. package/src/index.ts +23 -0
  78. package/src/middlewares/otel.ts +195 -120
  79. package/src/middlewares/usage-attributes.ts +65 -0
  80. package/src/types.ts +126 -13
  81. package/src/utilities/errors.ts +29 -0
  82. package/src/utilities/media-prompt.ts +86 -0
  83. package/src/utilities/numbers.ts +15 -0
package/src/types.ts CHANGED
@@ -820,14 +820,14 @@ export interface TextOptions<
820
820
  systemPrompts?: Array<SystemPrompt>
821
821
  agentLoopStrategy?: AgentLoopStrategy
822
822
  /**
823
- * Additional metadata to attach to the request.
824
- * Can be used for tracking, debugging, or passing custom information.
825
- * Structure and constraints vary by provider.
823
+ * Observability metadata attached to this call. Surfaced to middleware,
824
+ * devtools, and the event client; values may be arbitrarily structured
825
+ * (objects, arrays). Adapters never forward this field onto the provider
826
+ * wire request.
826
827
  *
827
- * Provider usage:
828
- * - OpenAI: `metadata` (Record<string, string>) - max 16 key-value pairs, keys max 64 chars, values max 512 chars
829
- * - Anthropic: `metadata` (Record<string, any>) - includes optional user_id (max 256 chars)
830
- * - Gemini: Not directly available in TextProviderOptions
828
+ * To send provider-side request metadata, use the provider's
829
+ * `modelOptions` field instead, where the provider supports one (e.g.
830
+ * OpenAI's and OpenRouter's `metadata` are both Record<string, string>).
831
831
  */
832
832
  metadata?: Record<string, any> | undefined
833
833
  modelOptions?: TProviderOptionsForModel
@@ -1470,6 +1470,99 @@ export interface SummarizationResult {
1470
1470
  // Image Generation Types
1471
1471
  // ============================================================================
1472
1472
 
1473
+ /**
1474
+ * Optional role hint on a media input part (image / video / audio). Adapters
1475
+ * read `metadata.role` to route the part to the provider-specific request
1476
+ * field — e.g. `'mask'` → OpenAI `mask` / fal `mask_url`, `'end_frame'` → fal
1477
+ * `end_image_url`, `'reference'` → fal `reference_image_urls`. When omitted
1478
+ * the adapter falls back to positional routing.
1479
+ */
1480
+ export type MediaInputRole =
1481
+ | 'reference'
1482
+ | 'mask'
1483
+ | 'control'
1484
+ | 'start_frame'
1485
+ | 'end_frame'
1486
+ | 'character'
1487
+
1488
+ /**
1489
+ * Metadata convention for image / video / audio inputs to media generation.
1490
+ * Carried on `ImagePart.metadata` / `VideoPart.metadata` / `AudioPart.metadata`
1491
+ * when used as conditioning inputs to `generateImage()` or `generateVideo()`.
1492
+ */
1493
+ export interface MediaInputMetadata {
1494
+ /** Optional role hint disambiguating the part's intent for the adapter */
1495
+ role?: MediaInputRole
1496
+ /**
1497
+ * Optional user-defined label for this input (e.g. `'woman-in-red-dress'`).
1498
+ * **Informational only** — adapters never read it and the SDK never
1499
+ * rewrites prompt text based on it. Use it to correlate parts with the
1500
+ * references you write in your prompt using the provider's own syntax
1501
+ * (fal's `@Image1`, OpenAI's "image 1", etc.), or for your own
1502
+ * bookkeeping/logging.
1503
+ */
1504
+ tag?: string
1505
+ }
1506
+
1507
+ /**
1508
+ * A single part of a multimodal media-generation prompt. Reuses the chat
1509
+ * content-part shapes: text parts carry the instruction, image / video /
1510
+ * audio parts carry conditioning inputs (with an optional
1511
+ * `metadata.role` hint — see {@link MediaInputRole}).
1512
+ */
1513
+ export type MediaPromptPart =
1514
+ | TextPart
1515
+ | ImagePart<MediaInputMetadata>
1516
+ | VideoPart<MediaInputMetadata>
1517
+ | AudioPart<MediaInputMetadata>
1518
+
1519
+ /**
1520
+ * Prompt accepted by `generateImage()` / `generateVideo()`: a plain string,
1521
+ * or an ordered array of content parts for image-conditioned generation
1522
+ * ("not like this *(image)*, more like this *(image)*"). Part order is
1523
+ * meaningful — adapters with native multimodal prompts (Gemini, OpenRouter)
1524
+ * preserve the interleaving; named-field providers (fal, OpenAI, xAI)
1525
+ * extract the media parts and flatten the text. Text is always sent
1526
+ * verbatim: to reference inputs from the prompt, write the provider's own
1527
+ * syntax yourself (e.g. fal's `@Image1`, OpenAI's "image 1"). An array may
1528
+ * be media-only (e.g. upscalers or pure img2img endpoints that take no
1529
+ * instruction text).
1530
+ */
1531
+ export type MediaPrompt = string | Array<MediaPromptPart>
1532
+
1533
+ /**
1534
+ * Non-text modalities a media-generation model can accept in its prompt.
1535
+ */
1536
+ export type MediaPromptModality = 'image' | 'video' | 'audio'
1537
+
1538
+ /** Maps a prompt modality to its content-part type. @internal */
1539
+ interface MediaPartByModality {
1540
+ image: ImagePart<MediaInputMetadata>
1541
+ video: VideoPart<MediaInputMetadata>
1542
+ audio: AudioPart<MediaInputMetadata>
1543
+ }
1544
+
1545
+ /**
1546
+ * Prompt type narrowed to the modalities a specific model supports.
1547
+ * `MediaPromptFor<never>` (a text-only model) is `string | Array<TextPart>`;
1548
+ * `MediaPromptFor<'image'>` additionally admits image parts, etc. Used by
1549
+ * the activity option types together with the adapter's per-model input
1550
+ * modality map so unsupported parts fail at compile time.
1551
+ */
1552
+ export type MediaPromptFor<TModalities extends MediaPromptModality = never> =
1553
+ | string
1554
+ | Array<TextPart | MediaPartByModality[TModalities]>
1555
+
1556
+ /**
1557
+ * Per-model map from model name to the prompt modalities it accepts, used as
1558
+ * an adapter type parameter (`TModelInputModalitiesByName`). Models absent
1559
+ * from the map fall back to the unconstrained {@link MediaPrompt}.
1560
+ */
1561
+ export type ModelInputModalitiesByName = Record<
1562
+ string,
1563
+ ReadonlyArray<MediaPromptModality>
1564
+ >
1565
+
1473
1566
  /**
1474
1567
  * Options for image generation.
1475
1568
  * These are the common options supported across providers.
@@ -1480,8 +1573,16 @@ export interface ImageGenerationOptions<
1480
1573
  > {
1481
1574
  /** The model to use for image generation */
1482
1575
  model: string
1483
- /** Text description of the desired image(s) */
1484
- prompt: string
1576
+ /**
1577
+ * Description of the desired image(s): a plain string, or an ordered array
1578
+ * of content parts for image-conditioned generation (image-to-image,
1579
+ * reference-guided, edit, multi-reference). Media parts may carry
1580
+ * `metadata.role` to disambiguate intent (mask, control, reference, …).
1581
+ * Adapters map parts onto the provider-native request — e.g. Gemini
1582
+ * multimodal `contents`, OpenAI `images.edit()`, fal `image_url` /
1583
+ * `mask_url` — and throw a clear runtime error for unsupported modalities.
1584
+ */
1585
+ prompt: MediaPrompt
1485
1586
  /** Number of images to generate (default: 1) */
1486
1587
  numberOfImages?: number
1487
1588
  /** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
@@ -1599,15 +1700,27 @@ export interface AudioGenerationResult {
1599
1700
  export interface VideoGenerationOptions<
1600
1701
  TProviderOptions extends object = object,
1601
1702
  TSize extends string | undefined = string,
1703
+ TDuration extends string | number | undefined = number,
1602
1704
  > {
1603
1705
  /** The model to use for video generation */
1604
1706
  model: string
1605
- /** Text description of the desired video */
1606
- prompt: string
1707
+ /**
1708
+ * Description of the desired video: a plain string, or an ordered array of
1709
+ * content parts for image-conditioned generation. Image parts may carry
1710
+ * `metadata.role` (`'start_frame' | 'end_frame' | 'reference' |
1711
+ * 'character'`) to disambiguate intent; adapters route them onto the
1712
+ * provider-native request (e.g. OpenAI Sora `input_reference`, fal
1713
+ * `image_url` / `end_image_url`) and throw at runtime if unsupported.
1714
+ */
1715
+ prompt: MediaPrompt
1607
1716
  /** Video size — format depends on the provider (e.g., "16:9", "1280x720") */
1608
1717
  size?: TSize
1609
- /** Video duration in seconds */
1610
- duration?: number
1718
+ /**
1719
+ * Video duration in seconds. Adapters that declare a per-model duration
1720
+ * map narrow this to the model's valid union; use
1721
+ * `adapter.snapDuration(seconds)` to coerce raw seconds to a valid value.
1722
+ */
1723
+ duration?: TDuration
1611
1724
  /** Model-specific options for video generation */
1612
1725
  modelOptions?: TProviderOptions
1613
1726
  /**
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Best-effort extraction of a human-readable message from an unknown thrown
3
+ * value, returning `undefined` when none can be found.
4
+ *
5
+ * Used by `otelMiddleware` so error reporting stays identical across chat and
6
+ * media spans.
7
+ */
8
+ export function errorMessage(err: unknown): string | undefined {
9
+ if (err instanceof Error) return err.message
10
+ if (typeof err === 'string') return err
11
+ if (err && typeof err === 'object' && 'message' in err) {
12
+ const m = (err as { message?: unknown }).message
13
+ if (typeof m === 'string') return m
14
+ }
15
+ return undefined
16
+ }
17
+
18
+ /**
19
+ * Best-effort extraction of an error's type name (used for the `error.type`
20
+ * metric attribute), falling back to `'Error'` when no name is available.
21
+ */
22
+ export function errorTypeName(err: unknown): string {
23
+ if (err instanceof Error) return err.name || 'Error'
24
+ if (err && typeof err === 'object' && 'name' in err) {
25
+ const n = (err as { name?: unknown }).name
26
+ if (typeof n === 'string') return n
27
+ }
28
+ return 'Error'
29
+ }
@@ -0,0 +1,86 @@
1
+ import type {
2
+ AudioPart,
3
+ ImagePart,
4
+ MediaInputMetadata,
5
+ MediaPrompt,
6
+ MediaPromptPart,
7
+ TextPart,
8
+ VideoPart,
9
+ } from '../types'
10
+
11
+ /**
12
+ * A {@link MediaPrompt} decomposed into the views adapters consume.
13
+ *
14
+ * Adapters with native multimodal prompts (Gemini `contents`, OpenRouter
15
+ * chat content parts) consume `parts` to preserve interleaving; named-field
16
+ * providers (fal, OpenAI) consume `text` plus the typed media buckets.
17
+ *
18
+ * Prompt text is **never rewritten**: text parts are concatenated verbatim.
19
+ * Providers that support referencing inputs from the prompt (e.g. fal's
20
+ * `@Image1`, OpenAI's "image 1" prose) expect the user to write that syntax
21
+ * themselves — the SDK does not inject or substitute markers.
22
+ */
23
+ export interface ResolvedMediaPrompt {
24
+ /**
25
+ * Text parts concatenated verbatim (paragraph-separated). Empty string
26
+ * for media-only prompts.
27
+ */
28
+ text: string
29
+ /** The prompt as ordered parts; a string prompt becomes one text part. */
30
+ parts: Array<MediaPromptPart>
31
+ /** Image parts in prompt order. */
32
+ images: Array<ImagePart<MediaInputMetadata>>
33
+ /** Video parts in prompt order. */
34
+ videos: Array<VideoPart<MediaInputMetadata>>
35
+ /** Audio parts in prompt order. */
36
+ audios: Array<AudioPart<MediaInputMetadata>>
37
+ }
38
+
39
+ /**
40
+ * Decompose a {@link MediaPrompt} into flattened text and per-modality part
41
+ * buckets, preserving prompt order everywhere. This is the single downrev
42
+ * point from the canonical interleaved prompt shape to the named-field
43
+ * request shapes most providers expose.
44
+ */
45
+ export function resolveMediaPrompt(prompt: MediaPrompt): ResolvedMediaPrompt {
46
+ if (typeof prompt === 'string') {
47
+ const textPart: TextPart = { type: 'text', content: prompt }
48
+ return {
49
+ text: prompt,
50
+ parts: [textPart],
51
+ images: [],
52
+ videos: [],
53
+ audios: [],
54
+ }
55
+ }
56
+
57
+ const images: Array<ImagePart<MediaInputMetadata>> = []
58
+ const videos: Array<VideoPart<MediaInputMetadata>> = []
59
+ const audios: Array<AudioPart<MediaInputMetadata>> = []
60
+ const textSegments: Array<string> = []
61
+
62
+ for (const part of prompt) {
63
+ switch (part.type) {
64
+ case 'text':
65
+ if (part.content) textSegments.push(part.content)
66
+ break
67
+ case 'image':
68
+ images.push(part)
69
+ break
70
+ case 'video':
71
+ videos.push(part)
72
+ break
73
+ case 'audio':
74
+ audios.push(part)
75
+ break
76
+ }
77
+ }
78
+
79
+ return {
80
+ text: textSegments.join('\n\n'),
81
+ parts: prompt,
82
+ images,
83
+ videos,
84
+ audios,
85
+ }
86
+ }
@@ -0,0 +1,15 @@
1
+ /**
2
+ * Return the first candidate that is a finite `number`, or `undefined`.
3
+ *
4
+ * Handy for picking a value from among several possible spellings/sources where
5
+ * only some are populated — e.g. the provider-native sampling option names read
6
+ * by the OTel middleware, or the optional numeric fields on `TokenUsage`.
7
+ */
8
+ export function firstNumber(...candidates: Array<unknown>): number | undefined {
9
+ for (const candidate of candidates) {
10
+ if (typeof candidate === 'number' && Number.isFinite(candidate)) {
11
+ return candidate
12
+ }
13
+ }
14
+ return undefined
15
+ }