@tanstack/ai 0.31.0 → 0.33.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/index.js +24 -3
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/middleware/types.d.ts +7 -0
- package/dist/esm/activities/chat/tools/lazy-tool-manager.d.ts +25 -1
- package/dist/esm/activities/chat/tools/lazy-tool-manager.js +26 -2
- package/dist/esm/activities/chat/tools/lazy-tool-manager.js.map +1 -1
- package/dist/esm/activities/generateAudio/index.d.ts +7 -0
- package/dist/esm/activities/generateAudio/index.js +26 -1
- package/dist/esm/activities/generateAudio/index.js.map +1 -1
- package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
- package/dist/esm/activities/generateImage/adapter.js.map +1 -1
- package/dist/esm/activities/generateImage/index.d.ts +26 -3
- package/dist/esm/activities/generateImage/index.js +38 -2
- package/dist/esm/activities/generateImage/index.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +7 -0
- package/dist/esm/activities/generateSpeech/index.js +26 -1
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/activities/generateTranscription/index.d.ts +7 -0
- package/dist/esm/activities/generateTranscription/index.js +26 -1
- package/dist/esm/activities/generateTranscription/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
- package/dist/esm/activities/generateVideo/adapter.js +14 -0
- package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
- package/dist/esm/activities/generateVideo/index.d.ts +40 -5
- package/dist/esm/activities/generateVideo/index.js +52 -2
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
- package/dist/esm/activities/generateVideo/snap.js +54 -0
- package/dist/esm/activities/generateVideo/snap.js.map +1 -0
- package/dist/esm/activities/index.d.ts +3 -2
- package/dist/esm/activities/index.js +2 -0
- package/dist/esm/activities/index.js.map +1 -1
- package/dist/esm/activities/middleware/index.d.ts +2 -0
- package/dist/esm/activities/middleware/run.d.ts +20 -0
- package/dist/esm/activities/middleware/run.js +42 -0
- package/dist/esm/activities/middleware/run.js.map +1 -0
- package/dist/esm/activities/middleware/types.d.ts +118 -0
- package/dist/esm/client.d.ts +1 -1
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -0
- package/dist/esm/index.js +4 -0
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/middlewares/otel.d.ts +8 -2
- package/dist/esm/middlewares/otel.js +145 -95
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/middlewares/usage-attributes.d.ts +24 -0
- package/dist/esm/middlewares/usage-attributes.js +43 -0
- package/dist/esm/middlewares/usage-attributes.js.map +1 -0
- package/dist/esm/types.d.ts +103 -14
- package/dist/esm/utilities/errors.d.ts +13 -0
- package/dist/esm/utilities/errors.js +22 -0
- package/dist/esm/utilities/errors.js.map +1 -0
- package/dist/esm/utilities/media-prompt.d.ts +35 -0
- package/dist/esm/utilities/media-prompt.js +43 -0
- package/dist/esm/utilities/media-prompt.js.map +1 -0
- package/dist/esm/utilities/numbers.d.ts +8 -0
- package/dist/esm/utilities/numbers.js +12 -0
- package/dist/esm/utilities/numbers.js.map +1 -0
- package/package.json +2 -2
- package/skills/ai-core/media-generation/SKILL.md +173 -3
- package/src/activities/chat/index.ts +32 -4
- package/src/activities/chat/middleware/types.ts +7 -0
- package/src/activities/chat/tools/lazy-tool-manager.ts +46 -4
- package/src/activities/generateAudio/index.ts +42 -1
- package/src/activities/generateImage/adapter.ts +16 -3
- package/src/activities/generateImage/index.ts +90 -5
- package/src/activities/generateSpeech/index.ts +42 -1
- package/src/activities/generateTranscription/index.ts +42 -1
- package/src/activities/generateVideo/adapter.ts +80 -4
- package/src/activities/generateVideo/index.ts +141 -6
- package/src/activities/generateVideo/snap.ts +100 -0
- package/src/activities/index.ts +4 -0
- package/src/activities/middleware/index.ts +20 -0
- package/src/activities/middleware/run.ts +88 -0
- package/src/activities/middleware/types.ts +173 -0
- package/src/client.ts +4 -0
- package/src/index.ts +23 -0
- package/src/middlewares/otel.ts +195 -120
- package/src/middlewares/usage-attributes.ts +65 -0
- package/src/types.ts +126 -13
- package/src/utilities/errors.ts +29 -0
- package/src/utilities/media-prompt.ts +86 -0
- package/src/utilities/numbers.ts +15 -0
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Return the first candidate that is a finite `number`, or `undefined`.
|
|
3
|
+
*
|
|
4
|
+
* Handy for picking a value from among several possible spellings/sources where
|
|
5
|
+
* only some are populated — e.g. the provider-native sampling option names read
|
|
6
|
+
* by the OTel middleware, or the optional numeric fields on `TokenUsage`.
|
|
7
|
+
*/
|
|
8
|
+
export declare function firstNumber(...candidates: Array<unknown>): number | undefined;
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
function firstNumber(...candidates) {
|
|
2
|
+
for (const candidate of candidates) {
|
|
3
|
+
if (typeof candidate === "number" && Number.isFinite(candidate)) {
|
|
4
|
+
return candidate;
|
|
5
|
+
}
|
|
6
|
+
}
|
|
7
|
+
return void 0;
|
|
8
|
+
}
|
|
9
|
+
export {
|
|
10
|
+
firstNumber
|
|
11
|
+
};
|
|
12
|
+
//# sourceMappingURL=numbers.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"numbers.js","sources":["../../../src/utilities/numbers.ts"],"sourcesContent":["/**\n * Return the first candidate that is a finite `number`, or `undefined`.\n *\n * Handy for picking a value from among several possible spellings/sources where\n * only some are populated — e.g. the provider-native sampling option names read\n * by the OTel middleware, or the optional numeric fields on `TokenUsage`.\n */\nexport function firstNumber(...candidates: Array<unknown>): number | undefined {\n for (const candidate of candidates) {\n if (typeof candidate === 'number' && Number.isFinite(candidate)) {\n return candidate\n }\n }\n return undefined\n}\n"],"names":[],"mappings":"AAOO,SAAS,eAAe,YAAgD;AAC7E,aAAW,aAAa,YAAY;AAClC,QAAI,OAAO,cAAc,YAAY,OAAO,SAAS,SAAS,GAAG;AAC/D,aAAO;AAAA,IACT;AAAA,EACF;AACA,SAAO;AACT;"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.33.0",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -76,7 +76,7 @@
|
|
|
76
76
|
"@ag-ui/core": "^0.0.52",
|
|
77
77
|
"@standard-schema/spec": "^1.1.0",
|
|
78
78
|
"partial-json": "^0.1.7",
|
|
79
|
-
"@tanstack/ai-event-client": "0.6.
|
|
79
|
+
"@tanstack/ai-event-client": "0.6.4"
|
|
80
80
|
},
|
|
81
81
|
"peerDependencies": {
|
|
82
82
|
"@opentelemetry/api": ">=1.9.0"
|
|
@@ -3,8 +3,9 @@ name: ai-core/media-generation
|
|
|
3
3
|
description: >
|
|
4
4
|
Image, audio, video, speech (TTS), and transcription generation using
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage,
|
|
6
|
-
generateAudio() with geminiAudio/falAudio, generateVideo() with
|
|
7
|
-
polling,
|
|
6
|
+
generateAudio() with geminiAudio/falAudio, generateVideo() with
|
|
7
|
+
openaiVideo/geminiVideo (async polling, per-model typed durations),
|
|
8
|
+
generateSpeech() with openaiSpeech, generateTranscription() with
|
|
8
9
|
openaiTranscription. React hooks: useGenerateImage, useGenerateAudio,
|
|
9
10
|
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
10
11
|
TanStack Start server function integration with toServerSentEventsResponse.
|
|
@@ -189,6 +190,103 @@ Result shape: `ImageGenerationResult` with `images` array where each entry
|
|
|
189
190
|
has `b64Json?`, `url?`, and `revisedPrompt?`. OpenAI image URLs expire
|
|
190
191
|
after 1 hour -- download or display immediately.
|
|
191
192
|
|
|
193
|
+
#### Image-conditioned generation: multimodal `prompt` parts
|
|
194
|
+
|
|
195
|
+
Both `generateImage()` and `generateVideo()` accept the `prompt` either as
|
|
196
|
+
a plain string or as an ordered array of content parts (`TextPart` /
|
|
197
|
+
`ImagePart` / `VideoPart` / `AudioPart` — the same shapes used elsewhere in
|
|
198
|
+
TanStack AI). Part order is meaningful: natively multimodal providers
|
|
199
|
+
(Gemini, OpenRouter) receive parts in order; named-field providers (OpenAI,
|
|
200
|
+
fal, xAI) extract media parts and flatten the text. Prompt text is always
|
|
201
|
+
sent verbatim — to reference inputs from the prompt, write the provider's
|
|
202
|
+
own syntax (fal `@Image1`, OpenAI "image 1" prose); the SDK never injects
|
|
203
|
+
or rewrites markers. Each media part may carry an optional
|
|
204
|
+
`metadata.role` hint that adapters use to route the part to the
|
|
205
|
+
provider-specific field. The accepted part types are narrowed per model at
|
|
206
|
+
compile time via the adapter's input-modality map.
|
|
207
|
+
|
|
208
|
+
```typescript
|
|
209
|
+
import { generateImage } from '@tanstack/ai'
|
|
210
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
211
|
+
|
|
212
|
+
// Image-to-image (OpenAI gpt-image-2 / gpt-image-1, dall-e-2)
|
|
213
|
+
await generateImage({
|
|
214
|
+
adapter: openaiImage('gpt-image-2'),
|
|
215
|
+
prompt: [
|
|
216
|
+
{ type: 'text', content: 'Turn this into a cinematic product photo' },
|
|
217
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
|
|
218
|
+
],
|
|
219
|
+
})
|
|
220
|
+
|
|
221
|
+
// Multi-reference (up to 16 for gpt-image models; up to ~14 for Gemini native
|
|
222
|
+
// — a provider limit, not enforced by the SDK)
|
|
223
|
+
await generateImage({
|
|
224
|
+
adapter: openaiImage('gpt-image-2'),
|
|
225
|
+
prompt: [
|
|
226
|
+
{ type: 'text', content: 'Apply the second image as style to the first' },
|
|
227
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
|
|
228
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/style.png' } },
|
|
229
|
+
],
|
|
230
|
+
})
|
|
231
|
+
|
|
232
|
+
// Inpaint via metadata.role === 'mask' (OpenAI gpt-image models, dall-e-2; fal mask_url)
|
|
233
|
+
await generateImage({
|
|
234
|
+
adapter: openaiImage('gpt-image-2'),
|
|
235
|
+
prompt: [
|
|
236
|
+
{ type: 'text', content: 'Replace the masked region with a tree' },
|
|
237
|
+
{ type: 'image', source: { type: 'url', value: photoUrl } },
|
|
238
|
+
{
|
|
239
|
+
type: 'image',
|
|
240
|
+
source: { type: 'url', value: maskUrl },
|
|
241
|
+
metadata: { role: 'mask' },
|
|
242
|
+
},
|
|
243
|
+
],
|
|
244
|
+
})
|
|
245
|
+
|
|
246
|
+
// Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional end_image_url)
|
|
247
|
+
import { generateVideo } from '@tanstack/ai'
|
|
248
|
+
import { falVideo } from '@tanstack/ai-fal'
|
|
249
|
+
|
|
250
|
+
await generateVideo({
|
|
251
|
+
adapter: falVideo('fal-ai/kling-video/v3/pro/image-to-video'),
|
|
252
|
+
prompt: [
|
|
253
|
+
{ type: 'image', source: { type: 'url', value: firstFrameUrl } },
|
|
254
|
+
{ type: 'text', content: 'Slow cinematic push-in' },
|
|
255
|
+
{
|
|
256
|
+
type: 'image',
|
|
257
|
+
source: { type: 'url', value: lastFrameUrl },
|
|
258
|
+
metadata: { role: 'end_frame' },
|
|
259
|
+
},
|
|
260
|
+
],
|
|
261
|
+
})
|
|
262
|
+
```
|
|
263
|
+
|
|
264
|
+
**Role hints** (`metadata.role`):
|
|
265
|
+
|
|
266
|
+
| Role | Maps to |
|
|
267
|
+
| --------------- | ----------------------------------------------------------------------------------------------------- |
|
|
268
|
+
| `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise |
|
|
269
|
+
| `'character'` | Same as `'reference'`; Veo `referenceImages` slot (planned — no Veo adapter yet) |
|
|
270
|
+
| `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
|
|
271
|
+
| `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
|
|
272
|
+
| `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image` (planned) |
|
|
273
|
+
| `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame` (planned) |
|
|
274
|
+
|
|
275
|
+
**Provider support matrix:**
|
|
276
|
+
|
|
277
|
+
| Provider | `generateImage` image parts | `generateVideo` image parts |
|
|
278
|
+
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
279
|
+
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
|
|
280
|
+
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | No native Veo adapter yet — deferred to a follow-up. |
|
|
281
|
+
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
|
|
282
|
+
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
|
|
283
|
+
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | n/a |
|
|
284
|
+
| Anthropic | n/a (no image generation API). | n/a |
|
|
285
|
+
|
|
286
|
+
Video and audio prompt parts follow the same `metadata.role` convention
|
|
287
|
+
for video-to-video and lipsync flows on fal; other providers throw when
|
|
288
|
+
they're passed.
|
|
289
|
+
|
|
192
290
|
### 2. Audio Generation (Music, Sound Effects)
|
|
193
291
|
|
|
194
292
|
Distinct from TTS — `generateAudio()` produces non-speech audio content.
|
|
@@ -331,6 +429,31 @@ const stream = generateVideo({
|
|
|
331
429
|
return toServerSentEventsResponse(stream)
|
|
332
430
|
```
|
|
333
431
|
|
|
432
|
+
Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
|
|
433
|
+
`duration` option is typed per model (e.g. `4 | 6 | 8` for Veo 3.x,
|
|
434
|
+
`5 | 6 | 8` for Veo 2); use `adapter.snapDuration(seconds)` to coerce raw
|
|
435
|
+
seconds and `adapter.availableDurations()` to enumerate the valid set.
|
|
436
|
+
Image prompt parts route by `metadata.role`: first un-roled /
|
|
437
|
+
`'start_frame'` image → input image, `'end_frame'` → `lastFrame`,
|
|
438
|
+
`'reference'` / `'character'` → `referenceImages`:
|
|
439
|
+
|
|
440
|
+
```typescript
|
|
441
|
+
import { geminiVideo } from '@tanstack/ai-gemini'
|
|
442
|
+
|
|
443
|
+
const adapter = geminiVideo('veo-3.1-generate-preview')
|
|
444
|
+
adapter.availableDurations() // { kind: 'discrete', values: [4, 6, 8] }
|
|
445
|
+
|
|
446
|
+
const { jobId } = await generateVideo({
|
|
447
|
+
adapter,
|
|
448
|
+
prompt: 'A golden retriever playing in sunflowers',
|
|
449
|
+
size: '16:9', // Veo sizes are aspect ratios: '16:9' | '9:16'
|
|
450
|
+
duration: adapter.snapDuration(7), // 6
|
|
451
|
+
modelOptions: { resolution: '1080p', generateAudio: true },
|
|
452
|
+
})
|
|
453
|
+
// Note: Veo result URLs require the Google API key to download
|
|
454
|
+
// (x-goog-api-key header or ?key= query parameter).
|
|
455
|
+
```
|
|
456
|
+
|
|
334
457
|
Client hook with job tracking:
|
|
335
458
|
|
|
336
459
|
```tsx
|
|
@@ -607,7 +730,54 @@ generateSpeech({
|
|
|
607
730
|
|
|
608
731
|
> Source: Gemini TTS adapter validation; CodeRabbit review of PR #463.
|
|
609
732
|
|
|
610
|
-
### h.
|
|
733
|
+
### h. HIGH: Passing image prompt parts to a model that doesn't support image-conditioned generation
|
|
734
|
+
|
|
735
|
+
Not every model accepts image-conditioned prompts. The `prompt` type is
|
|
736
|
+
narrowed per model, so passing an image part to a text-only model
|
|
737
|
+
(dall-e-3, Imagen, grok-2-image) is a **compile-time error**; adapters
|
|
738
|
+
also throw a clear runtime error as a backstop, so users learn at call
|
|
739
|
+
time rather than getting silently wrong output.
|
|
740
|
+
|
|
741
|
+
```typescript
|
|
742
|
+
// WRONG — dall-e-3 has no edit/inputs API; image parts are a type error
|
|
743
|
+
generateImage({
|
|
744
|
+
adapter: openaiImage('dall-e-3'),
|
|
745
|
+
prompt: [
|
|
746
|
+
{ type: 'text', content: 'Edit this' },
|
|
747
|
+
{ type: 'image', source: { type: 'url', value: url } }, // ❌ type error
|
|
748
|
+
],
|
|
749
|
+
})
|
|
750
|
+
|
|
751
|
+
// WRONG — Imagen is text-to-image only; same compile-time rejection
|
|
752
|
+
generateImage({
|
|
753
|
+
adapter: geminiImage('imagen-4.0-generate-001'),
|
|
754
|
+
prompt: [
|
|
755
|
+
{ type: 'text', content: 'Edit this' },
|
|
756
|
+
{ type: 'image', source: { type: 'url', value: url } }, // ❌ type error
|
|
757
|
+
],
|
|
758
|
+
})
|
|
759
|
+
|
|
760
|
+
// CORRECT — use a model that supports image-conditioned generation
|
|
761
|
+
generateImage({
|
|
762
|
+
adapter: openaiImage('gpt-image-2'), // edits up to 16 images
|
|
763
|
+
prompt: [
|
|
764
|
+
{ type: 'text', content: 'Edit this' },
|
|
765
|
+
{ type: 'image', source: { type: 'url', value: url } },
|
|
766
|
+
],
|
|
767
|
+
})
|
|
768
|
+
|
|
769
|
+
generateImage({
|
|
770
|
+
adapter: geminiImage('gemini-3.1-flash-image-preview'), // native multimodal
|
|
771
|
+
prompt: [
|
|
772
|
+
{ type: 'text', content: 'Edit this' },
|
|
773
|
+
{ type: 'image', source: { type: 'url', value: url } },
|
|
774
|
+
],
|
|
775
|
+
})
|
|
776
|
+
```
|
|
777
|
+
|
|
778
|
+
> Source: docs/media/image-generation.md, docs/media/video-generation.md.
|
|
779
|
+
|
|
780
|
+
### i. LOW: Writing a logging middleware to see media chunks flow through
|
|
611
781
|
|
|
612
782
|
Every media activity — `generateAudio`, `generateSpeech`,
|
|
613
783
|
`generateTranscription`, `generateImage`, `generateVideo` — accepts the
|
|
@@ -33,7 +33,11 @@ import type {
|
|
|
33
33
|
ClientToolRequest,
|
|
34
34
|
ToolResult,
|
|
35
35
|
} from './tools/tool-calls'
|
|
36
|
-
import type {
|
|
36
|
+
import type {
|
|
37
|
+
AnyTextAdapter,
|
|
38
|
+
StructuredOutputOptions,
|
|
39
|
+
StructuredOutputResult,
|
|
40
|
+
} from './adapter'
|
|
37
41
|
import type {
|
|
38
42
|
AgentLoopStrategy,
|
|
39
43
|
AnyTool,
|
|
@@ -646,6 +650,7 @@ class TextEngine<
|
|
|
646
650
|
this.deferredPromises.push(promise)
|
|
647
651
|
},
|
|
648
652
|
// Provider / adapter info
|
|
653
|
+
activity: 'chat',
|
|
649
654
|
provider: config.adapter.name,
|
|
650
655
|
model: config.params.model,
|
|
651
656
|
source: 'server',
|
|
@@ -1197,6 +1202,23 @@ class TextEngine<
|
|
|
1197
1202
|
}
|
|
1198
1203
|
}
|
|
1199
1204
|
|
|
1205
|
+
/**
|
|
1206
|
+
* Tools available for execution this turn. The discovery tool is dropped
|
|
1207
|
+
* from the advertised set (`this.tools`) once every lazy tool is discovered,
|
|
1208
|
+
* but a model may still re-request discovery; this widens execution lookup
|
|
1209
|
+
* to include it so such calls don't fail with "Unknown tool". Centralised so
|
|
1210
|
+
* both execution sites (`processToolCalls` and `checkForPendingToolCalls`)
|
|
1211
|
+
* stay in sync.
|
|
1212
|
+
*/
|
|
1213
|
+
private resolveExecutableTools(
|
|
1214
|
+
toolCalls: ReadonlyArray<ToolCall>,
|
|
1215
|
+
): ReadonlyArray<AnyTool> {
|
|
1216
|
+
return this.lazyToolManager.getExecutableTools(
|
|
1217
|
+
this.tools,
|
|
1218
|
+
toolCalls.map((tc) => tc.function.name),
|
|
1219
|
+
)
|
|
1220
|
+
}
|
|
1221
|
+
|
|
1200
1222
|
private async *checkForPendingToolCalls(): AsyncGenerator<
|
|
1201
1223
|
StreamChunk,
|
|
1202
1224
|
ToolPhaseResult,
|
|
@@ -1245,7 +1267,7 @@ class TextEngine<
|
|
|
1245
1267
|
|
|
1246
1268
|
const generator = executeToolCalls(
|
|
1247
1269
|
executablePendingCalls,
|
|
1248
|
-
this.
|
|
1270
|
+
this.resolveExecutableTools(executablePendingCalls),
|
|
1249
1271
|
approvals,
|
|
1250
1272
|
clientToolResults,
|
|
1251
1273
|
(eventName, data) => this.createCustomEventChunk(eventName, data),
|
|
@@ -1407,7 +1429,7 @@ class TextEngine<
|
|
|
1407
1429
|
|
|
1408
1430
|
const generator = executeToolCalls(
|
|
1409
1431
|
executableToolCalls,
|
|
1410
|
-
this.
|
|
1432
|
+
this.resolveExecutableTools(executableToolCalls),
|
|
1411
1433
|
approvals,
|
|
1412
1434
|
clientToolResults,
|
|
1413
1435
|
(eventName, data) => this.createCustomEventChunk(eventName, data),
|
|
@@ -2861,7 +2883,7 @@ async function* fallbackStructuredOutputStream(
|
|
|
2861
2883
|
timestamp,
|
|
2862
2884
|
}
|
|
2863
2885
|
|
|
2864
|
-
let result:
|
|
2886
|
+
let result: StructuredOutputResult<unknown>
|
|
2865
2887
|
try {
|
|
2866
2888
|
result = await adapter.structuredOutput(options)
|
|
2867
2889
|
} catch (error) {
|
|
@@ -2917,6 +2939,12 @@ async function* fallbackStructuredOutputStream(
|
|
|
2917
2939
|
model,
|
|
2918
2940
|
timestamp,
|
|
2919
2941
|
finishReason: 'stop',
|
|
2942
|
+
// Forward adapter-reported token usage so consumers reading
|
|
2943
|
+
// `RUN_FINISHED.usage` (and the engine's `runOnUsage` middleware hook) see
|
|
2944
|
+
// it on the fallback path, mirroring the native streaming path. The
|
|
2945
|
+
// conditional spread avoids emitting `usage: undefined` for adapters that
|
|
2946
|
+
// don't report it. See #758.
|
|
2947
|
+
...(result.usage ? { usage: result.usage } : {}),
|
|
2920
2948
|
}
|
|
2921
2949
|
}
|
|
2922
2950
|
|
|
@@ -80,6 +80,13 @@ export interface ChatMiddlewareContext<TContext = unknown> {
|
|
|
80
80
|
|
|
81
81
|
// --- Provider / adapter info (immutable for the lifetime of the request) ---
|
|
82
82
|
|
|
83
|
+
/**
|
|
84
|
+
* Which activity this context describes — always `'chat'`. Present so the
|
|
85
|
+
* chat context structurally satisfies the base `GenerationMiddlewareContext`,
|
|
86
|
+
* letting an observe-only middleware authored against the base (e.g.
|
|
87
|
+
* `otelMiddleware`) run on both chat and media activities.
|
|
88
|
+
*/
|
|
89
|
+
activity: 'chat'
|
|
83
90
|
/** Provider name (e.g., 'openai', 'anthropic') */
|
|
84
91
|
provider: string
|
|
85
92
|
/** Model identifier (e.g., 'gpt-4o') */
|
|
@@ -1,7 +1,14 @@
|
|
|
1
1
|
import { convertSchemaToJsonSchema } from './schema-converter'
|
|
2
|
-
import type { Tool } from '../../../types'
|
|
2
|
+
import type { AnyTool, Tool } from '../../../types'
|
|
3
3
|
|
|
4
|
-
|
|
4
|
+
/**
|
|
5
|
+
* Name of the synthetic tool the LLM calls to discover lazy tools.
|
|
6
|
+
*
|
|
7
|
+
* Exported so callers building custom message-compaction / history-trimming
|
|
8
|
+
* logic can reference the discovery tool by constant instead of hard-coding
|
|
9
|
+
* the string (which is an internal contract that could change).
|
|
10
|
+
*/
|
|
11
|
+
export const DISCOVERY_TOOL_NAME = '__lazy__tool__discovery__'
|
|
5
12
|
|
|
6
13
|
/**
|
|
7
14
|
* Manages lazy tool discovery for the chat agent loop.
|
|
@@ -87,6 +94,35 @@ export class LazyToolManager {
|
|
|
87
94
|
return active
|
|
88
95
|
}
|
|
89
96
|
|
|
97
|
+
/**
|
|
98
|
+
* Returns the tools that should be available for *execution* this turn.
|
|
99
|
+
*
|
|
100
|
+
* This is the advertised set (`getActiveTools()`, passed in as `activeTools`)
|
|
101
|
+
* plus the discovery tool when a pending call references it but it is no
|
|
102
|
+
* longer advertised. Once every lazy tool has been discovered the discovery
|
|
103
|
+
* tool is dropped from the advertised set, but a model may still re-request
|
|
104
|
+
* discovery (long context / hallucination); keeping it executable lets that
|
|
105
|
+
* call return the schemas again instead of failing with "Unknown tool".
|
|
106
|
+
*
|
|
107
|
+
* The advertised set is intentionally left unchanged — only execution lookup
|
|
108
|
+
* is widened. Operates on the already-built `activeTools`: it must NOT call
|
|
109
|
+
* `getActiveTools()`, which would reset `hasNewDiscoveries` before the
|
|
110
|
+
* post-execution refresh check in the agent loop.
|
|
111
|
+
*/
|
|
112
|
+
getExecutableTools(
|
|
113
|
+
activeTools: ReadonlyArray<AnyTool>,
|
|
114
|
+
pendingToolCallNames: ReadonlyArray<string>,
|
|
115
|
+
): ReadonlyArray<AnyTool> {
|
|
116
|
+
if (
|
|
117
|
+
this.discoveryTool &&
|
|
118
|
+
pendingToolCallNames.includes(DISCOVERY_TOOL_NAME) &&
|
|
119
|
+
!activeTools.some((t) => t.name === DISCOVERY_TOOL_NAME)
|
|
120
|
+
) {
|
|
121
|
+
return [...activeTools, this.discoveryTool]
|
|
122
|
+
}
|
|
123
|
+
return activeTools
|
|
124
|
+
}
|
|
125
|
+
|
|
90
126
|
/**
|
|
91
127
|
* Returns whether new tools have been discovered since the last getActiveTools() call.
|
|
92
128
|
*/
|
|
@@ -221,8 +257,14 @@ export class LazyToolManager {
|
|
|
221
257
|
for (const name of args.toolNames) {
|
|
222
258
|
const tool = lazyToolMap.get(name)
|
|
223
259
|
if (tool) {
|
|
224
|
-
|
|
225
|
-
|
|
260
|
+
// Only flag a refresh for genuinely new discoveries. Re-requesting
|
|
261
|
+
// an already-discovered tool still returns its schema below (the
|
|
262
|
+
// model asked for it), but must not trigger a redundant tool-list
|
|
263
|
+
// refresh + continue in the agent loop.
|
|
264
|
+
if (!manager.discoveredTools.has(name)) {
|
|
265
|
+
manager.discoveredTools.add(name)
|
|
266
|
+
manager.hasNewDiscoveries = true
|
|
267
|
+
}
|
|
226
268
|
const jsonSchema = tool.inputSchema
|
|
227
269
|
? convertSchemaToJsonSchema(tool.inputSchema)
|
|
228
270
|
: undefined
|
|
@@ -8,8 +8,16 @@
|
|
|
8
8
|
import { aiEventClient } from '@tanstack/ai-event-client'
|
|
9
9
|
import { streamGenerationResult } from '../stream-generation-result.js'
|
|
10
10
|
import { resolveDebugOption } from '../../logger/resolve'
|
|
11
|
+
import {
|
|
12
|
+
createGenerationContext,
|
|
13
|
+
runGenerationError,
|
|
14
|
+
runGenerationFinish,
|
|
15
|
+
runGenerationStart,
|
|
16
|
+
runGenerationUsage,
|
|
17
|
+
} from '../middleware'
|
|
11
18
|
import type { InternalLogger } from '../../logger/internal-logger'
|
|
12
19
|
import type { DebugOption } from '../../logger/types'
|
|
20
|
+
import type { GenerationMiddleware } from '../middleware'
|
|
13
21
|
import type { AudioAdapter } from './adapter'
|
|
14
22
|
import type { AudioGenerationResult, StreamChunk } from '../../types'
|
|
15
23
|
|
|
@@ -70,6 +78,12 @@ export interface AudioActivityOptions<
|
|
|
70
78
|
* control and/or a custom `Logger`.
|
|
71
79
|
*/
|
|
72
80
|
debug?: DebugOption
|
|
81
|
+
/**
|
|
82
|
+
* Observe-only middleware notified on start, usage, success, and error. Pass
|
|
83
|
+
* `otelMiddleware()` to emit OpenTelemetry spans, or implement the
|
|
84
|
+
* `GenerationMiddleware` contract for a custom backend.
|
|
85
|
+
*/
|
|
86
|
+
middleware?: Array<GenerationMiddleware>
|
|
73
87
|
}
|
|
74
88
|
|
|
75
89
|
// ===========================
|
|
@@ -135,7 +149,13 @@ async function runGenerateAudio<
|
|
|
135
149
|
>(
|
|
136
150
|
options: AudioActivityOptions<TAdapter, boolean>,
|
|
137
151
|
): Promise<AudioGenerationResult> {
|
|
138
|
-
const {
|
|
152
|
+
const {
|
|
153
|
+
adapter,
|
|
154
|
+
stream: _stream,
|
|
155
|
+
debug: _debug,
|
|
156
|
+
middleware,
|
|
157
|
+
...rest
|
|
158
|
+
} = options
|
|
139
159
|
const model = adapter.model
|
|
140
160
|
const requestId = createId('audio')
|
|
141
161
|
const startTime = Date.now()
|
|
@@ -145,6 +165,17 @@ async function runGenerateAudio<
|
|
|
145
165
|
(adapter as { name?: string }).name ??
|
|
146
166
|
'unknown'
|
|
147
167
|
|
|
168
|
+
const mwCtx = createGenerationContext({
|
|
169
|
+
requestId,
|
|
170
|
+
activity: 'audio',
|
|
171
|
+
provider: adapter.name,
|
|
172
|
+
model,
|
|
173
|
+
modelOptions: rest.modelOptions,
|
|
174
|
+
createId,
|
|
175
|
+
})
|
|
176
|
+
|
|
177
|
+
await runGenerationStart(middleware, mwCtx)
|
|
178
|
+
|
|
148
179
|
aiEventClient.emit('audio:request:started', {
|
|
149
180
|
requestId,
|
|
150
181
|
provider: adapter.name,
|
|
@@ -189,6 +220,12 @@ async function runGenerateAudio<
|
|
|
189
220
|
audioDuration: result.audio.duration,
|
|
190
221
|
})
|
|
191
222
|
|
|
223
|
+
if (result.usage) await runGenerationUsage(middleware, mwCtx, result.usage)
|
|
224
|
+
await runGenerationFinish(middleware, mwCtx, {
|
|
225
|
+
duration: elapsedMs,
|
|
226
|
+
usage: result.usage,
|
|
227
|
+
})
|
|
228
|
+
|
|
192
229
|
return result
|
|
193
230
|
} catch (error) {
|
|
194
231
|
const elapsedMs = Date.now() - startTime
|
|
@@ -202,6 +239,10 @@ async function runGenerateAudio<
|
|
|
202
239
|
modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
|
|
203
240
|
timestamp: Date.now(),
|
|
204
241
|
})
|
|
242
|
+
await runGenerationError(middleware, mwCtx, {
|
|
243
|
+
error,
|
|
244
|
+
duration: elapsedMs,
|
|
245
|
+
})
|
|
205
246
|
logger.errors('generateAudio activity failed', {
|
|
206
247
|
error,
|
|
207
248
|
source: 'generateAudio',
|
|
@@ -1,4 +1,8 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type {
|
|
2
|
+
ImageGenerationOptions,
|
|
3
|
+
ImageGenerationResult,
|
|
4
|
+
ModelInputModalitiesByName,
|
|
5
|
+
} from '../../types'
|
|
2
6
|
|
|
3
7
|
/**
|
|
4
8
|
* Resolve the size type for a model from the model-size map.
|
|
@@ -29,6 +33,8 @@ export interface ImageAdapterConfig {
|
|
|
29
33
|
* - TProviderOptions: Base provider-specific options (already resolved)
|
|
30
34
|
* - TModelProviderOptionsByName: Map from model name to its specific provider options
|
|
31
35
|
* - TModelSizeByName: Map from model name to its supported sizes
|
|
36
|
+
* - TModelInputModalitiesByName: Map from model name to the non-text prompt
|
|
37
|
+
* modalities it accepts (constrains the `prompt` part types at compile time)
|
|
32
38
|
*/
|
|
33
39
|
export interface ImageAdapter<
|
|
34
40
|
TModel extends string = string,
|
|
@@ -38,6 +44,8 @@ export interface ImageAdapter<
|
|
|
38
44
|
string,
|
|
39
45
|
string
|
|
40
46
|
>,
|
|
47
|
+
TModelInputModalitiesByName extends ModelInputModalitiesByName =
|
|
48
|
+
ModelInputModalitiesByName,
|
|
41
49
|
> {
|
|
42
50
|
/** Discriminator for adapter kind - used by generate() to determine API shape */
|
|
43
51
|
readonly kind: 'image'
|
|
@@ -53,6 +61,7 @@ export interface ImageAdapter<
|
|
|
53
61
|
providerOptions: TProviderOptions
|
|
54
62
|
modelProviderOptionsByName: TModelProviderOptionsByName
|
|
55
63
|
modelSizeByName: TModelSizeByName
|
|
64
|
+
modelInputModalitiesByName: TModelInputModalitiesByName
|
|
56
65
|
}
|
|
57
66
|
|
|
58
67
|
/**
|
|
@@ -67,7 +76,7 @@ export interface ImageAdapter<
|
|
|
67
76
|
* An ImageAdapter with any/unknown type parameters.
|
|
68
77
|
* Useful as a constraint in generic functions and interfaces.
|
|
69
78
|
*/
|
|
70
|
-
export type AnyImageAdapter = ImageAdapter<any, any, any, any>
|
|
79
|
+
export type AnyImageAdapter = ImageAdapter<any, any, any, any, any>
|
|
71
80
|
|
|
72
81
|
/**
|
|
73
82
|
* Abstract base class for image generation adapters.
|
|
@@ -83,11 +92,14 @@ export abstract class BaseImageAdapter<
|
|
|
83
92
|
string,
|
|
84
93
|
string
|
|
85
94
|
>,
|
|
95
|
+
TModelInputModalitiesByName extends ModelInputModalitiesByName =
|
|
96
|
+
ModelInputModalitiesByName,
|
|
86
97
|
> implements ImageAdapter<
|
|
87
98
|
TModel,
|
|
88
99
|
TProviderOptions,
|
|
89
100
|
TModelProviderOptionsByName,
|
|
90
|
-
TModelSizeByName
|
|
101
|
+
TModelSizeByName,
|
|
102
|
+
TModelInputModalitiesByName
|
|
91
103
|
> {
|
|
92
104
|
readonly kind = 'image' as const
|
|
93
105
|
abstract readonly name: string
|
|
@@ -98,6 +110,7 @@ export abstract class BaseImageAdapter<
|
|
|
98
110
|
providerOptions: TProviderOptions
|
|
99
111
|
modelProviderOptionsByName: TModelProviderOptionsByName
|
|
100
112
|
modelSizeByName: TModelSizeByName
|
|
113
|
+
modelInputModalitiesByName: TModelInputModalitiesByName
|
|
101
114
|
}
|
|
102
115
|
|
|
103
116
|
protected config: ImageAdapterConfig
|