@tanstack/ai 0.54.0 → 0.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -16
- package/dist/esm/activities/chat/index.js +17 -1
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/messages.d.ts +21 -1
- package/dist/esm/activities/chat/messages.js +50 -1
- package/dist/esm/activities/chat/messages.js.map +1 -1
- package/dist/esm/activities/chat/stream/processor.d.ts +17 -0
- package/dist/esm/activities/chat/stream/processor.js +27 -0
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/dist/esm/activities/chat/tools/tool-calls.js +1 -0
- package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
- package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
- package/dist/esm/activities/evaluate/adapter.js +23 -0
- package/dist/esm/activities/evaluate/adapter.js.map +1 -0
- package/dist/esm/activities/evaluate/index.d.ts +255 -0
- package/dist/esm/activities/evaluate/index.js +317 -0
- package/dist/esm/activities/evaluate/index.js.map +1 -0
- package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
- package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
- package/dist/esm/activities/generateSpeech/index.js +53 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
- package/dist/esm/activities/generateVoice/adapter.js +23 -0
- package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
- package/dist/esm/activities/generateVoice/index.d.ts +133 -0
- package/dist/esm/activities/generateVoice/index.js +184 -0
- package/dist/esm/activities/generateVoice/index.js.map +1 -0
- package/dist/esm/activities/index.d.ts +10 -4
- package/dist/esm/activities/index.js +14 -10
- package/dist/esm/activities/middleware/types.d.ts +1 -1
- package/dist/esm/client.d.ts +3 -2
- package/dist/esm/client.js +21 -3
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -2
- package/dist/esm/index.js +5 -2
- package/dist/esm/middlewares/otel.js +2 -0
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/realtime/index.d.ts +1 -1
- package/dist/esm/realtime/index.js +1 -1
- package/dist/esm/realtime/index.js.map +1 -1
- package/dist/esm/types.d.ts +225 -2
- package/package.json +2 -2
- package/skills/ai-core/adapter-configuration/SKILL.md +1 -1
- package/skills/ai-core/media-generation/SKILL.md +154 -18
- package/src/activities/chat/index.ts +23 -0
- package/src/activities/chat/messages.ts +60 -0
- package/src/activities/chat/stream/processor.ts +31 -0
- package/src/activities/chat/tools/tool-calls.ts +9 -0
- package/src/activities/evaluate/adapter.ts +212 -0
- package/src/activities/evaluate/index.ts +614 -0
- package/src/activities/generateSpeech/adapter.ts +47 -1
- package/src/activities/generateSpeech/index.ts +149 -8
- package/src/activities/generateVoice/adapter.ts +89 -0
- package/src/activities/generateVoice/index.ts +371 -0
- package/src/activities/index.ts +69 -0
- package/src/activities/middleware/types.ts +2 -0
- package/src/client.ts +35 -8
- package/src/index.ts +21 -0
- package/src/middlewares/otel.ts +2 -0
- package/src/realtime/index.ts +1 -1
- package/src/types.ts +246 -2
|
@@ -5,8 +5,10 @@ description: >
|
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
|
|
6
6
|
generateAudio() with geminiAudio/falAudio, generateVideo() with async
|
|
7
7
|
polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
|
|
8
|
-
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
|
|
9
|
-
generateTranscription() with openaiTranscription/byteplusTranscription
|
|
8
|
+
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech/elevenlabsSpeech,
|
|
9
|
+
generateTranscription() with openaiTranscription/byteplusTranscription,
|
|
10
|
+
generateVoice() with elevenlabsVoiceDesign (create a voice, then speak with it).
|
|
11
|
+
React hooks:
|
|
10
12
|
useGenerateImage, useGenerateAudio,
|
|
11
13
|
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
12
14
|
TanStack Start server function integration with toServerSentEventsResponse.
|
|
@@ -20,6 +22,8 @@ sources:
|
|
|
20
22
|
- 'TanStack/ai:docs/media/audio-generation.md'
|
|
21
23
|
- 'TanStack/ai:docs/media/video-generation.md'
|
|
22
24
|
- 'TanStack/ai:docs/media/text-to-speech.md'
|
|
25
|
+
- 'TanStack/ai:docs/adapters/elevenlabs.md'
|
|
26
|
+
- 'TanStack/ai:docs/media/voice-creation.md'
|
|
23
27
|
- 'TanStack/ai:docs/media/transcription.md'
|
|
24
28
|
- 'TanStack/ai:docs/advanced/debug-logging.md'
|
|
25
29
|
---
|
|
@@ -303,14 +307,14 @@ with `allowUrlFetch: true` on the adapter config
|
|
|
303
307
|
|
|
304
308
|
**Provider support matrix:**
|
|
305
309
|
|
|
306
|
-
| Provider | `generateImage` image parts | `generateVideo` image parts
|
|
307
|
-
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
308
|
-
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1.
|
|
309
|
-
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing).
|
|
310
|
-
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`.
|
|
311
|
-
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5).
|
|
312
|
-
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata.
|
|
313
|
-
| Anthropic | n/a (no image generation API). | n/a
|
|
310
|
+
| Provider | `generateImage` image parts | `generateVideo` image parts |
|
|
311
|
+
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
312
|
+
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
|
|
313
|
+
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
|
|
314
|
+
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
|
|
315
|
+
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). On 1.5 a starting frame can be combined with reference inputs (it pins the first frame). A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
|
|
316
|
+
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
|
|
317
|
+
| Anthropic | n/a (no image generation API). | n/a |
|
|
314
318
|
|
|
315
319
|
Video and audio prompt parts follow the same `metadata.role` convention
|
|
316
320
|
for video-to-video and lipsync flows on fal. Grok accepts one source
|
|
@@ -352,8 +356,14 @@ const { generate, result, isLoading } = useGenerateAudio({
|
|
|
352
356
|
|
|
353
357
|
### 3. Text-to-Speech
|
|
354
358
|
|
|
355
|
-
Adapters
|
|
356
|
-
`byteplusSpeech` (`seed-audio-1.0`).
|
|
359
|
+
Adapters include `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview),
|
|
360
|
+
`byteplusSpeech` (`seed-audio-1.0`), and `elevenlabsSpeech` (`eleven_v3`).
|
|
361
|
+
|
|
362
|
+
`elevenlabsSpeech` accepts `format: 'mp3' | 'pcm' | 'opus' | 'wav'`.
|
|
363
|
+
WAV output contains 44.1 kHz, 16-bit mono PCM with a RIFF header.
|
|
364
|
+
AAC and FLAC requests throw before the API call.
|
|
365
|
+
An explicit `modelOptions.outputFormat` overrides `format` and returns
|
|
366
|
+
the selected provider format without WAV wrapping.
|
|
357
367
|
|
|
358
368
|
> **BytePlus Seed Speech is a separate product from ModelArk** — it reads
|
|
359
369
|
> **`BYTEPLUS_VOICE_API_KEY`**, not `ARK_API_KEY`, and an Ark key there fails
|
|
@@ -364,7 +374,10 @@ Adapters: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview) and
|
|
|
364
374
|
> drops `voice`. Voice ids ending `_uranus_bigtts` are TTS 2.0,
|
|
365
375
|
> `_mars_bigtts` / `_moon_bigtts` are TTS 1.0, and `*_emo_v2_*` are the 1.0
|
|
366
376
|
> voices that accept emotion tags. Formats: `wav`, `mp3`, `pcm`, `ogg_opus`;
|
|
367
|
-
> `watermark`
|
|
377
|
+
> `modelOptions.watermark` takes an object here, not a boolean:
|
|
378
|
+
> `{ aigc_watermark }` for an audible marker and `{ aigc_metadata: { enable } }`
|
|
379
|
+
> for header provenance. `watermark: true` is shorthand for
|
|
380
|
+
> `{ aigc_watermark: true }`.
|
|
368
381
|
|
|
369
382
|
```typescript
|
|
370
383
|
import { generateSpeech } from '@tanstack/ai'
|
|
@@ -396,7 +409,126 @@ const { generate, result, isLoading } = useGenerateSpeech({
|
|
|
396
409
|
// Play: <audio src={`data:audio/${result.format};base64,${result.audio}`} controls />
|
|
397
410
|
```
|
|
398
411
|
|
|
399
|
-
|
|
412
|
+
**Dialogue (`turns`) and timings (`timestamps`).** `text` + `voice` is one
|
|
413
|
+
speaker. For a multi-voice script pass `turns` instead of `text` (they are
|
|
414
|
+
mutually exclusive), and set `timestamps: true` to get `result.alignment`
|
|
415
|
+
(per character or per word, `alignment.unit` says which) and `result.segments`
|
|
416
|
+
(one per turn or per sentence). All times are seconds.
|
|
417
|
+
|
|
418
|
+
```typescript
|
|
419
|
+
import { generateSpeech } from '@tanstack/ai'
|
|
420
|
+
import { byteplusSpeech } from '@tanstack/ai-byteplus'
|
|
421
|
+
|
|
422
|
+
// Second voice id comes from the BytePlus voice list.
|
|
423
|
+
const SECOND_VOICE = 'your-second-voice-id'
|
|
424
|
+
|
|
425
|
+
const result = await generateSpeech({
|
|
426
|
+
adapter: byteplusSpeech('seed-audio-1.0'),
|
|
427
|
+
turns: [
|
|
428
|
+
{ text: 'Do you sell picks?', voice: 'en_female_stokie_uranus_bigtts' },
|
|
429
|
+
{ text: 'By the till.', voice: SECOND_VOICE },
|
|
430
|
+
],
|
|
431
|
+
timestamps: true,
|
|
432
|
+
})
|
|
433
|
+
|
|
434
|
+
result.alignment?.endSeconds.at(-1) // where speech stops, not where the file does
|
|
435
|
+
result.segments?.[0] // { startSeconds, endSeconds, turnIndex?, voice?, text? }
|
|
436
|
+
```
|
|
437
|
+
|
|
438
|
+
Both are adapter capabilities, not universal. The activity rejects the request
|
|
439
|
+
before it reaches the provider when the adapter cannot do it, so read
|
|
440
|
+
`adapter.capabilities` rather than guessing:
|
|
441
|
+
|
|
442
|
+
| Adapter | `maxSpeakers` | `timestamps` |
|
|
443
|
+
| ----------------------- | ------------- | ------------------------------------------------ |
|
|
444
|
+
| `byteplusSpeech` | 3 | yes (`enable_subtitle`, word + sentence) |
|
|
445
|
+
| `elevenlabsSpeech` | 10 | yes (character, plus voice segments on dialogue) |
|
|
446
|
+
| `geminiSpeech` | 2 | no |
|
|
447
|
+
| every other TTS adapter | not supported | no |
|
|
448
|
+
|
|
449
|
+
### 4. Voice Creation
|
|
450
|
+
|
|
451
|
+
Adapter: `elevenlabsVoiceDesign` (`eleven_ttv_v3`, `eleven_multilingual_ttv_v2`).
|
|
452
|
+
|
|
453
|
+
`generateVoice()` makes a voice that does not exist in any catalog, either
|
|
454
|
+
from a text description or from a clip of a real speaker. It returns voice ids
|
|
455
|
+
you pass straight back to `generateSpeech()` as `voice`.
|
|
456
|
+
|
|
457
|
+
> Pass `prompt`, or `referenceAudio`, or both — the activity throws when
|
|
458
|
+
> neither is given. ElevenLabs always needs `prompt`, because its design
|
|
459
|
+
> endpoint requires a description, and only `eleven_ttv_v3` accepts
|
|
460
|
+
> `referenceAudio`. Without a `name` you get **previews**, which expire;
|
|
461
|
+
> with a `name` the best candidate is kept in the provider's voice library.
|
|
462
|
+
> Check `saved` on each returned voice rather than assuming. Remote audio
|
|
463
|
+
> URLs are rejected: read the file and pass bytes.
|
|
464
|
+
|
|
465
|
+
```typescript
|
|
466
|
+
import { generateSpeech, generateVoice } from '@tanstack/ai'
|
|
467
|
+
import {
|
|
468
|
+
elevenlabsSpeech,
|
|
469
|
+
elevenlabsVoiceDesign,
|
|
470
|
+
} from '@tanstack/ai-elevenlabs'
|
|
471
|
+
|
|
472
|
+
const designed = await generateVoice({
|
|
473
|
+
adapter: elevenlabsVoiceDesign('eleven_ttv_v3'),
|
|
474
|
+
prompt: 'A warm, gravelly narrator in his sixties with a slight Irish lilt',
|
|
475
|
+
name: 'Irish Narrator', // omit to audition previews instead
|
|
476
|
+
})
|
|
477
|
+
|
|
478
|
+
const [voice] = designed.voices
|
|
479
|
+
if (!voice) throw new Error('The provider returned no voices.')
|
|
480
|
+
|
|
481
|
+
// voice.voiceId -> pass to generateSpeech()
|
|
482
|
+
// voice.audio -> base64 preview, when the provider returns one
|
|
483
|
+
// voice.saved -> true only when it is in the provider's library
|
|
484
|
+
// voice.status -> 'ready' on every adapter today
|
|
485
|
+
|
|
486
|
+
const speech = await generateSpeech({
|
|
487
|
+
adapter: elevenlabsSpeech('eleven_v3'),
|
|
488
|
+
text: 'Once upon a time...',
|
|
489
|
+
voice: voice.voiceId,
|
|
490
|
+
})
|
|
491
|
+
```
|
|
492
|
+
|
|
493
|
+
**Status.** Every returned voice carries `status`. It is `'ready'` on every
|
|
494
|
+
adapter today, because they all finish the voice before returning. The
|
|
495
|
+
`'training'` and `'failed'` members exist for providers that build a voice
|
|
496
|
+
asynchronously; no adapter returns them yet, so do not write polling code
|
|
497
|
+
against them.
|
|
498
|
+
|
|
499
|
+
**Finding voices again.** `generateVoice()` hands back an id you are expected
|
|
500
|
+
to store. `listVoices({ adapter: <a TTS adapter>, origins })` reads the
|
|
501
|
+
account catalog back when you did not.
|
|
502
|
+
|
|
503
|
+
```typescript
|
|
504
|
+
import { listVoices } from '@tanstack/ai'
|
|
505
|
+
import { elevenlabsSpeech } from '@tanstack/ai-elevenlabs'
|
|
506
|
+
|
|
507
|
+
const { voices } = await listVoices({
|
|
508
|
+
adapter: elevenlabsSpeech('eleven_v3'),
|
|
509
|
+
origins: ['generated', 'cloned'],
|
|
510
|
+
})
|
|
511
|
+
```
|
|
512
|
+
|
|
513
|
+
`listVoices` hangs off the **TTS** adapter, not the voice adapter, because
|
|
514
|
+
`voice` is a `generateSpeech()` option — that is where the id gets consumed.
|
|
515
|
+
It is OPTIONAL, and only providers with a per-account catalog implement it.
|
|
516
|
+
Where the catalog is fixed the package publishes it instead — `GeminiTTSVoices`
|
|
517
|
+
from `@tanstack/ai-gemini`, or the `OpenAITTSVoice` union from
|
|
518
|
+
`@tanstack/ai-openai`. Prefer those: a type union beats a network call.
|
|
519
|
+
Calling `listVoices()` on such an adapter throws and points at them.
|
|
520
|
+
|
|
521
|
+
There is no React hook for this activity. Call it from a server route or
|
|
522
|
+
server function and return the result as JSON.
|
|
523
|
+
|
|
524
|
+
`elevenlabsVoiceDesign` is the only `generateVoice()` adapter in this repo.
|
|
525
|
+
xAI, BytePlus, and fal.ai each publish a voice-cloning API and are the
|
|
526
|
+
candidates for the next one, but none is implemented — do not write code
|
|
527
|
+
against them from this file.
|
|
528
|
+
|
|
529
|
+
OpenAI, Gemini, and Cloudflare have fixed voice catalogs and will not get one.
|
|
530
|
+
|
|
531
|
+
### 5. Audio Transcription
|
|
400
532
|
|
|
401
533
|
Adapters: `openaiTranscription` (whisper-1, gpt-4o-transcribe,
|
|
402
534
|
gpt-4o-mini-transcribe, gpt-4o-transcribe-diarize) and `byteplusTranscription`
|
|
@@ -476,7 +608,7 @@ const { generate, result, isLoading } = useTranscription({
|
|
|
476
608
|
// Trigger: generate({ audio: dataUrl, language: 'en' })
|
|
477
609
|
```
|
|
478
610
|
|
|
479
|
-
###
|
|
611
|
+
### 6. Video Generation (Experimental -- async polling)
|
|
480
612
|
|
|
481
613
|
Video generation uses a jobs/polling architecture. The server creates a job,
|
|
482
614
|
polls for status, and streams updates to the client. Adapters: `openaiVideo`
|
|
@@ -652,7 +784,7 @@ const { generate, result, jobId, videoStatus, isLoading } = useGenerateVideo({
|
|
|
652
784
|
// result (on completion): { url }
|
|
653
785
|
```
|
|
654
786
|
|
|
655
|
-
###
|
|
787
|
+
### 7. Cost tracking (fal billable units)
|
|
656
788
|
|
|
657
789
|
fal bills media generation by usage-based units, not tokens. Every fal media
|
|
658
790
|
adapter (`falImage`, `falAudio`, `falSpeech`, `falTranscription`, `falVideo`)
|
|
@@ -682,7 +814,7 @@ if (result.usage?.billed) {
|
|
|
682
814
|
For video, the units arrive with the completed result: `getVideoJobStatus()`
|
|
683
815
|
returns `usage` and emits a `video:usage` devtools event when fal reports it.
|
|
684
816
|
|
|
685
|
-
###
|
|
817
|
+
### 8. Durable persistence (job lifecycle + artifact bytes)
|
|
686
818
|
|
|
687
819
|
To make generations survive a server restart and be re-served later, add
|
|
688
820
|
`withGenerationPersistence` from `@tanstack/ai-persistence` as generation
|
|
@@ -1004,7 +1136,11 @@ generateAudio({
|
|
|
1004
1136
|
|
|
1005
1137
|
### g. MEDIUM: Gemini TTS multi-speaker with 0 or 3+ speakers
|
|
1006
1138
|
|
|
1007
|
-
`
|
|
1139
|
+
Prefer `turns` for new code: it builds `multiSpeakerVoiceConfig` and the
|
|
1140
|
+
labelled prompt for you, and the two-speaker cap is enforced by the activity
|
|
1141
|
+
from `capabilities.maxSpeakers`.
|
|
1142
|
+
|
|
1143
|
+
The hand-rolled form below still works. `multiSpeakerVoiceConfig.speakerVoiceConfigs` is validated to be length 1 or 2. Passing an empty array or three+ entries throws at the adapter boundary
|
|
1008
1144
|
(not at Gemini's API) with a clear error. Don't try to work around it with
|
|
1009
1145
|
`as any`.
|
|
1010
1146
|
|
|
@@ -63,10 +63,12 @@ import {
|
|
|
63
63
|
import { maxIterations as maxIterationsStrategy } from './agent-loop-strategies'
|
|
64
64
|
import { isCancelRequestedReason } from './cancel'
|
|
65
65
|
import {
|
|
66
|
+
appendUiResourceToModelMessages,
|
|
66
67
|
convertMessagesToModelMessages,
|
|
67
68
|
generateMessageId,
|
|
68
69
|
modelMessagesToUIMessages,
|
|
69
70
|
safeJsonStringify,
|
|
71
|
+
uiResourcePartFromCustomValue,
|
|
70
72
|
} from './messages'
|
|
71
73
|
import { MiddlewareRunner } from './middleware/compose'
|
|
72
74
|
import { getRunDetached } from './middleware/run-store'
|
|
@@ -2770,6 +2772,22 @@ class TextEngine<
|
|
|
2770
2772
|
}
|
|
2771
2773
|
}
|
|
2772
2774
|
|
|
2775
|
+
/**
|
|
2776
|
+
* Record a `ui-resource` CUSTOM chunk on the assistant ModelMessage owning
|
|
2777
|
+
* its `toolCallId` so the resource survives later MESSAGES_SNAPSHOT chunks
|
|
2778
|
+
* (e.g. the interrupt snapshot emitted when the run pauses on a client
|
|
2779
|
+
* tool). Mirrors the anchor-preserving approach used for
|
|
2780
|
+
* `toolCallMetadata` (#867). See #1397.
|
|
2781
|
+
*/
|
|
2782
|
+
private recordEmittedUiResource(value: unknown): void {
|
|
2783
|
+
const part = uiResourcePartFromCustomValue(value)
|
|
2784
|
+
if (!part) return
|
|
2785
|
+
const next = appendUiResourceToModelMessages(this.messages, part)
|
|
2786
|
+
if (next === this.messages) return
|
|
2787
|
+
this.messages = next
|
|
2788
|
+
this.middlewareCtx.messages = this.messages
|
|
2789
|
+
}
|
|
2790
|
+
|
|
2773
2791
|
private buildMessagesSnapshotChunk(): StreamChunk {
|
|
2774
2792
|
const withIds = this.messages.map((message, index) => ({
|
|
2775
2793
|
...message,
|
|
@@ -4431,6 +4449,11 @@ class TextEngine<
|
|
|
4431
4449
|
if (this.hasPublicRunStarted) continue
|
|
4432
4450
|
this.hasPublicRunStarted = true
|
|
4433
4451
|
}
|
|
4452
|
+
// Persist MCP Apps ui-resource emissions onto the tool-call anchor
|
|
4453
|
+
// message so interrupt MESSAGES_SNAPSHOT chunks keep them (#1397).
|
|
4454
|
+
if (spec.type === EventType.CUSTOM && spec.name === 'ui-resource') {
|
|
4455
|
+
this.recordEmittedUiResource((spec as CustomEvent).value)
|
|
4456
|
+
}
|
|
4434
4457
|
yield spec
|
|
4435
4458
|
this.middlewareCtx.chunkIndex++
|
|
4436
4459
|
}
|
|
@@ -387,6 +387,66 @@ function appendUiResources(
|
|
|
387
387
|
return { ...ui, parts: [...ui.parts, ...extra] }
|
|
388
388
|
}
|
|
389
389
|
|
|
390
|
+
/**
|
|
391
|
+
* Build a UIResourcePart from the value of a CUSTOM `ui-resource` chunk
|
|
392
|
+
* emitted via `ctx.emitCustomEvent('ui-resource', ...)` (MCP Apps). The
|
|
393
|
+
* emission-side value carries `resource`/`serverId`/`toolName` plus the
|
|
394
|
+
* `toolCallId` stamped by the tool-call context wrapper — the `type`
|
|
395
|
+
* discriminator is added here. Returns undefined when the value does not
|
|
396
|
+
* match the ui-resource shape.
|
|
397
|
+
*/
|
|
398
|
+
export function uiResourcePartFromCustomValue(
|
|
399
|
+
value: unknown,
|
|
400
|
+
): UIResourcePart | undefined {
|
|
401
|
+
if (!isRecord(value)) return undefined
|
|
402
|
+
const part: unknown = { type: 'ui-resource', ...value }
|
|
403
|
+
return isUiResourcePart(part) ? part : undefined
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
/**
|
|
407
|
+
* Store an emitted ui-resource part on the assistant ModelMessage that owns
|
|
408
|
+
* its `toolCallId` (the tool-call anchor), so it survives later
|
|
409
|
+
* MESSAGES_SNAPSHOT chunks — e.g. the interrupt snapshot emitted when the
|
|
410
|
+
* run pauses on a client tool (#1397). Mirrors how `toolCallMetadata` is
|
|
411
|
+
* preserved on the anchor (#867).
|
|
412
|
+
*
|
|
413
|
+
* Returns the SAME array reference when no anchor owns the tool call or the
|
|
414
|
+
* resource is already stored (idempotent).
|
|
415
|
+
*/
|
|
416
|
+
export function appendUiResourceToModelMessages(
|
|
417
|
+
messages: Array<ModelMessage>,
|
|
418
|
+
part: UIResourcePart,
|
|
419
|
+
): Array<ModelMessage> {
|
|
420
|
+
for (let index = messages.length - 1; index >= 0; index--) {
|
|
421
|
+
const message = messages[index]
|
|
422
|
+
if (!message || message.role !== 'assistant') continue
|
|
423
|
+
const ownsToolCall = message.toolCalls?.some(
|
|
424
|
+
(toolCall) => toolCall.id === part.toolCallId,
|
|
425
|
+
)
|
|
426
|
+
if (!ownsToolCall) continue
|
|
427
|
+
const previous = tanstackMetadata(message)?.uiResources ?? []
|
|
428
|
+
if (
|
|
429
|
+
previous.some((stored) => uiResourceKey(stored) === uiResourceKey(part))
|
|
430
|
+
) {
|
|
431
|
+
return messages
|
|
432
|
+
}
|
|
433
|
+
const nextMessage = {
|
|
434
|
+
...message,
|
|
435
|
+
metadata: {
|
|
436
|
+
...message.metadata,
|
|
437
|
+
tanstack: {
|
|
438
|
+
...tanstackMetadata(message),
|
|
439
|
+
uiResources: [...previous, part],
|
|
440
|
+
},
|
|
441
|
+
},
|
|
442
|
+
}
|
|
443
|
+
const next = messages.slice()
|
|
444
|
+
next[index] = nextMessage
|
|
445
|
+
return next
|
|
446
|
+
}
|
|
447
|
+
return messages
|
|
448
|
+
}
|
|
449
|
+
|
|
390
450
|
function assistantMetadata(
|
|
391
451
|
uiMessage: UIMessage,
|
|
392
452
|
): UIMessage['metadata'] | undefined {
|
|
@@ -242,6 +242,37 @@ export class StreamProcessor {
|
|
|
242
242
|
this.emitMessagesChange()
|
|
243
243
|
}
|
|
244
244
|
|
|
245
|
+
/**
|
|
246
|
+
* Put older UI messages at the front of the conversation.
|
|
247
|
+
*
|
|
248
|
+
* Skip a message if its id is already in the list. Keep the existing message.
|
|
249
|
+
* Then emit the same messages-change event as `setMessages`.
|
|
250
|
+
*
|
|
251
|
+
* Use this for older history pages. The first hydrate window uses `setMessages`.
|
|
252
|
+
*
|
|
253
|
+
* @param messages Older UI messages in insertion order. The first item is the oldest.
|
|
254
|
+
*
|
|
255
|
+
* @example
|
|
256
|
+
* ```ts
|
|
257
|
+
* processor.setMessages([newest])
|
|
258
|
+
* processor.prependMessages([oldest])
|
|
259
|
+
* ```
|
|
260
|
+
*/
|
|
261
|
+
prependMessages(messages: Array<UIMessage>) {
|
|
262
|
+
const existingIds = new Set(this.messages.map((message) => message.id))
|
|
263
|
+
const olderMessages: Array<UIMessage> = []
|
|
264
|
+
for (const message of messages) {
|
|
265
|
+
const isDuplicate = existingIds.has(message.id)
|
|
266
|
+
if (isDuplicate) {
|
|
267
|
+
continue
|
|
268
|
+
}
|
|
269
|
+
existingIds.add(message.id)
|
|
270
|
+
olderMessages.push(message)
|
|
271
|
+
}
|
|
272
|
+
this.messages = [...olderMessages, ...this.messages]
|
|
273
|
+
this.emitMessagesChange()
|
|
274
|
+
}
|
|
275
|
+
|
|
245
276
|
/**
|
|
246
277
|
* Add a user message to the conversation.
|
|
247
278
|
* Supports both simple string content and multimodal content arrays.
|
|
@@ -233,6 +233,15 @@ export class ToolCallManager<
|
|
|
233
233
|
* Add a TOOL_CALL_START event to begin tracking a tool call (AG-UI)
|
|
234
234
|
*/
|
|
235
235
|
addToolCallStartEvent(event: ToolCallStartEvent): void {
|
|
236
|
+
// AG-UI's TOOL_CALL_START carries no index, and a non-first-party or
|
|
237
|
+
// malformed producer can send a second START for a toolCallId that is
|
|
238
|
+
// already tracked. Without this guard, a repeat with the same index
|
|
239
|
+
// overwrites the slot (wiping any TOOL_CALL_ARGS already accumulated),
|
|
240
|
+
// and a repeat with a missing/different index inserts a duplicate row
|
|
241
|
+
// that getToolCalls() returns twice, running the tool twice.
|
|
242
|
+
for (const toolCall of this.toolCallsMap.values()) {
|
|
243
|
+
if (toolCall.id === event.toolCallId) return
|
|
244
|
+
}
|
|
236
245
|
const index = (event as AdapterYieldChunk).index ?? this.toolCallsMap.size
|
|
237
246
|
const name = event.toolCallName ?? event.toolName
|
|
238
247
|
this.toolCallsMap.set(index, {
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
import type { InternalLogger } from '../../logger/internal-logger'
|
|
2
|
+
import type { TokenUsage } from '../../types'
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Configuration for evaluate adapter instances.
|
|
6
|
+
*/
|
|
7
|
+
export interface EvaluateAdapterConfig {
|
|
8
|
+
apiKey?: string
|
|
9
|
+
baseUrl?: string
|
|
10
|
+
timeout?: number
|
|
11
|
+
headers?: Record<string, string>
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Shared JSON value for `state` and question `instructions`.
|
|
16
|
+
* A JSON array is one value, not a batch.
|
|
17
|
+
*/
|
|
18
|
+
export type EvaluateJsonValue = string | object | Array<unknown>
|
|
19
|
+
|
|
20
|
+
/** Content the model judges. A JSON array is one state, not a batch. */
|
|
21
|
+
export type EvaluateState = EvaluateJsonValue
|
|
22
|
+
|
|
23
|
+
/** Question text. Matches TypeSafe: string, object, or array. */
|
|
24
|
+
export type EvaluateInstructions = EvaluateJsonValue
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* TypeSafe choice question on the adapter wire.
|
|
28
|
+
*
|
|
29
|
+
* Generic parameters:
|
|
30
|
+
* - TOptions: option key to description (or `null` when the key is enough)
|
|
31
|
+
*/
|
|
32
|
+
export interface WireChoiceQuestion<
|
|
33
|
+
TOptions extends Record<string, string | null> = Record<
|
|
34
|
+
string,
|
|
35
|
+
string | null
|
|
36
|
+
>,
|
|
37
|
+
> {
|
|
38
|
+
type: 'choice'
|
|
39
|
+
instructions: EvaluateInstructions
|
|
40
|
+
criteria: TOptions
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* TypeSafe score question on the adapter wire.
|
|
45
|
+
*
|
|
46
|
+
* Generic parameters:
|
|
47
|
+
* - TLevels: ordered level labels, at least two
|
|
48
|
+
*/
|
|
49
|
+
export interface WireScoreQuestion<
|
|
50
|
+
TLevels extends ReadonlyArray<string> = ReadonlyArray<string>,
|
|
51
|
+
> {
|
|
52
|
+
type: 'score'
|
|
53
|
+
instructions: EvaluateInstructions
|
|
54
|
+
criteria: TLevels
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* TypeSafe yes/no question on the adapter wire.
|
|
59
|
+
* Public helpers call this `boolean`. The wire type is `noul`.
|
|
60
|
+
*/
|
|
61
|
+
export interface WireNoulQuestion {
|
|
62
|
+
type: 'noul'
|
|
63
|
+
instructions: EvaluateInstructions
|
|
64
|
+
criteria?: {
|
|
65
|
+
true?: string
|
|
66
|
+
false?: string
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Question payload adapters send to the provider. */
|
|
71
|
+
export type WireQuestion =
|
|
72
|
+
| WireChoiceQuestion
|
|
73
|
+
| WireScoreQuestion
|
|
74
|
+
| WireNoulQuestion
|
|
75
|
+
|
|
76
|
+
/** TypeSafe choice answer. Adapters do not invent a public `.value`. */
|
|
77
|
+
export interface WireChoiceAnswer {
|
|
78
|
+
type: 'choice'
|
|
79
|
+
choice: string
|
|
80
|
+
probabilities: Record<string, number>
|
|
81
|
+
confidence: number
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** TypeSafe score answer. `score` is the raw fraction. */
|
|
85
|
+
export interface WireScoreAnswer {
|
|
86
|
+
type: 'score'
|
|
87
|
+
score: number
|
|
88
|
+
legend: Record<string, string>
|
|
89
|
+
probabilities: Record<string, number>
|
|
90
|
+
confidence: number
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/** TypeSafe yes/no answer. `noul` is P(true). */
|
|
94
|
+
export interface WireNoulAnswer {
|
|
95
|
+
type: 'noul'
|
|
96
|
+
noul: number
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Provider payload for one question. The activity maps this to a unified answer. */
|
|
100
|
+
export type WireAnswer = WireChoiceAnswer | WireScoreAnswer | WireNoulAnswer
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Options passed to {@link EvaluateAdapter.evaluate}.
|
|
104
|
+
*/
|
|
105
|
+
export interface EvaluateOptions<
|
|
106
|
+
TProviderOptions extends object = Record<string, unknown>,
|
|
107
|
+
> {
|
|
108
|
+
model: string
|
|
109
|
+
/** Shared state every question judges. A JSON array is one state, not a batch. */
|
|
110
|
+
state: EvaluateState
|
|
111
|
+
/** TypeSafe wire questions, keyed by the caller's question ids. */
|
|
112
|
+
questions: Record<string, WireQuestion>
|
|
113
|
+
/** Provider-specific options forwarded by `decide()`. */
|
|
114
|
+
modelOptions?: TProviderOptions
|
|
115
|
+
/** Forwarded to the provider request for cancellation. */
|
|
116
|
+
abortSignal?: AbortSignal
|
|
117
|
+
/**
|
|
118
|
+
* Internal logger threaded from `decide()`. Adapters must call
|
|
119
|
+
* `logger.request()` before the provider call and `logger.errors()` in catch
|
|
120
|
+
* blocks.
|
|
121
|
+
*/
|
|
122
|
+
logger: InternalLogger
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Provider-level evaluate result. Adapters return the wire payload plus usage.
|
|
127
|
+
* The activity maps answers to the unified public shape.
|
|
128
|
+
*/
|
|
129
|
+
export interface EvaluateAdapterResult {
|
|
130
|
+
/** Resolved model id from the provider. */
|
|
131
|
+
model: string
|
|
132
|
+
answers: Record<string, WireAnswer>
|
|
133
|
+
usage: TokenUsage
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* Evaluate adapter interface with pre-resolved generics.
|
|
138
|
+
*
|
|
139
|
+
* An adapter is created by a provider function: `provider('model')` → `adapter`.
|
|
140
|
+
* All type resolution happens at the provider call site, not in this interface.
|
|
141
|
+
*
|
|
142
|
+
* Generic parameters:
|
|
143
|
+
* - TModel: The specific model name (e.g. `'jev-latest'`)
|
|
144
|
+
* - TProviderOptions: Provider-specific options (already resolved)
|
|
145
|
+
*/
|
|
146
|
+
export interface EvaluateAdapter<
|
|
147
|
+
TModel extends string = string,
|
|
148
|
+
TProviderOptions extends object = Record<string, unknown>,
|
|
149
|
+
> {
|
|
150
|
+
/** Discriminator for adapter kind */
|
|
151
|
+
readonly kind: 'evaluate'
|
|
152
|
+
/** Adapter name identifier */
|
|
153
|
+
readonly name: string
|
|
154
|
+
/** The model this adapter is configured for */
|
|
155
|
+
readonly model: TModel
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* @internal Type-only properties for inference. Not assigned at runtime.
|
|
159
|
+
*/
|
|
160
|
+
'~types': {
|
|
161
|
+
providerOptions: TProviderOptions
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Evaluate typed questions against `state`. Return the provider payload.
|
|
166
|
+
* Do not invent unified `.value` fields. The activity maps wire answers.
|
|
167
|
+
*/
|
|
168
|
+
evaluate: (
|
|
169
|
+
options: EvaluateOptions<TProviderOptions>,
|
|
170
|
+
) => Promise<EvaluateAdapterResult>
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* An EvaluateAdapter with any/unknown type parameters.
|
|
175
|
+
* Useful as a constraint in generic functions and interfaces.
|
|
176
|
+
*/
|
|
177
|
+
export type AnyEvaluateAdapter = EvaluateAdapter<any, any>
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* Abstract base class for evaluate adapters.
|
|
181
|
+
* Extend this class to implement an evaluate adapter for a specific provider.
|
|
182
|
+
*
|
|
183
|
+
* Generic parameters match EvaluateAdapter. The provider function resolves them.
|
|
184
|
+
*/
|
|
185
|
+
export abstract class BaseEvaluateAdapter<
|
|
186
|
+
TModel extends string = string,
|
|
187
|
+
TProviderOptions extends object = Record<string, unknown>,
|
|
188
|
+
> implements EvaluateAdapter<TModel, TProviderOptions> {
|
|
189
|
+
readonly kind = 'evaluate' as const
|
|
190
|
+
abstract readonly name: string
|
|
191
|
+
readonly model: TModel
|
|
192
|
+
|
|
193
|
+
// Type-only property - never assigned at runtime
|
|
194
|
+
declare '~types': {
|
|
195
|
+
providerOptions: TProviderOptions
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
protected config: EvaluateAdapterConfig
|
|
199
|
+
|
|
200
|
+
constructor(config: EvaluateAdapterConfig = {}, model: TModel) {
|
|
201
|
+
this.config = config
|
|
202
|
+
this.model = model
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
abstract evaluate(
|
|
206
|
+
options: EvaluateOptions<TProviderOptions>,
|
|
207
|
+
): Promise<EvaluateAdapterResult>
|
|
208
|
+
|
|
209
|
+
protected generateId(): string {
|
|
210
|
+
return `${this.name}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`
|
|
211
|
+
}
|
|
212
|
+
}
|