@tanstack/ai 0.52.3 → 0.54.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -13
- package/dist/esm/activities/chat/index.js +5 -3
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/generateLiveVideo/adapter.d.ts +69 -0
- package/dist/esm/activities/generateLiveVideo/adapter.js +23 -0
- package/dist/esm/activities/generateLiveVideo/adapter.js.map +1 -0
- package/dist/esm/activities/generateLiveVideo/index.d.ts +99 -0
- package/dist/esm/activities/generateLiveVideo/index.js +162 -0
- package/dist/esm/activities/generateLiveVideo/index.js.map +1 -0
- package/dist/esm/activities/generateVideo/index.js +3 -1
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/generateWorld/adapter.d.ts +69 -0
- package/dist/esm/activities/generateWorld/adapter.js +23 -0
- package/dist/esm/activities/generateWorld/adapter.js.map +1 -0
- package/dist/esm/activities/generateWorld/index.d.ts +99 -0
- package/dist/esm/activities/generateWorld/index.js +162 -0
- package/dist/esm/activities/generateWorld/index.js.map +1 -0
- package/dist/esm/activities/index.d.ts +8 -2
- package/dist/esm/activities/index.js +11 -7
- package/dist/esm/activities/middleware/types.d.ts +1 -1
- package/dist/esm/activities/summarize/chat-stream-summarize.js +2 -1
- package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
- package/dist/esm/byok/define-provider.d.ts +6 -0
- package/dist/esm/byok/define-provider.js +2 -1
- package/dist/esm/byok/define-provider.js.map +1 -1
- package/dist/esm/byok/get-key.d.ts +7 -0
- package/dist/esm/byok/get-key.js +8 -1
- package/dist/esm/byok/get-key.js.map +1 -1
- package/dist/esm/byok/server.d.ts +1 -1
- package/dist/esm/byok/server.js +2 -2
- package/dist/esm/client.d.ts +4 -2
- package/dist/esm/client.js +3 -1
- package/dist/esm/client.js.map +1 -1
- package/dist/esm/index.d.ts +4 -2
- package/dist/esm/index.js +3 -1
- package/dist/esm/middlewares/otel.js +3 -1
- package/dist/esm/middlewares/otel.js.map +1 -1
- package/dist/esm/types.d.ts +112 -0
- package/package.json +2 -2
- package/skills/ai-core/adapter-configuration/SKILL.md +103 -54
- package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +39 -21
- package/skills/ai-core/adapter-configuration/references/byteplus-adapter.md +5 -0
- package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +14 -6
- package/skills/ai-core/adapter-configuration/references/grok-adapter.md +33 -25
- package/skills/ai-core/adapter-configuration/references/groq-adapter.md +7 -2
- package/skills/ai-core/adapter-configuration/references/ollama-adapter.md +25 -12
- package/skills/ai-core/adapter-configuration/references/openai-adapter.md +19 -9
- package/skills/ai-core/adapter-configuration/references/openrouter-adapter.md +34 -21
- package/skills/ai-core/ag-ui-protocol/SKILL.md +16 -10
- package/skills/ai-core/chat-experience/SKILL.md +228 -108
- package/skills/ai-core/client-persistence/SKILL.md +21 -9
- package/skills/ai-core/custom-backend-integration/SKILL.md +86 -52
- package/skills/ai-core/debug-logging/SKILL.md +100 -18
- package/skills/ai-core/locks/SKILL.md +35 -7
- package/skills/ai-core/media-generation/SKILL.md +114 -49
- package/skills/ai-core/middleware/SKILL.md +174 -69
- package/skills/ai-core/structured-outputs/SKILL.md +99 -49
- package/skills/ai-core/tool-calling/SKILL.md +245 -158
- package/src/activities/chat/index.ts +6 -7
- package/src/activities/generateLiveVideo/adapter.ts +99 -0
- package/src/activities/generateLiveVideo/index.ts +339 -0
- package/src/activities/generateVideo/index.ts +3 -4
- package/src/activities/generateWorld/adapter.ts +96 -0
- package/src/activities/generateWorld/index.ts +339 -0
- package/src/activities/index.ts +44 -0
- package/src/activities/middleware/types.ts +2 -0
- package/src/activities/summarize/chat-stream-summarize.ts +2 -0
- package/src/byok/define-provider.ts +7 -0
- package/src/byok/get-key.ts +18 -0
- package/src/byok/server.ts +1 -1
- package/src/client.ts +8 -0
- package/src/index.ts +8 -0
- package/src/middlewares/otel.ts +2 -0
- package/src/types.ts +128 -0
|
@@ -110,9 +110,10 @@ parses it as SSE automatically:
|
|
|
110
110
|
import { createServerFn } from '@tanstack/react-start'
|
|
111
111
|
import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
|
|
112
112
|
import { openaiImage } from '@tanstack/ai-openai'
|
|
113
|
+
import type { OpenAIImageModel } from '@tanstack/ai-openai'
|
|
113
114
|
|
|
114
115
|
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
115
|
-
.inputValidator((data: { prompt: string; model?:
|
|
116
|
+
.inputValidator((data: { prompt: string; model?: OpenAIImageModel }) => data)
|
|
116
117
|
.handler(({ data }) => {
|
|
117
118
|
return toServerSentEventsResponse(
|
|
118
119
|
generateImage({
|
|
@@ -183,7 +184,7 @@ const openaiResult = await generateImage({
|
|
|
183
184
|
modelOptions: {
|
|
184
185
|
quality: 'high',
|
|
185
186
|
background: 'transparent',
|
|
186
|
-
|
|
187
|
+
output_format: 'png',
|
|
187
188
|
},
|
|
188
189
|
})
|
|
189
190
|
|
|
@@ -250,10 +251,10 @@ await generateImage({
|
|
|
250
251
|
adapter: openaiImage('gpt-image-2'),
|
|
251
252
|
prompt: [
|
|
252
253
|
{ type: 'text', content: 'Replace the masked region with a tree' },
|
|
253
|
-
{ type: 'image', source: { type: 'url', value:
|
|
254
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/photo.png' } },
|
|
254
255
|
{
|
|
255
256
|
type: 'image',
|
|
256
|
-
source: { type: 'url', value:
|
|
257
|
+
source: { type: 'url', value: 'https://…/mask.png' },
|
|
257
258
|
metadata: { role: 'mask' },
|
|
258
259
|
},
|
|
259
260
|
],
|
|
@@ -267,11 +268,11 @@ import { falVideo } from '@tanstack/ai-fal'
|
|
|
267
268
|
await generateVideo({
|
|
268
269
|
adapter: falVideo('fal-ai/kling-video/v3/pro/image-to-video'),
|
|
269
270
|
prompt: [
|
|
270
|
-
{ type: 'image', source: { type: 'url', value:
|
|
271
|
+
{ type: 'image', source: { type: 'url', value: 'https://…/first.png' } },
|
|
271
272
|
{ type: 'text', content: 'Slow cinematic push-in' },
|
|
272
273
|
{
|
|
273
274
|
type: 'image',
|
|
274
|
-
source: { type: 'url', value:
|
|
275
|
+
source: { type: 'url', value: 'https://…/last.png' },
|
|
275
276
|
metadata: { role: 'end_frame' },
|
|
276
277
|
},
|
|
277
278
|
],
|
|
@@ -404,35 +405,58 @@ gpt-4o-mini-transcribe, gpt-4o-transcribe-diarize) and `byteplusTranscription`
|
|
|
404
405
|
|
|
405
406
|
> **Capturing audio in the browser:** Use `useAudioRecorder` from `@tanstack/ai-react` to record directly in the browser, then pass the recording as the `audio` input to `generate()`, or use `recording.part` as a prompt part in chat/generation calls. No transcoding or extra dependencies required — the recorder returns the native browser format (`audio/webm` or `audio/mp4`). For transcription, wrap it as a `data:` URL so the provider gets the real content type; passing raw `recording.base64` makes the adapter assume `audio/mpeg` and mislabel the webm/mp4 bytes.
|
|
406
407
|
>
|
|
407
|
-
> ```
|
|
408
|
-
>
|
|
409
|
-
>
|
|
410
|
-
>
|
|
411
|
-
>
|
|
412
|
-
>
|
|
413
|
-
>
|
|
414
|
-
>
|
|
415
|
-
>
|
|
408
|
+
> ```tsx
|
|
409
|
+
> import {
|
|
410
|
+
> useAudioRecorder,
|
|
411
|
+
> useTranscription,
|
|
412
|
+
> fetchServerSentEvents,
|
|
413
|
+
> } from '@tanstack/ai-react'
|
|
414
|
+
>
|
|
415
|
+
> function VoiceNote() {
|
|
416
|
+
> const { isRecording, start, stop } = useAudioRecorder()
|
|
417
|
+
> const { generate } = useTranscription({
|
|
418
|
+
> connection: fetchServerSentEvents('/api/transcribe'),
|
|
419
|
+
> })
|
|
420
|
+
>
|
|
421
|
+
> async function finish() {
|
|
422
|
+
> const recording = await stop()
|
|
423
|
+
> const mimeType = recording.mimeType.split(';')[0] // strip ;codecs=...
|
|
424
|
+
> await generate({ audio: `data:${mimeType};base64,${recording.base64}` })
|
|
425
|
+
> }
|
|
426
|
+
>
|
|
427
|
+
> return (
|
|
428
|
+
> <button onClick={isRecording ? finish : start}>
|
|
429
|
+
> {isRecording ? 'Stop & transcribe' : 'Record'}
|
|
430
|
+
> </button>
|
|
431
|
+
> )
|
|
432
|
+
> }
|
|
416
433
|
> ```
|
|
417
434
|
|
|
418
435
|
```typescript
|
|
419
|
-
|
|
436
|
+
// routes/api/transcribe.ts
|
|
437
|
+
import { generateTranscription, toServerSentEventsResponse } from '@tanstack/ai'
|
|
420
438
|
import { openaiTranscription } from '@tanstack/ai-openai'
|
|
421
439
|
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
440
|
+
export async function POST(request: Request) {
|
|
441
|
+
// The client hook below posts { data: { audio: dataUrl, language } }
|
|
442
|
+
const { audio, language } = (await request.json()).data
|
|
443
|
+
|
|
444
|
+
const stream = generateTranscription({
|
|
445
|
+
adapter: openaiTranscription('whisper-1'),
|
|
446
|
+
audio, // File, Blob, base64 string, or data URL
|
|
447
|
+
language,
|
|
448
|
+
responseFormat: 'verbose_json',
|
|
449
|
+
modelOptions: {
|
|
450
|
+
timestamp_granularities: ['word', 'segment'],
|
|
451
|
+
},
|
|
452
|
+
stream: true,
|
|
453
|
+
})
|
|
431
454
|
|
|
432
|
-
// result.text
|
|
433
|
-
// result.
|
|
434
|
-
//
|
|
435
|
-
|
|
455
|
+
// On the client, result.text is the transcript, result.language the
|
|
456
|
+
// detected language, result.duration the seconds, result.segments the
|
|
457
|
+
// timestamped segments (word-level timestamps are in result.words).
|
|
458
|
+
return toServerSentEventsResponse(stream)
|
|
459
|
+
}
|
|
436
460
|
```
|
|
437
461
|
|
|
438
462
|
For speaker diarization, use `openaiTranscription('gpt-4o-transcribe-diarize')`.
|
|
@@ -486,14 +510,17 @@ while (status.status !== 'completed' && status.status !== 'failed') {
|
|
|
486
510
|
}
|
|
487
511
|
|
|
488
512
|
// Streaming: server handles polling, client gets real-time updates
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
513
|
+
export async function POST(request: Request) {
|
|
514
|
+
const { prompt } = await request.json()
|
|
515
|
+
const stream = generateVideo({
|
|
516
|
+
adapter: openaiVideo('sora-2'),
|
|
517
|
+
prompt,
|
|
518
|
+
stream: true,
|
|
519
|
+
pollingInterval: 3000,
|
|
520
|
+
maxDuration: 600_000,
|
|
521
|
+
})
|
|
522
|
+
return toServerSentEventsResponse(stream)
|
|
523
|
+
}
|
|
497
524
|
```
|
|
498
525
|
|
|
499
526
|
Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
|
|
@@ -505,6 +532,7 @@ Image prompt parts route by `metadata.role`: first un-roled /
|
|
|
505
532
|
`'reference'` / `'character'` → `referenceImages`:
|
|
506
533
|
|
|
507
534
|
```typescript
|
|
535
|
+
import { generateVideo } from '@tanstack/ai'
|
|
508
536
|
import { geminiVideo } from '@tanstack/ai-gemini'
|
|
509
537
|
|
|
510
538
|
const adapter = geminiVideo('veo-3.1-generate-preview')
|
|
@@ -538,6 +566,7 @@ media). For conversational editing, pass a prior generation's `jobId` as
|
|
|
538
566
|
on 2026-09-30.
|
|
539
567
|
|
|
540
568
|
```typescript
|
|
569
|
+
import { generateVideo } from '@tanstack/ai'
|
|
541
570
|
import { geminiVideo } from '@tanstack/ai-gemini'
|
|
542
571
|
|
|
543
572
|
const omni = geminiVideo('gemini-omni-1.1-flash')
|
|
@@ -590,6 +619,7 @@ from OpenRouter's published metadata, with the same `availableDurations()` /
|
|
|
590
619
|
`snapDuration()` helpers:
|
|
591
620
|
|
|
592
621
|
```typescript
|
|
622
|
+
import { generateVideo } from '@tanstack/ai'
|
|
593
623
|
import { openRouterVideo } from '@tanstack/ai-openrouter'
|
|
594
624
|
|
|
595
625
|
const adapter = openRouterVideo('bytedance/seedance-2.0')
|
|
@@ -643,6 +673,7 @@ const result = await generateImage({
|
|
|
643
673
|
|
|
644
674
|
// usage.billed.quantity is the priced quantity. Multiply by the endpoint unit
|
|
645
675
|
// price (GET https://api.fal.ai/v1/models/pricing?endpoint_id=…) for exact cost.
|
|
676
|
+
const unitPrice = 0.025 // USD per unit, from the pricing endpoint
|
|
646
677
|
if (result.usage?.billed) {
|
|
647
678
|
const cost = result.usage.billed.quantity * unitPrice
|
|
648
679
|
}
|
|
@@ -769,6 +800,8 @@ Provide either `connection` (streaming SSE transport) or `fetcher`
|
|
|
769
800
|
to transform what is stored:
|
|
770
801
|
|
|
771
802
|
```tsx
|
|
803
|
+
import { useGenerateSpeech, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
804
|
+
|
|
772
805
|
const { result } = useGenerateSpeech({
|
|
773
806
|
connection: fetchServerSentEvents('/api/generate/speech'),
|
|
774
807
|
onResult: (raw) => ({
|
|
@@ -790,7 +823,7 @@ Agents trained on older code may still generate this pattern.
|
|
|
790
823
|
|
|
791
824
|
**Wrong:**
|
|
792
825
|
|
|
793
|
-
```typescript
|
|
826
|
+
```typescript ignore
|
|
794
827
|
import { embedding } from '@tanstack/ai'
|
|
795
828
|
import { openaiEmbed } from '@tanstack/ai-openai'
|
|
796
829
|
|
|
@@ -825,27 +858,34 @@ stream from a server function will not work.
|
|
|
825
858
|
|
|
826
859
|
**Wrong:**
|
|
827
860
|
|
|
828
|
-
```typescript
|
|
829
|
-
|
|
830
|
-
|
|
861
|
+
```typescript ignore
|
|
862
|
+
import { createServerFn } from '@tanstack/react-start'
|
|
863
|
+
import { generateImage } from '@tanstack/ai'
|
|
864
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
865
|
+
|
|
866
|
+
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
867
|
+
.inputValidator((data: { prompt: string }) => data)
|
|
868
|
+
.handler(({ data }) => {
|
|
831
869
|
// BUG: returning raw stream -- client cannot parse this
|
|
870
|
+
// (also a type error: an AsyncIterable is not a valid server-function return)
|
|
832
871
|
return generateImage({
|
|
833
872
|
adapter: openaiImage('gpt-image-1'),
|
|
834
873
|
prompt: data.prompt,
|
|
835
874
|
stream: true,
|
|
836
875
|
})
|
|
837
|
-
}
|
|
838
|
-
)
|
|
876
|
+
})
|
|
839
877
|
```
|
|
840
878
|
|
|
841
879
|
**Correct:**
|
|
842
880
|
|
|
843
881
|
```typescript
|
|
882
|
+
import { createServerFn } from '@tanstack/react-start'
|
|
844
883
|
import { generateImage, toServerSentEventsResponse } from '@tanstack/ai'
|
|
845
884
|
import { openaiImage } from '@tanstack/ai-openai'
|
|
846
885
|
|
|
847
|
-
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
848
|
-
({
|
|
886
|
+
export const generateImageStreamFn = createServerFn({ method: 'POST' })
|
|
887
|
+
.inputValidator((data: { prompt: string }) => data)
|
|
888
|
+
.handler(({ data }) => {
|
|
849
889
|
return toServerSentEventsResponse(
|
|
850
890
|
generateImage({
|
|
851
891
|
adapter: openaiImage('gpt-image-1'),
|
|
@@ -853,8 +893,7 @@ export const generateImageStreamFn = createServerFn({ method: 'POST' }).handler(
|
|
|
853
893
|
stream: true,
|
|
854
894
|
}),
|
|
855
895
|
)
|
|
856
|
-
}
|
|
857
|
-
)
|
|
896
|
+
})
|
|
858
897
|
```
|
|
859
898
|
|
|
860
899
|
> Source: maintainer interview.
|
|
@@ -866,6 +905,9 @@ later, the image will silently break. Always download or display the image
|
|
|
866
905
|
immediately, or convert to base64 for persistence.
|
|
867
906
|
|
|
868
907
|
```typescript
|
|
908
|
+
import { generateImage } from '@tanstack/ai'
|
|
909
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
910
|
+
|
|
869
911
|
const result = await generateImage({
|
|
870
912
|
adapter: openaiImage('dall-e-3'),
|
|
871
913
|
prompt: 'A mountain landscape',
|
|
@@ -904,7 +946,7 @@ Gemini's `GenerateContentConfig` (used by Lyria 3 Pro / Lyria 3 Clip) does
|
|
|
904
946
|
returns 30-second `audio/mp3`; Lyria 3 Pro returns `audio/mp3`. These fields
|
|
905
947
|
are not in `GeminiAudioProviderOptions` — don't reach for them via `as any`.
|
|
906
948
|
|
|
907
|
-
```typescript
|
|
949
|
+
```typescript ignore
|
|
908
950
|
// WRONG — both fields are silently ignored or rejected by the SDK
|
|
909
951
|
generateAudio({
|
|
910
952
|
adapter: geminiAudio('lyria-3-pro-preview'),
|
|
@@ -914,6 +956,11 @@ generateAudio({
|
|
|
914
956
|
negativePrompt: 'vocals', // unsupported
|
|
915
957
|
} as any,
|
|
916
958
|
})
|
|
959
|
+
```
|
|
960
|
+
|
|
961
|
+
```typescript
|
|
962
|
+
import { generateAudio } from '@tanstack/ai'
|
|
963
|
+
import { geminiAudio } from '@tanstack/ai-gemini'
|
|
917
964
|
|
|
918
965
|
// CORRECT — shape the prompt itself for what you want
|
|
919
966
|
generateAudio({
|
|
@@ -934,6 +981,10 @@ model's native field like `music_length_ms` or `seconds_total`), but not
|
|
|
934
981
|
for Lyria.
|
|
935
982
|
|
|
936
983
|
```typescript
|
|
984
|
+
import { generateAudio } from '@tanstack/ai'
|
|
985
|
+
import { geminiAudio } from '@tanstack/ai-gemini'
|
|
986
|
+
import { falAudio } from '@tanstack/ai-fal'
|
|
987
|
+
|
|
937
988
|
// For Lyria: put length guidance in the prompt
|
|
938
989
|
generateAudio({
|
|
939
990
|
adapter: geminiAudio('lyria-3-pro-preview'),
|
|
@@ -958,6 +1009,9 @@ generateAudio({
|
|
|
958
1009
|
`as any`.
|
|
959
1010
|
|
|
960
1011
|
```typescript
|
|
1012
|
+
import { generateSpeech } from '@tanstack/ai'
|
|
1013
|
+
import { geminiSpeech } from '@tanstack/ai-gemini'
|
|
1014
|
+
|
|
961
1015
|
generateSpeech({
|
|
962
1016
|
adapter: geminiSpeech('gemini-2.5-pro-preview-tts'),
|
|
963
1017
|
text: '[Alice] Hi. [Bob] Hello!',
|
|
@@ -988,7 +1042,7 @@ narrowed per model, so passing an image part to a text-only model
|
|
|
988
1042
|
also throw a clear runtime error as a backstop, so users learn at call
|
|
989
1043
|
time rather than getting silently wrong output.
|
|
990
1044
|
|
|
991
|
-
```typescript
|
|
1045
|
+
```typescript ignore
|
|
992
1046
|
// WRONG — dall-e-3 has no edit/inputs API; image parts are a type error
|
|
993
1047
|
generateImage({
|
|
994
1048
|
adapter: openaiImage('dall-e-3'),
|
|
@@ -1006,6 +1060,14 @@ generateImage({
|
|
|
1006
1060
|
{ type: 'image', source: { type: 'url', value: url } }, // ❌ type error
|
|
1007
1061
|
],
|
|
1008
1062
|
})
|
|
1063
|
+
```
|
|
1064
|
+
|
|
1065
|
+
```typescript
|
|
1066
|
+
import { generateImage } from '@tanstack/ai'
|
|
1067
|
+
import { openaiImage } from '@tanstack/ai-openai'
|
|
1068
|
+
import { geminiImage } from '@tanstack/ai-gemini'
|
|
1069
|
+
|
|
1070
|
+
const url = 'https://…/photo.png'
|
|
1009
1071
|
|
|
1010
1072
|
// CORRECT — use a model that supports image-conditioned generation
|
|
1011
1073
|
generateImage({
|
|
@@ -1035,6 +1097,9 @@ same `debug?: DebugOption` option that `chat()` does. Reach for `debug`
|
|
|
1035
1097
|
instead of wiring up logging middleware.
|
|
1036
1098
|
|
|
1037
1099
|
```typescript
|
|
1100
|
+
import { generateSpeech } from '@tanstack/ai'
|
|
1101
|
+
import { openaiSpeech } from '@tanstack/ai-openai'
|
|
1102
|
+
|
|
1038
1103
|
// When a speech generation sounds wrong or a transcription returns garbage
|
|
1039
1104
|
generateSpeech({
|
|
1040
1105
|
adapter: openaiSpeech('tts-1'),
|