@tanstack/ai 0.55.0 → 0.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/README.md +42 -16
  2. package/dist/esm/activities/chat/index.js +8 -6
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/middleware/types.d.ts +4 -3
  5. package/dist/esm/activities/chat/middleware/types.js.map +1 -1
  6. package/dist/esm/activities/chat/tools/tool-calls.d.ts +2 -2
  7. package/dist/esm/activities/chat/tools/tool-calls.js +3 -2
  8. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  9. package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
  10. package/dist/esm/activities/evaluate/adapter.js +23 -0
  11. package/dist/esm/activities/evaluate/adapter.js.map +1 -0
  12. package/dist/esm/activities/evaluate/index.d.ts +255 -0
  13. package/dist/esm/activities/evaluate/index.js +317 -0
  14. package/dist/esm/activities/evaluate/index.js.map +1 -0
  15. package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
  16. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
  17. package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
  18. package/dist/esm/activities/generateSpeech/index.js +53 -3
  19. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  20. package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
  21. package/dist/esm/activities/generateVoice/adapter.js +23 -0
  22. package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
  23. package/dist/esm/activities/generateVoice/index.d.ts +133 -0
  24. package/dist/esm/activities/generateVoice/index.js +184 -0
  25. package/dist/esm/activities/generateVoice/index.js.map +1 -0
  26. package/dist/esm/activities/index.d.ts +10 -4
  27. package/dist/esm/activities/index.js +14 -10
  28. package/dist/esm/activities/middleware/types.d.ts +1 -1
  29. package/dist/esm/client.d.ts +3 -2
  30. package/dist/esm/client.js +21 -3
  31. package/dist/esm/client.js.map +1 -1
  32. package/dist/esm/index.d.ts +4 -2
  33. package/dist/esm/index.js +6 -3
  34. package/dist/esm/middlewares/otel.js +2 -0
  35. package/dist/esm/middlewares/otel.js.map +1 -1
  36. package/dist/esm/realtime/index.d.ts +1 -1
  37. package/dist/esm/realtime/index.js +1 -1
  38. package/dist/esm/realtime/index.js.map +1 -1
  39. package/dist/esm/stream-to-response.js +12 -6
  40. package/dist/esm/stream-to-response.js.map +1 -1
  41. package/dist/esm/strip-to-spec-middleware.js +2 -1
  42. package/dist/esm/strip-to-spec-middleware.js.map +1 -1
  43. package/dist/esm/types.d.ts +243 -3
  44. package/dist/esm/utilities/durability-batch.d.ts +8 -0
  45. package/dist/esm/utilities/durability-batch.js +45 -0
  46. package/dist/esm/utilities/durability-batch.js.map +1 -0
  47. package/package.json +3 -3
  48. package/skills/ai-core/media-generation/SKILL.md +132 -6
  49. package/src/activities/chat/index.ts +11 -5
  50. package/src/activities/chat/middleware/types.ts +8 -2
  51. package/src/activities/chat/tools/tool-calls.ts +24 -5
  52. package/src/activities/evaluate/adapter.ts +212 -0
  53. package/src/activities/evaluate/index.ts +614 -0
  54. package/src/activities/generateSpeech/adapter.ts +47 -1
  55. package/src/activities/generateSpeech/index.ts +149 -8
  56. package/src/activities/generateVoice/adapter.ts +89 -0
  57. package/src/activities/generateVoice/index.ts +371 -0
  58. package/src/activities/index.ts +69 -0
  59. package/src/activities/middleware/types.ts +2 -0
  60. package/src/client.ts +35 -8
  61. package/src/index.ts +21 -0
  62. package/src/middlewares/otel.ts +2 -0
  63. package/src/realtime/index.ts +1 -1
  64. package/src/stream-to-response.ts +16 -6
  65. package/src/strip-to-spec-middleware.ts +2 -1
  66. package/src/types.ts +269 -3
  67. package/src/utilities/durability-batch.ts +48 -0
package/src/types.ts CHANGED
@@ -629,6 +629,22 @@ type RuntimeContextField<TContext> =
629
629
  context: TContext
630
630
  }
631
631
 
632
+ /**
633
+ * Options for a single `emitCustomEvent` call, on both the tool-execution and
634
+ * middleware contexts.
635
+ */
636
+ export interface EmitCustomEventOptions {
637
+ /**
638
+ * Keep this event in the durability batch with later chunks.
639
+ * CUSTOM events flush as soon as they are emitted, so a progress
640
+ * indicator can render at emit time. Pass `{ batch: true }` for a
641
+ * high-volume stream that should share appends with later output.
642
+ * `process.stdout`, `process.stderr`, `sandbox.file`, and
643
+ * `sandbox.file.diff` already batch.
644
+ */
645
+ batch?: boolean
646
+ }
647
+
632
648
  /**
633
649
  * Context passed to tool execute functions, providing capabilities like
634
650
  * emitting custom events during execution.
@@ -649,6 +665,8 @@ export type ToolExecutionContext<TContext = unknown> =
649
665
  *
650
666
  * @param eventName - Name of the custom event
651
667
  * @param value - Event payload value
668
+ * @param options - Pass `{ batch: true }` to keep this event in the
669
+ * durability batch instead of flushing it immediately
652
670
  *
653
671
  * @example
654
672
  * ```ts
@@ -661,7 +679,11 @@ export type ToolExecutionContext<TContext = unknown> =
661
679
  * })
662
680
  * ```
663
681
  */
664
- emitCustomEvent: (eventName: string, value: Record<string, any>) => void
682
+ emitCustomEvent: (
683
+ eventName: string,
684
+ value: Record<string, any>,
685
+ options?: EmitCustomEventOptions,
686
+ ) => void
665
687
  }
666
688
 
667
689
  export type ToolExecuteFunction<
@@ -2350,6 +2372,56 @@ export interface LiveVideoGenerationResult {
2350
2372
  // Text-to-Speech (TTS) Types
2351
2373
  // ============================================================================
2352
2374
 
2375
+ /**
2376
+ * One turn of a multi-voice dialogue request.
2377
+ *
2378
+ * Providers that expose a dedicated dialogue endpoint (ElevenLabs
2379
+ * `textToDialogue`, Gemini multi-speaker) take these natively instead of a
2380
+ * single `text` + `voice` pair.
2381
+ */
2382
+ export interface TTSTurn {
2383
+ /** The text this voice speaks. */
2384
+ text: string
2385
+ /** Provider voice id (ElevenLabs) or voice name (Gemini) for this turn. */
2386
+ voice: string
2387
+ }
2388
+
2389
+ /**
2390
+ * Timings for the generated audio, returned when `timestamps: true` was
2391
+ * requested and the adapter declares `capabilities.timestamps`.
2392
+ *
2393
+ * Granularity differs per provider — ElevenLabs reports characters, BytePlus
2394
+ * reports words — so `unit` says which, and the three arrays are parallel.
2395
+ * All times are **seconds**; adapters convert.
2396
+ */
2397
+ export interface TTSAlignment {
2398
+ /** Granularity of each entry. */
2399
+ unit: 'character' | 'word'
2400
+ /** Entry text, in audio order. */
2401
+ texts: Array<string>
2402
+ /** Start of each entry in seconds. Same length as `texts`. */
2403
+ startSeconds: Array<number>
2404
+ /** End of each entry in seconds. Same length as `texts`. */
2405
+ endSeconds: Array<number>
2406
+ }
2407
+
2408
+ /**
2409
+ * A stretch of audio attributable to one turn (multi-voice) or one utterance
2410
+ * (single voice). This is what tells a consumer which turn is where.
2411
+ */
2412
+ export interface TTSSegment {
2413
+ /** Start of the segment in seconds. */
2414
+ startSeconds: number
2415
+ /** End of the segment in seconds. */
2416
+ endSeconds: number
2417
+ /** Index into the request's `turns`, when the provider reports it. */
2418
+ turnIndex?: number
2419
+ /** Voice heard in this segment, when the provider reports it. */
2420
+ voice?: string
2421
+ /** Text spoken in this segment, when the provider reports it. */
2422
+ text?: string
2423
+ }
2424
+
2353
2425
  /**
2354
2426
  * Options for text-to-speech generation.
2355
2427
  * These are the common options supported across providers.
@@ -2357,8 +2429,24 @@ export interface LiveVideoGenerationResult {
2357
2429
  export interface TTSOptions<TProviderOptions extends object = object> {
2358
2430
  /** The model to use for TTS generation */
2359
2431
  model: string
2360
- /** The text to convert to speech */
2432
+ /**
2433
+ * The text to convert to speech. When the caller passed `turns`, the
2434
+ * activity fills this with the turn texts joined by newlines so adapters
2435
+ * that only read `text` still receive the full script.
2436
+ */
2361
2437
  text: string
2438
+ /**
2439
+ * Multi-voice dialogue turns, when the caller asked for dialogue. Only
2440
+ * adapters that declare `capabilities.maxSpeakers` ever see this — the
2441
+ * activity rejects `turns` for the rest.
2442
+ */
2443
+ turns?: Array<TTSTurn>
2444
+ /**
2445
+ * Ask for `alignment` / `segments` on the result. Rejected by the activity
2446
+ * unless the adapter declares `capabilities.timestamps`, because on some
2447
+ * providers this is a different endpoint rather than free metadata.
2448
+ */
2449
+ timestamps?: boolean
2362
2450
  /** The voice to use for generation */
2363
2451
  voice?: string
2364
2452
  /** The output audio format */
@@ -2393,8 +2481,18 @@ export interface TTSResult {
2393
2481
  audio: string
2394
2482
  /** Audio format of the generated audio */
2395
2483
  format: string
2396
- /** Duration of the audio in seconds, if available */
2484
+ /** Duration of the audio file in seconds, if available */
2397
2485
  duration?: number
2486
+ /**
2487
+ * Character- or word-level timings, present when `timestamps: true` was
2488
+ * requested. Use this rather than `duration` to find where *speech* ends.
2489
+ */
2490
+ alignment?: TTSAlignment
2491
+ /**
2492
+ * Per-turn (or per-utterance) spans of the audio, present when
2493
+ * `timestamps: true` was requested and the provider reports segmentation.
2494
+ */
2495
+ segments?: Array<TTSSegment>
2398
2496
  /** Content type of the audio (e.g., 'audio/mp3') */
2399
2497
  contentType?: string
2400
2498
  /** Token usage information (if provided by the adapter) */
@@ -2403,6 +2501,174 @@ export interface TTSResult {
2403
2501
  artifacts?: Array<PersistedArtifactRef>
2404
2502
  }
2405
2503
 
2504
+ // ============================================================================
2505
+ // Voice Catalog Types
2506
+ // ============================================================================
2507
+
2508
+ /**
2509
+ * Where a voice in a provider's catalog came from.
2510
+ *
2511
+ * `'premade'` is the provider's own stock catalog. `'generated'` and
2512
+ * `'cloned'` are voices the account made, which is what `generateVoice()`
2513
+ * produces. `'professional'` covers a provider's curated or paid tiers.
2514
+ */
2515
+ export type VoiceOrigin = 'premade' | 'generated' | 'cloned' | 'professional'
2516
+
2517
+ /** One voice from a provider's catalog. */
2518
+ export interface CatalogVoice {
2519
+ /** Pass this to `generateSpeech()` as `voice` */
2520
+ voiceId: string
2521
+ /** Display name, when the provider stores one */
2522
+ name?: string
2523
+ /** Where the voice came from */
2524
+ origin?: VoiceOrigin
2525
+ /** Provider description of the voice */
2526
+ description?: string
2527
+ /** URL of a sample, when the provider hosts one */
2528
+ previewUrl?: string
2529
+ /** Provider labels, such as accent, age, or use case */
2530
+ labels?: Record<string, string>
2531
+ }
2532
+
2533
+ /** Options for listing a provider's voices. */
2534
+ export interface ListVoicesOptions {
2535
+ /**
2536
+ * Restrict the result to voices of these origins. Adapters filter server
2537
+ * side when the provider supports it, and in memory otherwise.
2538
+ */
2539
+ origins?: Array<VoiceOrigin>
2540
+ /**
2541
+ * Effective abort signal. Adapters forward this to the provider SDK when
2542
+ * supported.
2543
+ */
2544
+ abortSignal?: AbortSignal
2545
+ }
2546
+
2547
+ /** Result of listing a provider's voices. */
2548
+ export interface ListVoicesResult {
2549
+ /** The voices available to this account */
2550
+ voices: Array<CatalogVoice>
2551
+ }
2552
+
2553
+ // ============================================================================
2554
+ // Voice Creation Types
2555
+ // ============================================================================
2556
+
2557
+ /**
2558
+ * Options for creating a voice.
2559
+ *
2560
+ * Providers create voices in one of two ways, and some support both:
2561
+ * - **design** — synthesize a brand new voice from a text {@link prompt}.
2562
+ * - **clone** — derive a voice from {@link referenceAudio} of a real speaker.
2563
+ *
2564
+ * At least one of `prompt` / `referenceAudio` is required; which ones an
2565
+ * adapter accepts depends on the model. An adapter may require both — the
2566
+ * only adapter today, `elevenlabsVoiceDesign`, always needs `prompt` and
2567
+ * takes `referenceAudio` as an additional design reference.
2568
+ */
2569
+ export interface VoiceGenerationOptions<
2570
+ TProviderOptions extends object = object,
2571
+ > {
2572
+ /** The model to use for voice creation */
2573
+ model: string
2574
+ /** Text description of the voice to create, for design-capable models */
2575
+ prompt?: string
2576
+ /**
2577
+ * Reference audio of the speaker to clone - base64 string, base64 data URL,
2578
+ * File, Blob, or ArrayBuffer. For clone-capable models. Remote URLs are not
2579
+ * accepted; read the file and pass the bytes.
2580
+ */
2581
+ referenceAudio?: string | File | Blob | ArrayBuffer
2582
+ /**
2583
+ * Name to store the voice under in the provider's voice library. Providers
2584
+ * differ on what this implies — ElevenLabs only persists a designed voice
2585
+ * when a name is given. Read {@link GeneratedVoice.saved} to find out what
2586
+ * actually happened.
2587
+ */
2588
+ name?: string
2589
+ /** Human-readable description stored alongside the voice */
2590
+ description?: string
2591
+ /** Model-specific options for voice creation */
2592
+ modelOptions?: TProviderOptions
2593
+ /**
2594
+ * Internal logger threaded from the generateVoice() entry point. Adapters
2595
+ * must call logger.request() before the SDK call and logger.errors() in
2596
+ * catch blocks.
2597
+ */
2598
+ logger: InternalLogger
2599
+ /**
2600
+ * Effective abort signal composed by the activity from caller `abortSignal`
2601
+ * and/or `timeout`. Adapters should forward this to the provider SDK when
2602
+ * supported. Request-specific - never store on a global client config.
2603
+ */
2604
+ abortSignal?: AbortSignal
2605
+ }
2606
+
2607
+ /**
2608
+ * A single voice produced by {@link VoiceGenerationOptions}.
2609
+ */
2610
+ export interface GeneratedVoice {
2611
+ /**
2612
+ * The provider's voice identifier. Pass it straight back as the `voice`
2613
+ * option on `generateSpeech()`.
2614
+ */
2615
+ voiceId: string
2616
+ /** Base64-encoded preview audio, when the provider returns one */
2617
+ audio?: string
2618
+ /** Audio format of the preview (e.g. 'mp3') */
2619
+ format?: string
2620
+ /** Content type of the preview (e.g. 'audio/mpeg') */
2621
+ contentType?: string
2622
+ /** Duration of the preview in seconds, if available */
2623
+ duration?: number
2624
+ /** Language of the preview, if reported */
2625
+ language?: string
2626
+ /**
2627
+ * Whether the voice is persisted in the provider's voice library. Unsaved
2628
+ * voices are previews and generally expire.
2629
+ */
2630
+ saved: boolean
2631
+ /**
2632
+ * Whether the voice can be used in `generateSpeech()` yet. Required so a
2633
+ * caller never has to guess: every adapter states it outright.
2634
+ */
2635
+ status: VoiceTrainingStatus
2636
+ }
2637
+
2638
+ /**
2639
+ * Whether a created voice is usable.
2640
+ *
2641
+ * - `'ready'` — usable in `generateSpeech()` now. Every adapter today returns
2642
+ * this, because they all finish the voice inside `generateVoice()`.
2643
+ * - `'training'` — the provider accepted the request but is still building
2644
+ * the voice, so it is not usable yet. Reserved for providers that train
2645
+ * asynchronously; no adapter returns it yet, and reading the state back
2646
+ * will land with the first adapter that needs it.
2647
+ * - `'failed'` — the provider finished without producing a usable voice.
2648
+ */
2649
+ export type VoiceTrainingStatus = 'ready' | 'training' | 'failed'
2650
+
2651
+ /**
2652
+ * Result of voice creation.
2653
+ *
2654
+ * Design models typically return several candidates to choose between; clone
2655
+ * models return exactly one.
2656
+ */
2657
+ export interface VoiceResult {
2658
+ /** Unique identifier for the generation */
2659
+ id: string
2660
+ /** Model used for generation */
2661
+ model: string
2662
+ /** The voices produced, best-first when the provider ranks them */
2663
+ voices: Array<GeneratedVoice>
2664
+ /** The line spoken in the previews, when the provider generated one */
2665
+ previewText?: string
2666
+ /** Token usage information (if provided by the adapter) */
2667
+ usage?: TokenUsage
2668
+ /** Persisted artifact references for generated assets, when available */
2669
+ artifacts?: Array<PersistedArtifactRef>
2670
+ }
2671
+
2406
2672
  // ============================================================================
2407
2673
  // Transcription (Speech-to-Text) Types
2408
2674
  // ============================================================================
@@ -0,0 +1,48 @@
1
+ import { CUSTOM_EVENT } from '../custom-events'
2
+ import type { CustomEvent, StreamChunk } from '../types'
3
+ import { tanstackMetadata, withTanstackMetadata } from './merge-metadata'
4
+
5
+ /**
6
+ * High-volume CUSTOM names that stay in the durability batch.
7
+ * Everything else flushes as soon as it is emitted.
8
+ */
9
+ const BATCHED_CUSTOM_EVENT_NAMES = new Set<string>([
10
+ CUSTOM_EVENT.PROCESS_STDOUT,
11
+ CUSTOM_EVENT.PROCESS_STDERR,
12
+ 'sandbox.file',
13
+ 'sandbox.file.diff',
14
+ ])
15
+
16
+ function hasBatchHint(chunk: StreamChunk): boolean {
17
+ const tanstack = tanstackMetadata(chunk)
18
+ if (tanstack == null) return false
19
+ return Reflect.get(tanstack, 'batch') === true
20
+ }
21
+
22
+ /** Mark a CUSTOM chunk so the durability producer keeps it in the batch. */
23
+ export function withDurabilityBatchHint(chunk: CustomEvent): CustomEvent {
24
+ return withTanstackMetadata(chunk, { batch: true })
25
+ }
26
+
27
+ export function isDurabilityBatchedCustom(chunk: StreamChunk): boolean {
28
+ if (chunk.type !== 'CUSTOM') return false
29
+ if (BATCHED_CUSTOM_EVENT_NAMES.has(chunk.name)) return true
30
+ return hasBatchHint(chunk)
31
+ }
32
+
33
+ /**
34
+ * Drop the in-process batch hint so it does not sit in the log or on the wire.
35
+ */
36
+ export function stripDurabilityBatchHint(chunk: StreamChunk): StreamChunk {
37
+ const tanstack = tanstackMetadata(chunk)
38
+ if (tanstack == null || Reflect.get(tanstack, 'batch') !== true) return chunk
39
+ Reflect.deleteProperty(tanstack, 'batch')
40
+ const metadata = chunk.metadata
41
+ if (metadata != null && Object.keys(tanstack).length === 0) {
42
+ Reflect.deleteProperty(metadata, 'tanstack')
43
+ if (Object.keys(metadata).length === 0) {
44
+ Reflect.deleteProperty(chunk, 'metadata')
45
+ }
46
+ }
47
+ return chunk
48
+ }