@tanstack/ai 0.55.0 → 0.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/README.md +42 -16
  2. package/dist/esm/activities/chat/index.js +8 -6
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/middleware/types.d.ts +4 -3
  5. package/dist/esm/activities/chat/middleware/types.js.map +1 -1
  6. package/dist/esm/activities/chat/tools/tool-calls.d.ts +2 -2
  7. package/dist/esm/activities/chat/tools/tool-calls.js +3 -2
  8. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  9. package/dist/esm/activities/evaluate/adapter.d.ts +160 -0
  10. package/dist/esm/activities/evaluate/adapter.js +23 -0
  11. package/dist/esm/activities/evaluate/adapter.js.map +1 -0
  12. package/dist/esm/activities/evaluate/index.d.ts +255 -0
  13. package/dist/esm/activities/evaluate/index.js +317 -0
  14. package/dist/esm/activities/evaluate/index.js.map +1 -0
  15. package/dist/esm/activities/generateSpeech/adapter.d.ts +39 -1
  16. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -1
  17. package/dist/esm/activities/generateSpeech/index.d.ts +55 -5
  18. package/dist/esm/activities/generateSpeech/index.js +53 -3
  19. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  20. package/dist/esm/activities/generateVoice/adapter.d.ts +62 -0
  21. package/dist/esm/activities/generateVoice/adapter.js +23 -0
  22. package/dist/esm/activities/generateVoice/adapter.js.map +1 -0
  23. package/dist/esm/activities/generateVoice/index.d.ts +133 -0
  24. package/dist/esm/activities/generateVoice/index.js +184 -0
  25. package/dist/esm/activities/generateVoice/index.js.map +1 -0
  26. package/dist/esm/activities/index.d.ts +10 -4
  27. package/dist/esm/activities/index.js +14 -10
  28. package/dist/esm/activities/middleware/types.d.ts +1 -1
  29. package/dist/esm/client.d.ts +3 -2
  30. package/dist/esm/client.js +21 -3
  31. package/dist/esm/client.js.map +1 -1
  32. package/dist/esm/index.d.ts +4 -2
  33. package/dist/esm/index.js +6 -3
  34. package/dist/esm/middlewares/otel.js +2 -0
  35. package/dist/esm/middlewares/otel.js.map +1 -1
  36. package/dist/esm/realtime/index.d.ts +1 -1
  37. package/dist/esm/realtime/index.js +1 -1
  38. package/dist/esm/realtime/index.js.map +1 -1
  39. package/dist/esm/stream-to-response.js +12 -6
  40. package/dist/esm/stream-to-response.js.map +1 -1
  41. package/dist/esm/strip-to-spec-middleware.js +2 -1
  42. package/dist/esm/strip-to-spec-middleware.js.map +1 -1
  43. package/dist/esm/types.d.ts +243 -3
  44. package/dist/esm/utilities/durability-batch.d.ts +8 -0
  45. package/dist/esm/utilities/durability-batch.js +45 -0
  46. package/dist/esm/utilities/durability-batch.js.map +1 -0
  47. package/package.json +3 -3
  48. package/skills/ai-core/media-generation/SKILL.md +132 -6
  49. package/src/activities/chat/index.ts +11 -5
  50. package/src/activities/chat/middleware/types.ts +8 -2
  51. package/src/activities/chat/tools/tool-calls.ts +24 -5
  52. package/src/activities/evaluate/adapter.ts +212 -0
  53. package/src/activities/evaluate/index.ts +614 -0
  54. package/src/activities/generateSpeech/adapter.ts +47 -1
  55. package/src/activities/generateSpeech/index.ts +149 -8
  56. package/src/activities/generateVoice/adapter.ts +89 -0
  57. package/src/activities/generateVoice/index.ts +371 -0
  58. package/src/activities/index.ts +69 -0
  59. package/src/activities/middleware/types.ts +2 -0
  60. package/src/client.ts +35 -8
  61. package/src/index.ts +21 -0
  62. package/src/middlewares/otel.ts +2 -0
  63. package/src/realtime/index.ts +1 -1
  64. package/src/stream-to-response.ts +16 -6
  65. package/src/strip-to-spec-middleware.ts +2 -1
  66. package/src/types.ts +269 -3
  67. package/src/utilities/durability-batch.ts +48 -0
@@ -20,9 +20,11 @@ import type { AnyImageAdapter } from './generateImage/adapter'
20
20
  import type { AnyAudioAdapter } from './generateAudio/adapter'
21
21
  import type { AnyVideoAdapter } from './generateVideo/adapter'
22
22
  import type { AnyTTSAdapter } from './generateSpeech/adapter'
23
+ import type { AnyVoiceAdapter } from './generateVoice/adapter'
23
24
  import type { AnyTranscriptionAdapter } from './generateTranscription/adapter'
24
25
  import type { AnyEmbeddingAdapter } from './embed/adapter'
25
26
  import type { AnyRerankAdapter } from './rerank/adapter'
27
+ import type { AnyEvaluateAdapter } from './evaluate/adapter'
26
28
  import type { AnyWorldAdapter } from './generateWorld/adapter'
27
29
  import type { AnyLiveVideoAdapter } from './generateLiveVideo/adapter'
28
30
 
@@ -89,6 +91,46 @@ export {
89
91
  type AnyRerankAdapter,
90
92
  } from './rerank/adapter'
91
93
 
94
+ // ===========================
95
+ // Evaluate Activity
96
+ // ===========================
97
+
98
+ export {
99
+ kind as evaluateKind,
100
+ decide,
101
+ choice,
102
+ score,
103
+ boolean,
104
+ type EvaluateActivityOptions,
105
+ type EvaluateResult,
106
+ type EvaluateResultMeta,
107
+ type EvaluateProviderOptions,
108
+ type ChoiceAnswer,
109
+ type ScoreAnswer,
110
+ type BooleanAnswer,
111
+ type InferEvaluateAnswer,
112
+ } from './evaluate/index'
113
+
114
+ export {
115
+ BaseEvaluateAdapter,
116
+ type EvaluateAdapter,
117
+ type EvaluateAdapterConfig,
118
+ type AnyEvaluateAdapter,
119
+ type EvaluateOptions,
120
+ type EvaluateAdapterResult,
121
+ type EvaluateState,
122
+ type EvaluateInstructions,
123
+ type EvaluateJsonValue,
124
+ type WireQuestion,
125
+ type WireAnswer,
126
+ type WireChoiceQuestion,
127
+ type WireScoreQuestion,
128
+ type WireNoulQuestion,
129
+ type WireChoiceAnswer,
130
+ type WireScoreAnswer,
131
+ type WireNoulAnswer,
132
+ } from './evaluate/adapter'
133
+
92
134
  // ===========================
93
135
  // Image Activity
94
136
  // ===========================
@@ -162,6 +204,8 @@ export { snapToDurationOption } from './generateVideo/snap'
162
204
  export {
163
205
  kind as ttsKind,
164
206
  generateSpeech,
207
+ listVoices,
208
+ type ListVoicesActivityOptions,
165
209
  type TTSActivityOptions,
166
210
  type TTSActivityResult,
167
211
  type TTSProviderOptions,
@@ -171,9 +215,30 @@ export {
171
215
  BaseTTSAdapter,
172
216
  type TTSAdapter,
173
217
  type TTSAdapterConfig,
218
+ type TTSCapabilities,
174
219
  type AnyTTSAdapter,
175
220
  } from './generateSpeech/adapter'
176
221
 
222
+ // ===========================
223
+ // Voice Activity
224
+ // ===========================
225
+
226
+ export {
227
+ kind as voiceKind,
228
+ generateVoice,
229
+ createVoiceOptions,
230
+ type VoiceActivityOptions,
231
+ type VoiceActivityResult,
232
+ type VoiceProviderOptions,
233
+ } from './generateVoice/index'
234
+
235
+ export {
236
+ BaseVoiceAdapter,
237
+ type VoiceAdapter,
238
+ type VoiceAdapterConfig,
239
+ type AnyVoiceAdapter,
240
+ } from './generateVoice/adapter'
241
+
177
242
  // ===========================
178
243
  // Transcription Activity
179
244
  // ===========================
@@ -262,9 +327,11 @@ export type AIAdapter =
262
327
  | AnyAudioAdapter
263
328
  | AnyVideoAdapter
264
329
  | AnyTTSAdapter
330
+ | AnyVoiceAdapter
265
331
  | AnyTranscriptionAdapter
266
332
  | AnyEmbeddingAdapter
267
333
  | AnyRerankAdapter
334
+ | AnyEvaluateAdapter
268
335
  | AnyWorldAdapter
269
336
  | AnyLiveVideoAdapter
270
337
 
@@ -276,8 +343,10 @@ export type AdapterKind =
276
343
  | 'audio'
277
344
  | 'video'
278
345
  | 'tts'
346
+ | 'voice'
279
347
  | 'transcription'
280
348
  | 'embedding'
281
349
  | 'rerank'
350
+ | 'evaluate'
282
351
  | 'world'
283
352
  | 'liveVideo'
@@ -40,9 +40,11 @@ export type GenerationActivity =
40
40
  | 'video'
41
41
  | 'audio'
42
42
  | 'tts'
43
+ | 'voice'
43
44
  | 'transcription'
44
45
  | 'embedding'
45
46
  | 'rerank'
47
+ | 'evaluate'
46
48
  | 'summarize'
47
49
  | 'world'
48
50
  | 'liveVideo'
package/src/client.ts CHANGED
@@ -3,6 +3,7 @@ import type {
3
3
  ImageGenerationOptions,
4
4
  TTSOptions,
5
5
  TranscriptionOptions,
6
+ VoiceGenerationOptions,
6
7
  VideoGenerationOptions,
7
8
  WorldGenerationOptions,
8
9
  LiveVideoGenerationOptions,
@@ -12,6 +13,7 @@ export type GenerationKind =
12
13
  | 'image'
13
14
  | 'audio'
14
15
  | 'tts'
16
+ | 'voice'
15
17
  | 'video'
16
18
  | 'transcription'
17
19
  | 'world'
@@ -21,6 +23,7 @@ type GenerationInputByKind = {
21
23
  image: Omit<ImageGenerationOptions, 'logger' | 'model'>
22
24
  audio: Omit<AudioGenerationOptions, 'logger' | 'model'>
23
25
  tts: Omit<TTSOptions, 'logger' | 'model'>
26
+ voice: Omit<VoiceGenerationOptions, 'logger' | 'model'>
24
27
  video: Omit<VideoGenerationOptions, 'logger' | 'model'>
25
28
  transcription: Omit<TranscriptionOptions, 'logger' | 'model'>
26
29
  world: Omit<WorldGenerationOptions, 'logger' | 'model'>
@@ -38,6 +41,7 @@ const generationKinds = [
38
41
  'image',
39
42
  'audio',
40
43
  'tts',
44
+ 'voice',
41
45
  'video',
42
46
  'transcription',
43
47
  'world',
@@ -69,6 +73,31 @@ function assertGenerationKind(kind: unknown): asserts kind is GenerationKind {
69
73
  }
70
74
  }
71
75
 
76
+ /**
77
+ * The input field(s) that identify a generation body for a kind. Most kinds
78
+ * have exactly one; `voice` accepts either of its two creation modes, so any
79
+ * one of its keys is enough.
80
+ */
81
+ function requiredKeysForKind(kind: GenerationKind): Array<string> {
82
+ // Enumerated rather than defaulted so a new generation kind has to declare
83
+ // the field that identifies its body instead of silently inheriting
84
+ // `prompt`.
85
+ switch (kind) {
86
+ case 'tts':
87
+ return ['text']
88
+ case 'transcription':
89
+ return ['audio']
90
+ case 'voice':
91
+ return ['prompt', 'referenceAudio']
92
+ case 'image':
93
+ case 'audio':
94
+ case 'video':
95
+ case 'world':
96
+ case 'liveVideo':
97
+ return ['prompt']
98
+ }
99
+ }
100
+
72
101
  function assertInputForKind(
73
102
  kind: GenerationKind,
74
103
  input: unknown,
@@ -77,21 +106,19 @@ function assertInputForKind(
77
106
  throw new Error(`Generation ${kind} input must be an object.`)
78
107
  }
79
108
 
80
- const requiredKey =
81
- kind === 'tts' ? 'text' : kind === 'transcription' ? 'audio' : 'prompt'
109
+ const requiredKeys = requiredKeysForKind(kind)
82
110
 
83
- if (!hasOwnKey(input, requiredKey)) {
84
- throw new Error(`Generation ${kind} input must include ${requiredKey}.`)
111
+ if (!requiredKeys.some((key) => hasOwnKey(input, key))) {
112
+ throw new Error(
113
+ `Generation ${kind} input must include ${requiredKeys.join(' or ')}.`,
114
+ )
85
115
  }
86
116
  }
87
117
 
88
118
  function isInputForKind(kind: GenerationKind, input: unknown): boolean {
89
119
  if (!isRecord(input)) return false
90
120
 
91
- const requiredKey =
92
- kind === 'tts' ? 'text' : kind === 'transcription' ? 'audio' : 'prompt'
93
-
94
- return hasOwnKey(input, requiredKey)
121
+ return requiredKeysForKind(kind).some((key) => hasOwnKey(input, key))
95
122
  }
96
123
 
97
124
  function forwardedPropsFromEnvelope(
package/src/index.ts CHANGED
@@ -3,11 +3,17 @@ export {
3
3
  chat,
4
4
  summarize,
5
5
  rerank,
6
+ decide,
7
+ choice,
8
+ score,
9
+ boolean,
6
10
  generateImage,
7
11
  generateAudio,
8
12
  generateVideo,
9
13
  getVideoJobStatus,
10
14
  generateSpeech,
15
+ listVoices,
16
+ generateVoice,
11
17
  generateTranscription,
12
18
  embed,
13
19
  generateWorld,
@@ -22,6 +28,7 @@ export { createImageOptions } from './activities/generateImage/index'
22
28
  export { createAudioOptions } from './activities/generateAudio/index'
23
29
  export { createVideoOptions } from './activities/generateVideo/index'
24
30
  export { createSpeechOptions } from './activities/generateSpeech/index'
31
+ export { createVoiceOptions } from './activities/generateVoice/index'
25
32
  export { createTranscriptionOptions } from './activities/generateTranscription/index'
26
33
  export { createEmbedOptions } from './activities/embed/index'
27
34
  export { createWorldOptions } from './activities/generateWorld/index'
@@ -40,6 +47,9 @@ export type {
40
47
  AudioAdapter,
41
48
  AnyTTSAdapter,
42
49
  TTSAdapter,
50
+ TTSCapabilities,
51
+ AnyVoiceAdapter,
52
+ VoiceAdapter,
43
53
  AnyTranscriptionAdapter,
44
54
  TranscriptionAdapter,
45
55
  AnyVideoAdapter,
@@ -48,6 +58,14 @@ export type {
48
58
  EmbeddingAdapter,
49
59
  AnyRerankAdapter,
50
60
  RerankAdapter,
61
+ AnyEvaluateAdapter,
62
+ EvaluateAdapter,
63
+ ChoiceAnswer,
64
+ ScoreAnswer,
65
+ BooleanAnswer,
66
+ EvaluateResult,
67
+ WireQuestion,
68
+ WireAnswer,
51
69
  AnyWorldAdapter,
52
70
  WorldAdapter,
53
71
  AnyLiveVideoAdapter,
@@ -57,6 +75,9 @@ export type {
57
75
  // Rerank adapter base + types
58
76
  export { BaseRerankAdapter } from './activities/rerank/adapter'
59
77
 
78
+ // Evaluate adapter base + types
79
+ export { BaseEvaluateAdapter } from './activities/evaluate/adapter'
80
+
60
81
  // Tool definition
61
82
  export {
62
83
  toolDefinition,
@@ -87,9 +87,11 @@ const OPERATION_NAME: Record<GenerationActivity, string> = {
87
87
  video: 'video_generation',
88
88
  audio: 'audio_generation',
89
89
  tts: 'text_to_speech',
90
+ voice: 'voice_generation',
90
91
  transcription: 'transcription',
91
92
  embedding: 'embeddings',
92
93
  rerank: 'rerank',
94
+ evaluate: 'evaluate',
93
95
  summarize: 'summarize',
94
96
  world: 'world_generation',
95
97
  liveVideo: 'live_video_generation',
@@ -22,7 +22,7 @@ export type * from './types'
22
22
  * // On the server (e.g. inside a server route or framework server
23
23
  * // function), mint an ephemeral token for the client:
24
24
  * const token = await realtimeToken({
25
- * adapter: openaiRealtimeToken({ model: 'gpt-realtime' }),
25
+ * adapter: openaiRealtimeToken({ model: 'gpt-realtime-2.1' }),
26
26
  * })
27
27
  * ```
28
28
  */
@@ -9,6 +9,10 @@ import { notifyRunDisconnected } from './delivery-disconnect'
9
9
  import { resolveResumeRunId } from './stream-durability'
10
10
  import { EventType } from './types'
11
11
  import { toWireChunk } from './strip-to-spec-middleware'
12
+ import {
13
+ isDurabilityBatchedCustom,
14
+ stripDurabilityBatchHint,
15
+ } from './utilities/durability-batch'
12
16
  import { resolveDebugOption } from './logger/resolve'
13
17
  import { runErrorEventToError } from './utilities/errors'
14
18
  import type { LockStore } from './activities/chat/middleware/locks'
@@ -319,23 +323,29 @@ function resolveBatchSize(batch: number | undefined): number {
319
323
 
320
324
  /**
321
325
  * Boundaries at which the batching producer flushes early, regardless of the
322
- * batch size — the run-start marker, terminal events, and tool-call ends.
323
- * Flushing here keeps the durability log promptly consistent at semantically
324
- * meaningful points.
326
+ * batch size: run-start, terminals, tool-call ends, and CUSTOM events that
327
+ * are not high-volume adapter output.
325
328
  *
326
329
  * `RUN_STARTED` matters especially for one-shot activities (image, speech,
327
330
  * transcription, summarize): they emit `RUN_STARTED`, then await the provider
328
331
  * for seconds, then a terminal. Without flushing `RUN_STARTED` the log stays
329
332
  * empty for the whole run, so a mount-time `joinRun` finds nothing and its
330
- * empty-log deadline fast-fails as "run gone" — even though the run is alive.
333
+ * empty-log deadline fast-fails as "run gone" even though the run is alive.
331
334
  * Flushing it immediately makes the run resumable from the instant it starts.
335
+ *
336
+ * CUSTOM progress events (compaction, tool progress, middleware) flush at
337
+ * emit time so a live indicator can render. `process.stdout`,
338
+ * `process.stderr`, `sandbox.file`, and `sandbox.file.diff` stay batched.
339
+ * `emitCustomEvent(name, value, { batch: true })` opts a single event into
340
+ * that same batch.
332
341
  */
333
342
  function isDurabilityFlushBoundary(chunk: StreamChunk): boolean {
334
343
  return (
335
344
  chunk.type === 'RUN_STARTED' ||
336
345
  chunk.type === 'RUN_FINISHED' ||
337
346
  chunk.type === 'RUN_ERROR' ||
338
- chunk.type === 'TOOL_CALL_END'
347
+ chunk.type === 'TOOL_CALL_END' ||
348
+ (chunk.type === 'CUSTOM' && !isDurabilityBatchedCustom(chunk))
339
349
  )
340
350
  }
341
351
 
@@ -446,7 +456,7 @@ export function durableStreamSource<TOffset extends string>(
446
456
 
447
457
  async function* flush(): AsyncIterable<StreamChunk> {
448
458
  if (batch.length === 0) return
449
- const toForward = batch
459
+ const toForward = batch.map(stripDurabilityBatchHint)
450
460
  batch = []
451
461
  // Tag each chunk with the exact backend offset. Requiring one opaque
452
462
  // token per chunk preserves exact-once resume at any batch size.
@@ -6,6 +6,7 @@ import {
6
6
  tanstackMetadata,
7
7
  withTanstackMetadata,
8
8
  } from './utilities/merge-metadata'
9
+ import { stripDurabilityBatchHint } from './utilities/durability-batch'
9
10
  import { normalizeStreamChunk } from './utilities/normalize-stream-chunk'
10
11
  import { isSpecTopLevelKey } from './utilities/spec-event-keys'
11
12
 
@@ -54,5 +55,5 @@ export function toWireChunk(
54
55
  chunk: StreamChunk | AdapterYieldChunk,
55
56
  ): StreamChunk {
56
57
  const [normalized] = normalizeStreamChunk(chunk)
57
- return stripToSpec(normalized ?? chunk)
58
+ return stripDurabilityBatchHint(stripToSpec(normalized ?? chunk))
58
59
  }