@markusylisiurunen/tau 0.3.55 → 0.3.57

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/dist/core/auth/credential_store.js +1 -9
  2. package/dist/core/auth/credential_store.js.map +1 -1
  3. package/dist/core/cli.js +3 -0
  4. package/dist/core/cli.js.map +1 -1
  5. package/dist/core/config/runtime.js +4 -4
  6. package/dist/core/config/runtime.js.map +1 -1
  7. package/dist/core/config/runtime_config_snapshot.js +1 -1
  8. package/dist/core/config/runtime_config_snapshot.js.map +1 -1
  9. package/dist/core/history/history_manager.js +35 -25
  10. package/dist/core/history/history_manager.js.map +1 -1
  11. package/dist/core/models/catalog.js +11 -45
  12. package/dist/core/models/catalog.js.map +1 -1
  13. package/dist/core/models/cli.js +49 -0
  14. package/dist/core/models/cli.js.map +1 -0
  15. package/dist/core/models/remote_catalog.js +394 -0
  16. package/dist/core/models/remote_catalog.js.map +1 -0
  17. package/dist/core/personas.js +3 -3
  18. package/dist/core/static/tau_docs/getting-started.md +5 -1
  19. package/dist/core/static/tau_docs/history.md +1 -1
  20. package/dist/core/static/tau_docs/models.md +12 -2
  21. package/dist/core/static/tau_docs/node-sdk.md +7 -1
  22. package/dist/core/static/tau_docs/security.md +5 -0
  23. package/dist/core/static/tau_docs/telegram.md +2 -2
  24. package/dist/core/static/tau_docs/tui.md +2 -2
  25. package/dist/core/telegram/adapter.js +2 -2
  26. package/dist/core/telegram/adapter.js.map +1 -1
  27. package/dist/core/telegram/local_session_client.js +2 -1
  28. package/dist/core/telegram/local_session_client.js.map +1 -1
  29. package/dist/core/telegram/tts.js +3 -4
  30. package/dist/core/telegram/tts.js.map +1 -1
  31. package/dist/core/utils/gemini_speech.js +435 -150
  32. package/dist/core/utils/gemini_speech.js.map +1 -1
  33. package/dist/core/utils/model_stream.js +0 -10
  34. package/dist/core/utils/model_stream.js.map +1 -1
  35. package/dist/core/utils/openai_transcription.js +164 -23
  36. package/dist/core/utils/openai_transcription.js.map +1 -1
  37. package/dist/core/utils/speech_to_text.js +1 -0
  38. package/dist/core/utils/speech_to_text.js.map +1 -1
  39. package/dist/core/version.js +2 -1
  40. package/dist/core/version.js.map +1 -1
  41. package/dist/execution/execution_environment.js.map +1 -1
  42. package/dist/execution/tool_backend_execution_environment.js +2 -1
  43. package/dist/execution/tool_backend_execution_environment.js.map +1 -1
  44. package/dist/history/worker/index.js +22 -7
  45. package/dist/history/worker/index.js.map +1 -1
  46. package/dist/host/execution_runtime.js +3 -1
  47. package/dist/host/execution_runtime.js.map +1 -1
  48. package/dist/host/local_session_host.js +37 -11
  49. package/dist/host/local_session_host.js.map +1 -1
  50. package/dist/main.js +74 -8
  51. package/dist/main.js.map +1 -1
  52. package/dist/sdk/index.d.ts +1 -1
  53. package/dist/sdk/index.js.map +1 -1
  54. package/dist/sdk/local_client.d.ts +4 -1
  55. package/dist/sdk/local_client.js +35 -9
  56. package/dist/sdk/local_client.js.map +1 -1
  57. package/dist/sdk/types.d.ts +14 -0
  58. package/dist/tui/session_chat_controller.js +3 -0
  59. package/dist/tui/session_chat_controller.js.map +1 -1
  60. package/dist/tui/speech_playback.js +145 -82
  61. package/dist/tui/speech_playback.js.map +1 -1
  62. package/package.json +4 -4
  63. package/dist/core/models/tau_extensions.js +0 -27
  64. package/dist/core/models/tau_extensions.js.map +0 -1
@@ -5,15 +5,15 @@ const DEFAULT_GEMINI_SPEECH_REWRITE_MODEL = "gemini-3.7-flash";
5
5
  const DEFAULT_GEMINI_SPEECH_REWRITE_THINKING_LEVEL = "low";
6
6
  const DEFAULT_GEMINI_SPEECH_TTS_MODEL = "gemini-3.1-flash-tts-preview";
7
7
  const DEFAULT_GEMINI_TTS_VOICE_NAME = "Despina";
8
- const DEFAULT_TTS_SAMPLE_RATE_HZ = 24000;
9
- const DEFAULT_TTS_CHANNEL_COUNT = 1;
10
- const DEFAULT_TTS_BITS_PER_SAMPLE = 16;
8
+ export const GEMINI_SPEECH_SAMPLE_RATE_HZ = 24000;
9
+ export const GEMINI_SPEECH_CHANNEL_COUNT = 1;
10
+ export const GEMINI_SPEECH_BITS_PER_SAMPLE = 16;
11
11
  const DEFAULT_TTS_MAX_ATTEMPTS = 3;
12
12
  const SPEECH_REWRITE_TIMEOUT_MS = 60_000;
13
- const PROGRESSIVE_TTS_CONCURRENCY = 6;
14
13
  const COMPLETE_TTS_CONCURRENCY = 3;
15
- const PROGRESSIVE_SPEECH_CHUNK_CHARACTERS = 500;
16
- const COMPLETE_SPEECH_CHUNK_CHARACTERS = 1000;
14
+ const MAX_SPEECH_SEGMENT_SECONDS = 120;
15
+ const ESTIMATED_SPEECH_CHARACTERS_PER_SECOND = 17;
16
+ const MAX_SPEECH_SEGMENT_WEIGHT = MAX_SPEECH_SEGMENT_SECONDS * ESTIMATED_SPEECH_CHARACTERS_PER_SECOND;
17
17
  const MAX_SPEECH_SOURCE_CHARACTERS = 10_000;
18
18
  const MAX_SPOKEN_TEXT_CHARACTERS = 10_000;
19
19
  const MAX_SPEECH_PCM_BYTES = 32 * 1024 * 1024;
@@ -48,77 +48,26 @@ class GeminiTtsOutputLimitError extends Error {
48
48
  this.name = "GeminiTtsOutputLimitError";
49
49
  }
50
50
  }
51
- export async function* streamGeminiSpeechAudio(options) {
52
- const sourceText = options.sourceText.trim();
53
- if (!sourceText) {
54
- throw new Error("speech source text was empty");
55
- }
56
- if (exceedsUnicodeCharacterLimit(sourceText, MAX_SPEECH_SOURCE_CHARACTERS)) {
57
- throw new Error("speech source text exceeds 10,000 characters");
58
- }
59
- const apiKey = options.apiKey.trim();
60
- if (!apiKey) {
61
- throw new Error("missing Gemini API key");
62
- }
63
- const fetchImpl = options.fetchImpl ?? fetch;
64
- const rewriteModel = options.rewriteModel ?? DEFAULT_GEMINI_SPEECH_REWRITE_MODEL;
65
- const ttsModel = options.ttsModel ?? DEFAULT_GEMINI_SPEECH_TTS_MODEL;
66
- const voiceName = options.voiceName ?? DEFAULT_GEMINI_TTS_VOICE_NAME;
51
+ export async function* generateGeminiSpeechAudio(options) {
67
52
  const abortController = createLinkedAbortController(options.signal);
68
53
  let completed = false;
69
54
  try {
70
- await options.onStageChange?.("rewriting");
71
- const rewriteController = createLinkedAbortController(abortController.signal);
72
- const rewriteTimeout = setTimeout(() => rewriteController.abort(), SPEECH_REWRITE_TIMEOUT_MS);
73
- rewriteTimeout.unref?.();
74
- let spokenText;
75
- try {
76
- spokenText = await rewriteTextForSpeech({
77
- apiKey,
78
- model: rewriteModel,
79
- sourceText,
80
- fetchImpl,
81
- signal: rewriteController.signal,
82
- });
83
- }
84
- catch (error) {
85
- if (!abortController.signal.aborted && rewriteController.signal.aborted) {
86
- throw new Error("speech rewrite timed out after 1 minute");
87
- }
88
- throw error;
89
- }
90
- finally {
91
- clearTimeout(rewriteTimeout);
92
- rewriteController.dispose();
93
- }
94
- if (exceedsUnicodeCharacterLimit(spokenText, MAX_SPOKEN_TEXT_CHARACTERS)) {
95
- throw new Error("rewritten speech text exceeds 10,000 characters");
96
- }
97
- const spokenChunks = splitSpeechChunks(spokenText, options.deliveryMode);
98
- await options.onStageChange?.("generating");
99
- await options.onChunkProgress?.({ ready: 0, total: spokenChunks.length });
100
- for await (const chunk of synthesizeSpeechAudioChunksInOrder({
101
- apiKey,
102
- model: ttsModel,
103
- voiceName,
104
- spokenChunks,
105
- fetchImpl,
55
+ const prepared = await prepareGeminiSpeech(options, abortController.signal);
56
+ for await (const chunk of synthesizeSpeechAudioSegmentsInOrder({
57
+ ...prepared,
106
58
  signal: abortController.signal,
107
- maxAttempts: options.maxTtsAttempts ?? DEFAULT_TTS_MAX_ATTEMPTS,
108
- concurrency: options.deliveryMode === "progressive"
109
- ? PROGRESSIVE_TTS_CONCURRENCY
110
- : COMPLETE_TTS_CONCURRENCY,
111
- onChunkProgress: options.onChunkProgress,
59
+ concurrency: COMPLETE_TTS_CONCURRENCY,
60
+ onSegmentProgress: options.onSegmentProgress,
112
61
  abortOnFailure: () => abortController.abort(),
113
62
  })) {
114
63
  yield {
115
64
  index: chunk.index,
116
- total: spokenChunks.length,
65
+ total: prepared.spokenSegments.length,
117
66
  audio: encodeWaveFile({
118
67
  pcmAudio: chunk.pcmAudio,
119
- sampleRateHz: DEFAULT_TTS_SAMPLE_RATE_HZ,
120
- channelCount: DEFAULT_TTS_CHANNEL_COUNT,
121
- bitsPerSample: DEFAULT_TTS_BITS_PER_SAMPLE,
68
+ sampleRateHz: GEMINI_SPEECH_SAMPLE_RATE_HZ,
69
+ channelCount: GEMINI_SPEECH_CHANNEL_COUNT,
70
+ bitsPerSample: GEMINI_SPEECH_BITS_PER_SAMPLE,
122
71
  }),
123
72
  mimeType: "audio/wav",
124
73
  };
@@ -132,6 +81,119 @@ export async function* streamGeminiSpeechAudio(options) {
132
81
  abortController.dispose();
133
82
  }
134
83
  }
84
+ export async function* streamGeminiSpeechPcm(options) {
85
+ const abortController = createLinkedAbortController(options.signal);
86
+ let completed = false;
87
+ let totalPcmBytes = 0;
88
+ try {
89
+ const prepared = await prepareGeminiSpeech(options, abortController.signal);
90
+ const total = prepared.spokenSegments.length;
91
+ const accountAudio = (audio) => {
92
+ totalPcmBytes += audio.length;
93
+ if (totalPcmBytes > MAX_SPEECH_PCM_BYTES) {
94
+ throw new Error("generated speech audio exceeds the 32 MiB limit");
95
+ }
96
+ };
97
+ const prefetch = (index) => collectStreamingSpeechSegment({
98
+ ...prepared,
99
+ spokenText: prepared.spokenSegments[index],
100
+ signal: abortController.signal,
101
+ accountAudio,
102
+ }).then((audio) => ({ audio }), (error) => ({ error }));
103
+ let ready = 0;
104
+ const firstStream = streamSpeechSegmentWithRetries({
105
+ ...prepared,
106
+ spokenText: prepared.spokenSegments[0],
107
+ signal: abortController.signal,
108
+ initialBufferBytes: Math.max(0, Math.trunc(options.initialBufferBytes ?? 0)),
109
+ });
110
+ const firstIterator = firstStream[Symbol.asyncIterator]();
111
+ let nextAudio = firstIterator.next();
112
+ let prefetched = total > 1 ? prefetch(1) : undefined;
113
+ while (true) {
114
+ const next = await nextAudio;
115
+ if (next.done) {
116
+ break;
117
+ }
118
+ accountAudio(next.value);
119
+ yield { index: 0, total, audio: next.value };
120
+ nextAudio = firstIterator.next();
121
+ }
122
+ ready += 1;
123
+ await options.onSegmentProgress?.({ ready, total });
124
+ for (let index = 1; index < total; index += 1) {
125
+ const outcome = await prefetched;
126
+ if ("error" in outcome) {
127
+ throw outcome.error instanceof Error
128
+ ? outcome.error
129
+ : new Error("Gemini TTS request failed");
130
+ }
131
+ prefetched = index + 1 < total ? prefetch(index + 1) : undefined;
132
+ yield { index, total, audio: outcome.audio };
133
+ ready += 1;
134
+ await options.onSegmentProgress?.({ ready, total });
135
+ }
136
+ completed = true;
137
+ }
138
+ finally {
139
+ if (!completed) {
140
+ abortController.abort();
141
+ }
142
+ abortController.dispose();
143
+ }
144
+ }
145
+ async function prepareGeminiSpeech(options, signal) {
146
+ const sourceText = options.sourceText.trim();
147
+ if (!sourceText) {
148
+ throw new Error("speech source text was empty");
149
+ }
150
+ if (exceedsUnicodeCharacterLimit(sourceText, MAX_SPEECH_SOURCE_CHARACTERS)) {
151
+ throw new Error("speech source text exceeds 10,000 characters");
152
+ }
153
+ const apiKey = options.apiKey.trim();
154
+ if (!apiKey) {
155
+ throw new Error("missing Gemini API key");
156
+ }
157
+ const fetchImpl = options.fetchImpl ?? fetch;
158
+ await options.onStageChange?.("rewriting");
159
+ const rewriteController = createLinkedAbortController(signal);
160
+ const rewriteTimeout = setTimeout(() => rewriteController.abort(), SPEECH_REWRITE_TIMEOUT_MS);
161
+ rewriteTimeout.unref?.();
162
+ let spokenText;
163
+ try {
164
+ spokenText = await rewriteTextForSpeech({
165
+ apiKey,
166
+ model: options.rewriteModel ?? DEFAULT_GEMINI_SPEECH_REWRITE_MODEL,
167
+ sourceText,
168
+ fetchImpl,
169
+ signal: rewriteController.signal,
170
+ });
171
+ }
172
+ catch (error) {
173
+ if (!signal.aborted && rewriteController.signal.aborted) {
174
+ throw new Error("speech rewrite timed out after 1 minute");
175
+ }
176
+ throw error;
177
+ }
178
+ finally {
179
+ clearTimeout(rewriteTimeout);
180
+ rewriteController.dispose();
181
+ }
182
+ if (exceedsUnicodeCharacterLimit(spokenText, MAX_SPOKEN_TEXT_CHARACTERS)) {
183
+ throw new Error("rewritten speech text exceeds 10,000 characters");
184
+ }
185
+ const spokenSegments = splitSpeechSegments(spokenText);
186
+ await options.onStageChange?.("generating");
187
+ await options.onSegmentProgress?.({ ready: 0, total: spokenSegments.length });
188
+ return {
189
+ apiKey,
190
+ model: options.ttsModel ?? DEFAULT_GEMINI_SPEECH_TTS_MODEL,
191
+ voiceName: options.voiceName ?? DEFAULT_GEMINI_TTS_VOICE_NAME,
192
+ spokenSegments,
193
+ fetchImpl,
194
+ maxAttempts: options.maxTtsAttempts ?? DEFAULT_TTS_MAX_ATTEMPTS,
195
+ };
196
+ }
135
197
  async function rewriteTextForSpeech(args) {
136
198
  const payload = await requestGeminiGenerateContent({
137
199
  apiKey: args.apiKey,
@@ -161,8 +223,8 @@ async function rewriteTextForSpeech(args) {
161
223
  }
162
224
  return rewrittenText;
163
225
  }
164
- async function* synthesizeSpeechAudioChunksInOrder(args) {
165
- const total = args.spokenChunks.length;
226
+ async function* synthesizeSpeechAudioSegmentsInOrder(args) {
227
+ const total = args.spokenSegments.length;
166
228
  const results = new Array(total);
167
229
  const concurrency = Math.max(1, Math.trunc(args.concurrency));
168
230
  let nextIndex = 0;
@@ -202,14 +264,9 @@ async function* synthesizeSpeechAudioChunksInOrder(args) {
202
264
  return;
203
265
  }
204
266
  try {
205
- const pcmAudio = await synthesizeSpeechAudioChunk({
206
- apiKey: args.apiKey,
207
- model: args.model,
208
- voiceName: args.voiceName,
209
- spokenText: args.spokenChunks[index],
210
- fetchImpl: args.fetchImpl,
211
- signal: args.signal,
212
- maxAttempts: args.maxAttempts,
267
+ const pcmAudio = await synthesizeSpeechAudioSegment({
268
+ ...args,
269
+ spokenText: args.spokenSegments[index],
213
270
  });
214
271
  totalPcmBytes += pcmAudio.length;
215
272
  if (totalPcmBytes > MAX_SPEECH_PCM_BYTES) {
@@ -217,12 +274,12 @@ async function* synthesizeSpeechAudioChunksInOrder(args) {
217
274
  }
218
275
  results[index] = pcmAudio;
219
276
  ready += 1;
220
- await args.onChunkProgress?.({ ready, total });
277
+ await args.onSegmentProgress?.({ ready, total });
221
278
  wake();
222
279
  }
223
- catch (err) {
280
+ catch (error) {
224
281
  if (failure === undefined) {
225
- failure = err;
282
+ failure = error;
226
283
  }
227
284
  args.abortOnFailure?.();
228
285
  wake();
@@ -252,40 +309,19 @@ async function* synthesizeSpeechAudioChunksInOrder(args) {
252
309
  args.signal?.removeEventListener("abort", onAbort);
253
310
  }
254
311
  }
255
- async function synthesizeSpeechAudioChunk(args) {
312
+ async function synthesizeSpeechAudioSegment(args) {
256
313
  const maxAttempts = Math.max(1, Math.trunc(args.maxAttempts));
257
314
  let lastError;
258
- for (let attempt = 1; attempt <= maxAttempts; attempt++) {
315
+ for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
259
316
  try {
260
317
  const payload = await requestGeminiGenerateContent({
261
318
  apiKey: args.apiKey,
262
319
  model: args.model,
263
320
  fetchImpl: args.fetchImpl,
264
321
  signal: args.signal,
265
- body: {
266
- contents: [
267
- {
268
- parts: [
269
- {
270
- text: buildSpeechSynthesisPrompt(args.spokenText),
271
- },
272
- ],
273
- },
274
- ],
275
- generationConfig: {
276
- responseModalities: ["AUDIO"],
277
- maxOutputTokens: TTS_MAX_OUTPUT_TOKENS,
278
- speechConfig: {
279
- voiceConfig: {
280
- prebuiltVoiceConfig: {
281
- voiceName: args.voiceName,
282
- },
283
- },
284
- },
285
- },
286
- },
322
+ body: buildSpeechSynthesisRequest(args.spokenText, args.voiceName),
287
323
  });
288
- assertGeminiTtsDidNotReachOutputLimit(payload);
324
+ assertGeminiTtsCompleted(payload);
289
325
  const audioData = extractGeminiInlineAudioData(payload);
290
326
  if (!audioData) {
291
327
  throw new GeminiTtsResponseError("Gemini TTS response did not include audio data");
@@ -302,8 +338,139 @@ async function synthesizeSpeechAudioChunk(args) {
302
338
  }
303
339
  throw lastError instanceof Error ? lastError : new Error("Gemini TTS request failed");
304
340
  }
341
+ async function* streamSpeechSegmentWithRetries(args) {
342
+ const maxAttempts = Math.max(1, Math.trunc(args.maxAttempts));
343
+ let lastError;
344
+ for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
345
+ const bufferedAudio = [];
346
+ let bufferedAudioBytes = 0;
347
+ let emittedAudio = false;
348
+ try {
349
+ for await (const audio of requestGeminiSpeechStream(args)) {
350
+ if (!emittedAudio && bufferedAudioBytes < args.initialBufferBytes) {
351
+ bufferedAudio.push(audio);
352
+ bufferedAudioBytes += audio.length;
353
+ if (bufferedAudioBytes < args.initialBufferBytes) {
354
+ continue;
355
+ }
356
+ emittedAudio = true;
357
+ yield Buffer.concat(bufferedAudio);
358
+ continue;
359
+ }
360
+ emittedAudio = true;
361
+ yield audio;
362
+ }
363
+ if (bufferedAudio.length > 0 && !emittedAudio) {
364
+ emittedAudio = true;
365
+ yield Buffer.concat(bufferedAudio);
366
+ }
367
+ return;
368
+ }
369
+ catch (error) {
370
+ lastError = error;
371
+ if (emittedAudio ||
372
+ args.signal?.aborted ||
373
+ !isRetryableTtsError(error) ||
374
+ attempt >= maxAttempts) {
375
+ throw error;
376
+ }
377
+ await waitForRetryDelay(attempt, args.signal);
378
+ }
379
+ }
380
+ throw lastError instanceof Error ? lastError : new Error("Gemini TTS request failed");
381
+ }
382
+ async function collectStreamingSpeechSegment(args) {
383
+ const maxAttempts = Math.max(1, Math.trunc(args.maxAttempts));
384
+ let lastError;
385
+ for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
386
+ const chunks = [];
387
+ try {
388
+ for await (const audio of requestGeminiSpeechStream(args)) {
389
+ chunks.push(audio);
390
+ }
391
+ }
392
+ catch (error) {
393
+ lastError = error;
394
+ if (args.signal?.aborted || !isRetryableTtsError(error) || attempt >= maxAttempts) {
395
+ throw error;
396
+ }
397
+ await waitForRetryDelay(attempt, args.signal);
398
+ continue;
399
+ }
400
+ const audio = Buffer.concat(chunks);
401
+ args.accountAudio(audio);
402
+ return audio;
403
+ }
404
+ throw lastError instanceof Error ? lastError : new Error("Gemini TTS request failed");
405
+ }
406
+ async function* requestGeminiSpeechStream(args) {
407
+ const response = await requestGeminiResponse({
408
+ apiKey: args.apiKey,
409
+ model: args.model,
410
+ method: "streamGenerateContent?alt=sse",
411
+ fetchImpl: args.fetchImpl,
412
+ signal: args.signal,
413
+ body: buildSpeechSynthesisRequest(args.spokenText, args.voiceName),
414
+ });
415
+ if (!response.body) {
416
+ throw new GeminiTtsResponseError("Gemini TTS streaming response did not include a body");
417
+ }
418
+ let receivedAudio = false;
419
+ let completed = false;
420
+ for await (const payload of parseGeminiSse(response.body)) {
421
+ const finishReason = getGeminiFinishReason(payload);
422
+ if (finishReason) {
423
+ assertGeminiTtsFinishReason(finishReason);
424
+ completed = finishReason === "STOP";
425
+ }
426
+ for (const audioData of extractGeminiInlineAudioDataParts(payload)) {
427
+ receivedAudio = true;
428
+ yield Buffer.from(audioData, "base64");
429
+ }
430
+ }
431
+ if (!receivedAudio) {
432
+ throw new GeminiTtsResponseError("Gemini TTS response did not include audio data");
433
+ }
434
+ if (!completed) {
435
+ throw new GeminiTtsResponseError("Gemini TTS stream ended without a stop response");
436
+ }
437
+ }
438
+ function buildSpeechSynthesisRequest(spokenText, voiceName) {
439
+ return {
440
+ contents: [
441
+ {
442
+ parts: [
443
+ {
444
+ text: buildSpeechSynthesisPrompt(spokenText),
445
+ },
446
+ ],
447
+ },
448
+ ],
449
+ generationConfig: {
450
+ responseModalities: ["AUDIO"],
451
+ maxOutputTokens: TTS_MAX_OUTPUT_TOKENS,
452
+ speechConfig: {
453
+ voiceConfig: {
454
+ prebuiltVoiceConfig: {
455
+ voiceName,
456
+ },
457
+ },
458
+ },
459
+ },
460
+ };
461
+ }
305
462
  async function requestGeminiGenerateContent(args) {
306
- const response = await args.fetchImpl(`${GEMINI_GENERATE_CONTENT_BASE_URL}/${encodeURIComponent(args.model)}:generateContent`, {
463
+ const response = await requestGeminiResponse({ ...args, method: "generateContent" });
464
+ const responseText = await response.text();
465
+ try {
466
+ return responseText ? JSON.parse(responseText) : undefined;
467
+ }
468
+ catch {
469
+ return undefined;
470
+ }
471
+ }
472
+ async function requestGeminiResponse(args) {
473
+ const response = await args.fetchImpl(`${GEMINI_GENERATE_CONTENT_BASE_URL}/${encodeURIComponent(args.model)}:${args.method}`, {
307
474
  method: "POST",
308
475
  headers: {
309
476
  "Content-Type": "application/json",
@@ -312,6 +479,9 @@ async function requestGeminiGenerateContent(args) {
312
479
  body: JSON.stringify(args.body),
313
480
  signal: args.signal,
314
481
  });
482
+ if (response.ok) {
483
+ return response;
484
+ }
315
485
  const responseText = await response.text();
316
486
  let payload;
317
487
  try {
@@ -320,15 +490,69 @@ async function requestGeminiGenerateContent(args) {
320
490
  catch {
321
491
  payload = undefined;
322
492
  }
323
- if (!response.ok) {
324
- const parsed = errorPayloadSchema.safeParse(payload);
325
- const fallbackMessage = responseText.trim() || `HTTP ${response.status}`;
326
- const message = parsed.success
327
- ? (parsed.data.error?.message ?? fallbackMessage)
328
- : fallbackMessage;
329
- throw new GeminiApiError(message, response.status);
493
+ const parsed = errorPayloadSchema.safeParse(payload);
494
+ const fallbackMessage = responseText.trim() || `HTTP ${response.status}`;
495
+ const message = parsed.success
496
+ ? (parsed.data.error?.message ?? fallbackMessage)
497
+ : fallbackMessage;
498
+ throw new GeminiApiError(message, response.status);
499
+ }
500
+ async function* parseGeminiSse(body) {
501
+ const reader = body.getReader();
502
+ const decoder = new TextDecoder();
503
+ let buffered = "";
504
+ let completed = false;
505
+ try {
506
+ while (true) {
507
+ const { done, value } = await reader.read();
508
+ buffered += decoder.decode(value, { stream: !done });
509
+ let boundary = findSseEventBoundary(buffered);
510
+ while (boundary) {
511
+ const event = buffered.slice(0, boundary.index);
512
+ buffered = buffered.slice(boundary.index + boundary.length);
513
+ const payload = parseSseEvent(event);
514
+ if (payload !== undefined) {
515
+ yield payload;
516
+ }
517
+ boundary = findSseEventBoundary(buffered);
518
+ }
519
+ if (done) {
520
+ const payload = parseSseEvent(buffered);
521
+ if (payload !== undefined) {
522
+ yield payload;
523
+ }
524
+ completed = true;
525
+ return;
526
+ }
527
+ }
528
+ }
529
+ finally {
530
+ if (!completed) {
531
+ await reader.cancel().catch(() => { });
532
+ }
533
+ reader.releaseLock();
534
+ }
535
+ }
536
+ function findSseEventBoundary(text) {
537
+ const match = /\r?\n\r?\n/u.exec(text);
538
+ return match ? { index: match.index, length: match[0].length } : undefined;
539
+ }
540
+ function parseSseEvent(event) {
541
+ const data = event
542
+ .split(/\r?\n/u)
543
+ .filter((line) => line.startsWith("data:"))
544
+ .map((line) => line.slice(5).trimStart())
545
+ .join("\n")
546
+ .trim();
547
+ if (!data || data === "[DONE]") {
548
+ return undefined;
549
+ }
550
+ try {
551
+ return JSON.parse(data);
552
+ }
553
+ catch {
554
+ throw new GeminiTtsResponseError("Gemini TTS returned malformed streaming data");
330
555
  }
331
- return payload;
332
556
  }
333
557
  function extractGeminiText(payload) {
334
558
  if (!isObject(payload) || !Array.isArray(payload.candidates)) {
@@ -350,18 +574,39 @@ function extractGeminiText(payload) {
350
574
  }
351
575
  return "";
352
576
  }
353
- function assertGeminiTtsDidNotReachOutputLimit(payload) {
354
- if (!isObject(payload) || !Array.isArray(payload.candidates)) {
355
- return;
577
+ function assertGeminiTtsCompleted(payload) {
578
+ const finishReason = getGeminiFinishReason(payload);
579
+ if (finishReason) {
580
+ assertGeminiTtsFinishReason(finishReason);
356
581
  }
357
- if (payload.candidates.some((candidate) => isObject(candidate) && candidate.finishReason === "MAX_TOKENS")) {
582
+ }
583
+ function assertGeminiTtsFinishReason(finishReason) {
584
+ if (finishReason === "MAX_TOKENS") {
358
585
  throw new GeminiTtsOutputLimitError();
359
586
  }
587
+ if (finishReason !== "STOP") {
588
+ throw new GeminiTtsResponseError(`Gemini TTS stopped with finish reason '${finishReason}'`);
589
+ }
360
590
  }
361
- function extractGeminiInlineAudioData(payload) {
591
+ function getGeminiFinishReason(payload) {
362
592
  if (!isObject(payload) || !Array.isArray(payload.candidates)) {
363
593
  return undefined;
364
594
  }
595
+ for (const candidate of payload.candidates) {
596
+ if (isObject(candidate) && typeof candidate.finishReason === "string") {
597
+ return candidate.finishReason;
598
+ }
599
+ }
600
+ return undefined;
601
+ }
602
+ function extractGeminiInlineAudioData(payload) {
603
+ return extractGeminiInlineAudioDataParts(payload)[0];
604
+ }
605
+ function extractGeminiInlineAudioDataParts(payload) {
606
+ if (!isObject(payload) || !Array.isArray(payload.candidates)) {
607
+ return [];
608
+ }
609
+ const audioData = [];
365
610
  for (const candidate of payload.candidates) {
366
611
  if (!isObject(candidate) ||
367
612
  !isObject(candidate.content) ||
@@ -376,11 +621,11 @@ function extractGeminiInlineAudioData(payload) {
376
621
  }
377
622
  const data = part.inlineData.data.trim();
378
623
  if (data) {
379
- return data;
624
+ audioData.push(data);
380
625
  }
381
626
  }
382
627
  }
383
- return undefined;
628
+ return audioData;
384
629
  }
385
630
  function buildSpeechRewritePrompt(sourceText) {
386
631
  return [
@@ -422,46 +667,86 @@ function exceedsUnicodeCharacterLimit(text, limit) {
422
667
  }
423
668
  return false;
424
669
  }
425
- function splitSpeechChunks(spokenText, deliveryMode) {
670
+ function splitSpeechSegments(spokenText) {
426
671
  const normalizedText = spokenText
427
672
  .trim()
428
673
  .split(/\n\s*\n/gu)
429
674
  .map((paragraph) => paragraph.replace(/\s+/gu, " ").trim())
430
675
  .filter(Boolean)
431
676
  .join("\n\n");
432
- const maxCharacters = deliveryMode === "progressive"
433
- ? PROGRESSIVE_SPEECH_CHUNK_CHARACTERS
434
- : COMPLETE_SPEECH_CHUNK_CHARACTERS;
435
- const chunks = [];
436
- let remaining = Array.from(normalizedText);
437
- while (remaining.length > maxCharacters) {
438
- const boundary = findSpeechChunkBoundary(remaining, maxCharacters);
439
- chunks.push(remaining.slice(0, boundary).join("").trim());
440
- remaining = Array.from(remaining.slice(boundary).join("").trim());
441
- }
442
- if (remaining.length > 0) {
443
- chunks.push(remaining.join(""));
444
- }
445
- return chunks.length > 0 ? chunks : [spokenText.trim()];
677
+ const characters = Array.from(normalizedText);
678
+ const cumulativeWeights = [0];
679
+ for (const character of characters) {
680
+ cumulativeWeights.push(cumulativeWeights.at(-1) + speechCharacterWeight(character));
681
+ }
682
+ const totalWeight = cumulativeWeights.at(-1);
683
+ const minimumSegmentCounts = minimumSpeechSegmentCounts(cumulativeWeights);
684
+ const segmentCount = minimumSegmentCounts[0];
685
+ if (segmentCount === 1) {
686
+ return [normalizedText];
687
+ }
688
+ const boundaries = [0];
689
+ for (let segment = 1; segment < segmentCount; segment += 1) {
690
+ const previousBoundary = boundaries.at(-1);
691
+ const previousWeight = cumulativeWeights[previousBoundary];
692
+ const remainingSegments = segmentCount - segment;
693
+ const idealWeight = (totalWeight * segment) / segmentCount;
694
+ const maximumBoundary = characters.length - remainingSegments;
695
+ const candidates = Array.from({ length: maximumBoundary - previousBoundary }, (_, offset) => previousBoundary + offset + 1).filter((index) => cumulativeWeights[index] - previousWeight <= MAX_SPEECH_SEGMENT_WEIGHT &&
696
+ minimumSegmentCounts[index] <= remainingSegments);
697
+ const naturalCandidates = candidates.filter((index) => isNaturalSpeechBoundary(characters, index));
698
+ const wordCandidates = candidates.filter((index) => /\s/u.test(characters[index] ?? ""));
699
+ boundaries.push(selectSpeechBoundary(naturalCandidates, wordCandidates.length > 0 ? wordCandidates : candidates, cumulativeWeights, idealWeight));
700
+ }
701
+ boundaries.push(characters.length);
702
+ return boundaries
703
+ .slice(1)
704
+ .map((boundary, index) => characters.slice(boundaries[index], boundary).join("").trim());
446
705
  }
447
- function findSpeechChunkBoundary(characters, maxCharacters) {
448
- const preferredMinimum = Math.floor(maxCharacters * 0.6);
449
- for (let index = maxCharacters; index >= preferredMinimum; index -= 1) {
450
- if (characters[index] === "\n" && characters[index + 1] === "\n") {
451
- return index;
452
- }
453
- }
454
- for (let index = maxCharacters; index >= preferredMinimum; index -= 1) {
455
- if (isSpeechSentenceBoundary(characters[index - 1] ?? "", characters[index] ?? "")) {
456
- return index;
706
+ function speechCharacterWeight(character) {
707
+ return /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u.test(character)
708
+ ? 3
709
+ : 1;
710
+ }
711
+ function minimumSpeechSegmentCounts(cumulativeWeights) {
712
+ const counts = new Array(cumulativeWeights.length).fill(0);
713
+ let boundary = cumulativeWeights.length - 1;
714
+ for (let index = cumulativeWeights.length - 2; index >= 0; index -= 1) {
715
+ while (cumulativeWeights[boundary] - cumulativeWeights[index] > MAX_SPEECH_SEGMENT_WEIGHT) {
716
+ boundary -= 1;
457
717
  }
718
+ counts[index] = counts[boundary] + 1;
458
719
  }
459
- for (let index = maxCharacters; index >= 1; index -= 1) {
460
- if (/\s/u.test(characters[index] ?? "")) {
461
- return index;
720
+ return counts;
721
+ }
722
+ function isNaturalSpeechBoundary(characters, index) {
723
+ const previous = characters[index - 1] ?? "";
724
+ const next = characters[index] ?? "";
725
+ return (previous === "\n" && next === "\n") || isSpeechSentenceBoundary(previous, next);
726
+ }
727
+ function selectSpeechBoundary(naturalCandidates, fallbackCandidates, cumulativeWeights, idealWeight) {
728
+ const fallback = closestSpeechBoundary(fallbackCandidates, cumulativeWeights, idealWeight);
729
+ if (naturalCandidates.length === 0) {
730
+ return fallback;
731
+ }
732
+ const natural = closestSpeechBoundary(naturalCandidates, cumulativeWeights, idealWeight);
733
+ const naturalDistance = Math.abs(cumulativeWeights[natural] - idealWeight);
734
+ const fallbackDistance = Math.abs(cumulativeWeights[fallback] - idealWeight);
735
+ return naturalDistance <= fallbackDistance + MAX_SPEECH_SEGMENT_WEIGHT * 0.05
736
+ ? natural
737
+ : fallback;
738
+ }
739
+ function closestSpeechBoundary(candidates, cumulativeWeights, idealWeight) {
740
+ let closest = candidates[0];
741
+ let closestDistance = Math.abs(cumulativeWeights[closest] - idealWeight);
742
+ for (const candidate of candidates.slice(1)) {
743
+ const distance = Math.abs(cumulativeWeights[candidate] - idealWeight);
744
+ if (distance < closestDistance) {
745
+ closest = candidate;
746
+ closestDistance = distance;
462
747
  }
463
748
  }
464
- return maxCharacters;
749
+ return closest;
465
750
  }
466
751
  function isSpeechSentenceBoundary(previous, next) {
467
752
  return /[。!?]/u.test(previous) || (/[.!?;:]/u.test(previous) && /\s/u.test(next));