@markusylisiurunen/tau 0.3.55 → 0.3.57
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/auth/credential_store.js +1 -9
- package/dist/core/auth/credential_store.js.map +1 -1
- package/dist/core/cli.js +3 -0
- package/dist/core/cli.js.map +1 -1
- package/dist/core/config/runtime.js +4 -4
- package/dist/core/config/runtime.js.map +1 -1
- package/dist/core/config/runtime_config_snapshot.js +1 -1
- package/dist/core/config/runtime_config_snapshot.js.map +1 -1
- package/dist/core/history/history_manager.js +35 -25
- package/dist/core/history/history_manager.js.map +1 -1
- package/dist/core/models/catalog.js +11 -45
- package/dist/core/models/catalog.js.map +1 -1
- package/dist/core/models/cli.js +49 -0
- package/dist/core/models/cli.js.map +1 -0
- package/dist/core/models/remote_catalog.js +394 -0
- package/dist/core/models/remote_catalog.js.map +1 -0
- package/dist/core/personas.js +3 -3
- package/dist/core/static/tau_docs/getting-started.md +5 -1
- package/dist/core/static/tau_docs/history.md +1 -1
- package/dist/core/static/tau_docs/models.md +12 -2
- package/dist/core/static/tau_docs/node-sdk.md +7 -1
- package/dist/core/static/tau_docs/security.md +5 -0
- package/dist/core/static/tau_docs/telegram.md +2 -2
- package/dist/core/static/tau_docs/tui.md +2 -2
- package/dist/core/telegram/adapter.js +2 -2
- package/dist/core/telegram/adapter.js.map +1 -1
- package/dist/core/telegram/local_session_client.js +2 -1
- package/dist/core/telegram/local_session_client.js.map +1 -1
- package/dist/core/telegram/tts.js +3 -4
- package/dist/core/telegram/tts.js.map +1 -1
- package/dist/core/utils/gemini_speech.js +435 -150
- package/dist/core/utils/gemini_speech.js.map +1 -1
- package/dist/core/utils/model_stream.js +0 -10
- package/dist/core/utils/model_stream.js.map +1 -1
- package/dist/core/utils/openai_transcription.js +164 -23
- package/dist/core/utils/openai_transcription.js.map +1 -1
- package/dist/core/utils/speech_to_text.js +1 -0
- package/dist/core/utils/speech_to_text.js.map +1 -1
- package/dist/core/version.js +2 -1
- package/dist/core/version.js.map +1 -1
- package/dist/execution/execution_environment.js.map +1 -1
- package/dist/execution/tool_backend_execution_environment.js +2 -1
- package/dist/execution/tool_backend_execution_environment.js.map +1 -1
- package/dist/history/worker/index.js +22 -7
- package/dist/history/worker/index.js.map +1 -1
- package/dist/host/execution_runtime.js +3 -1
- package/dist/host/execution_runtime.js.map +1 -1
- package/dist/host/local_session_host.js +37 -11
- package/dist/host/local_session_host.js.map +1 -1
- package/dist/main.js +74 -8
- package/dist/main.js.map +1 -1
- package/dist/sdk/index.d.ts +1 -1
- package/dist/sdk/index.js.map +1 -1
- package/dist/sdk/local_client.d.ts +4 -1
- package/dist/sdk/local_client.js +35 -9
- package/dist/sdk/local_client.js.map +1 -1
- package/dist/sdk/types.d.ts +14 -0
- package/dist/tui/session_chat_controller.js +3 -0
- package/dist/tui/session_chat_controller.js.map +1 -1
- package/dist/tui/speech_playback.js +145 -82
- package/dist/tui/speech_playback.js.map +1 -1
- package/package.json +4 -4
- package/dist/core/models/tau_extensions.js +0 -27
- package/dist/core/models/tau_extensions.js.map +0 -1
|
@@ -5,15 +5,15 @@ const DEFAULT_GEMINI_SPEECH_REWRITE_MODEL = "gemini-3.7-flash";
|
|
|
5
5
|
const DEFAULT_GEMINI_SPEECH_REWRITE_THINKING_LEVEL = "low";
|
|
6
6
|
const DEFAULT_GEMINI_SPEECH_TTS_MODEL = "gemini-3.1-flash-tts-preview";
|
|
7
7
|
const DEFAULT_GEMINI_TTS_VOICE_NAME = "Despina";
|
|
8
|
-
const
|
|
9
|
-
const
|
|
10
|
-
const
|
|
8
|
+
export const GEMINI_SPEECH_SAMPLE_RATE_HZ = 24000;
|
|
9
|
+
export const GEMINI_SPEECH_CHANNEL_COUNT = 1;
|
|
10
|
+
export const GEMINI_SPEECH_BITS_PER_SAMPLE = 16;
|
|
11
11
|
const DEFAULT_TTS_MAX_ATTEMPTS = 3;
|
|
12
12
|
const SPEECH_REWRITE_TIMEOUT_MS = 60_000;
|
|
13
|
-
const PROGRESSIVE_TTS_CONCURRENCY = 6;
|
|
14
13
|
const COMPLETE_TTS_CONCURRENCY = 3;
|
|
15
|
-
const
|
|
16
|
-
const
|
|
14
|
+
const MAX_SPEECH_SEGMENT_SECONDS = 120;
|
|
15
|
+
const ESTIMATED_SPEECH_CHARACTERS_PER_SECOND = 17;
|
|
16
|
+
const MAX_SPEECH_SEGMENT_WEIGHT = MAX_SPEECH_SEGMENT_SECONDS * ESTIMATED_SPEECH_CHARACTERS_PER_SECOND;
|
|
17
17
|
const MAX_SPEECH_SOURCE_CHARACTERS = 10_000;
|
|
18
18
|
const MAX_SPOKEN_TEXT_CHARACTERS = 10_000;
|
|
19
19
|
const MAX_SPEECH_PCM_BYTES = 32 * 1024 * 1024;
|
|
@@ -48,77 +48,26 @@ class GeminiTtsOutputLimitError extends Error {
|
|
|
48
48
|
this.name = "GeminiTtsOutputLimitError";
|
|
49
49
|
}
|
|
50
50
|
}
|
|
51
|
-
export async function*
|
|
52
|
-
const sourceText = options.sourceText.trim();
|
|
53
|
-
if (!sourceText) {
|
|
54
|
-
throw new Error("speech source text was empty");
|
|
55
|
-
}
|
|
56
|
-
if (exceedsUnicodeCharacterLimit(sourceText, MAX_SPEECH_SOURCE_CHARACTERS)) {
|
|
57
|
-
throw new Error("speech source text exceeds 10,000 characters");
|
|
58
|
-
}
|
|
59
|
-
const apiKey = options.apiKey.trim();
|
|
60
|
-
if (!apiKey) {
|
|
61
|
-
throw new Error("missing Gemini API key");
|
|
62
|
-
}
|
|
63
|
-
const fetchImpl = options.fetchImpl ?? fetch;
|
|
64
|
-
const rewriteModel = options.rewriteModel ?? DEFAULT_GEMINI_SPEECH_REWRITE_MODEL;
|
|
65
|
-
const ttsModel = options.ttsModel ?? DEFAULT_GEMINI_SPEECH_TTS_MODEL;
|
|
66
|
-
const voiceName = options.voiceName ?? DEFAULT_GEMINI_TTS_VOICE_NAME;
|
|
51
|
+
export async function* generateGeminiSpeechAudio(options) {
|
|
67
52
|
const abortController = createLinkedAbortController(options.signal);
|
|
68
53
|
let completed = false;
|
|
69
54
|
try {
|
|
70
|
-
await options.
|
|
71
|
-
const
|
|
72
|
-
|
|
73
|
-
rewriteTimeout.unref?.();
|
|
74
|
-
let spokenText;
|
|
75
|
-
try {
|
|
76
|
-
spokenText = await rewriteTextForSpeech({
|
|
77
|
-
apiKey,
|
|
78
|
-
model: rewriteModel,
|
|
79
|
-
sourceText,
|
|
80
|
-
fetchImpl,
|
|
81
|
-
signal: rewriteController.signal,
|
|
82
|
-
});
|
|
83
|
-
}
|
|
84
|
-
catch (error) {
|
|
85
|
-
if (!abortController.signal.aborted && rewriteController.signal.aborted) {
|
|
86
|
-
throw new Error("speech rewrite timed out after 1 minute");
|
|
87
|
-
}
|
|
88
|
-
throw error;
|
|
89
|
-
}
|
|
90
|
-
finally {
|
|
91
|
-
clearTimeout(rewriteTimeout);
|
|
92
|
-
rewriteController.dispose();
|
|
93
|
-
}
|
|
94
|
-
if (exceedsUnicodeCharacterLimit(spokenText, MAX_SPOKEN_TEXT_CHARACTERS)) {
|
|
95
|
-
throw new Error("rewritten speech text exceeds 10,000 characters");
|
|
96
|
-
}
|
|
97
|
-
const spokenChunks = splitSpeechChunks(spokenText, options.deliveryMode);
|
|
98
|
-
await options.onStageChange?.("generating");
|
|
99
|
-
await options.onChunkProgress?.({ ready: 0, total: spokenChunks.length });
|
|
100
|
-
for await (const chunk of synthesizeSpeechAudioChunksInOrder({
|
|
101
|
-
apiKey,
|
|
102
|
-
model: ttsModel,
|
|
103
|
-
voiceName,
|
|
104
|
-
spokenChunks,
|
|
105
|
-
fetchImpl,
|
|
55
|
+
const prepared = await prepareGeminiSpeech(options, abortController.signal);
|
|
56
|
+
for await (const chunk of synthesizeSpeechAudioSegmentsInOrder({
|
|
57
|
+
...prepared,
|
|
106
58
|
signal: abortController.signal,
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
? PROGRESSIVE_TTS_CONCURRENCY
|
|
110
|
-
: COMPLETE_TTS_CONCURRENCY,
|
|
111
|
-
onChunkProgress: options.onChunkProgress,
|
|
59
|
+
concurrency: COMPLETE_TTS_CONCURRENCY,
|
|
60
|
+
onSegmentProgress: options.onSegmentProgress,
|
|
112
61
|
abortOnFailure: () => abortController.abort(),
|
|
113
62
|
})) {
|
|
114
63
|
yield {
|
|
115
64
|
index: chunk.index,
|
|
116
|
-
total:
|
|
65
|
+
total: prepared.spokenSegments.length,
|
|
117
66
|
audio: encodeWaveFile({
|
|
118
67
|
pcmAudio: chunk.pcmAudio,
|
|
119
|
-
sampleRateHz:
|
|
120
|
-
channelCount:
|
|
121
|
-
bitsPerSample:
|
|
68
|
+
sampleRateHz: GEMINI_SPEECH_SAMPLE_RATE_HZ,
|
|
69
|
+
channelCount: GEMINI_SPEECH_CHANNEL_COUNT,
|
|
70
|
+
bitsPerSample: GEMINI_SPEECH_BITS_PER_SAMPLE,
|
|
122
71
|
}),
|
|
123
72
|
mimeType: "audio/wav",
|
|
124
73
|
};
|
|
@@ -132,6 +81,119 @@ export async function* streamGeminiSpeechAudio(options) {
|
|
|
132
81
|
abortController.dispose();
|
|
133
82
|
}
|
|
134
83
|
}
|
|
84
|
+
export async function* streamGeminiSpeechPcm(options) {
|
|
85
|
+
const abortController = createLinkedAbortController(options.signal);
|
|
86
|
+
let completed = false;
|
|
87
|
+
let totalPcmBytes = 0;
|
|
88
|
+
try {
|
|
89
|
+
const prepared = await prepareGeminiSpeech(options, abortController.signal);
|
|
90
|
+
const total = prepared.spokenSegments.length;
|
|
91
|
+
const accountAudio = (audio) => {
|
|
92
|
+
totalPcmBytes += audio.length;
|
|
93
|
+
if (totalPcmBytes > MAX_SPEECH_PCM_BYTES) {
|
|
94
|
+
throw new Error("generated speech audio exceeds the 32 MiB limit");
|
|
95
|
+
}
|
|
96
|
+
};
|
|
97
|
+
const prefetch = (index) => collectStreamingSpeechSegment({
|
|
98
|
+
...prepared,
|
|
99
|
+
spokenText: prepared.spokenSegments[index],
|
|
100
|
+
signal: abortController.signal,
|
|
101
|
+
accountAudio,
|
|
102
|
+
}).then((audio) => ({ audio }), (error) => ({ error }));
|
|
103
|
+
let ready = 0;
|
|
104
|
+
const firstStream = streamSpeechSegmentWithRetries({
|
|
105
|
+
...prepared,
|
|
106
|
+
spokenText: prepared.spokenSegments[0],
|
|
107
|
+
signal: abortController.signal,
|
|
108
|
+
initialBufferBytes: Math.max(0, Math.trunc(options.initialBufferBytes ?? 0)),
|
|
109
|
+
});
|
|
110
|
+
const firstIterator = firstStream[Symbol.asyncIterator]();
|
|
111
|
+
let nextAudio = firstIterator.next();
|
|
112
|
+
let prefetched = total > 1 ? prefetch(1) : undefined;
|
|
113
|
+
while (true) {
|
|
114
|
+
const next = await nextAudio;
|
|
115
|
+
if (next.done) {
|
|
116
|
+
break;
|
|
117
|
+
}
|
|
118
|
+
accountAudio(next.value);
|
|
119
|
+
yield { index: 0, total, audio: next.value };
|
|
120
|
+
nextAudio = firstIterator.next();
|
|
121
|
+
}
|
|
122
|
+
ready += 1;
|
|
123
|
+
await options.onSegmentProgress?.({ ready, total });
|
|
124
|
+
for (let index = 1; index < total; index += 1) {
|
|
125
|
+
const outcome = await prefetched;
|
|
126
|
+
if ("error" in outcome) {
|
|
127
|
+
throw outcome.error instanceof Error
|
|
128
|
+
? outcome.error
|
|
129
|
+
: new Error("Gemini TTS request failed");
|
|
130
|
+
}
|
|
131
|
+
prefetched = index + 1 < total ? prefetch(index + 1) : undefined;
|
|
132
|
+
yield { index, total, audio: outcome.audio };
|
|
133
|
+
ready += 1;
|
|
134
|
+
await options.onSegmentProgress?.({ ready, total });
|
|
135
|
+
}
|
|
136
|
+
completed = true;
|
|
137
|
+
}
|
|
138
|
+
finally {
|
|
139
|
+
if (!completed) {
|
|
140
|
+
abortController.abort();
|
|
141
|
+
}
|
|
142
|
+
abortController.dispose();
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
async function prepareGeminiSpeech(options, signal) {
|
|
146
|
+
const sourceText = options.sourceText.trim();
|
|
147
|
+
if (!sourceText) {
|
|
148
|
+
throw new Error("speech source text was empty");
|
|
149
|
+
}
|
|
150
|
+
if (exceedsUnicodeCharacterLimit(sourceText, MAX_SPEECH_SOURCE_CHARACTERS)) {
|
|
151
|
+
throw new Error("speech source text exceeds 10,000 characters");
|
|
152
|
+
}
|
|
153
|
+
const apiKey = options.apiKey.trim();
|
|
154
|
+
if (!apiKey) {
|
|
155
|
+
throw new Error("missing Gemini API key");
|
|
156
|
+
}
|
|
157
|
+
const fetchImpl = options.fetchImpl ?? fetch;
|
|
158
|
+
await options.onStageChange?.("rewriting");
|
|
159
|
+
const rewriteController = createLinkedAbortController(signal);
|
|
160
|
+
const rewriteTimeout = setTimeout(() => rewriteController.abort(), SPEECH_REWRITE_TIMEOUT_MS);
|
|
161
|
+
rewriteTimeout.unref?.();
|
|
162
|
+
let spokenText;
|
|
163
|
+
try {
|
|
164
|
+
spokenText = await rewriteTextForSpeech({
|
|
165
|
+
apiKey,
|
|
166
|
+
model: options.rewriteModel ?? DEFAULT_GEMINI_SPEECH_REWRITE_MODEL,
|
|
167
|
+
sourceText,
|
|
168
|
+
fetchImpl,
|
|
169
|
+
signal: rewriteController.signal,
|
|
170
|
+
});
|
|
171
|
+
}
|
|
172
|
+
catch (error) {
|
|
173
|
+
if (!signal.aborted && rewriteController.signal.aborted) {
|
|
174
|
+
throw new Error("speech rewrite timed out after 1 minute");
|
|
175
|
+
}
|
|
176
|
+
throw error;
|
|
177
|
+
}
|
|
178
|
+
finally {
|
|
179
|
+
clearTimeout(rewriteTimeout);
|
|
180
|
+
rewriteController.dispose();
|
|
181
|
+
}
|
|
182
|
+
if (exceedsUnicodeCharacterLimit(spokenText, MAX_SPOKEN_TEXT_CHARACTERS)) {
|
|
183
|
+
throw new Error("rewritten speech text exceeds 10,000 characters");
|
|
184
|
+
}
|
|
185
|
+
const spokenSegments = splitSpeechSegments(spokenText);
|
|
186
|
+
await options.onStageChange?.("generating");
|
|
187
|
+
await options.onSegmentProgress?.({ ready: 0, total: spokenSegments.length });
|
|
188
|
+
return {
|
|
189
|
+
apiKey,
|
|
190
|
+
model: options.ttsModel ?? DEFAULT_GEMINI_SPEECH_TTS_MODEL,
|
|
191
|
+
voiceName: options.voiceName ?? DEFAULT_GEMINI_TTS_VOICE_NAME,
|
|
192
|
+
spokenSegments,
|
|
193
|
+
fetchImpl,
|
|
194
|
+
maxAttempts: options.maxTtsAttempts ?? DEFAULT_TTS_MAX_ATTEMPTS,
|
|
195
|
+
};
|
|
196
|
+
}
|
|
135
197
|
async function rewriteTextForSpeech(args) {
|
|
136
198
|
const payload = await requestGeminiGenerateContent({
|
|
137
199
|
apiKey: args.apiKey,
|
|
@@ -161,8 +223,8 @@ async function rewriteTextForSpeech(args) {
|
|
|
161
223
|
}
|
|
162
224
|
return rewrittenText;
|
|
163
225
|
}
|
|
164
|
-
async function*
|
|
165
|
-
const total = args.
|
|
226
|
+
async function* synthesizeSpeechAudioSegmentsInOrder(args) {
|
|
227
|
+
const total = args.spokenSegments.length;
|
|
166
228
|
const results = new Array(total);
|
|
167
229
|
const concurrency = Math.max(1, Math.trunc(args.concurrency));
|
|
168
230
|
let nextIndex = 0;
|
|
@@ -202,14 +264,9 @@ async function* synthesizeSpeechAudioChunksInOrder(args) {
|
|
|
202
264
|
return;
|
|
203
265
|
}
|
|
204
266
|
try {
|
|
205
|
-
const pcmAudio = await
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
voiceName: args.voiceName,
|
|
209
|
-
spokenText: args.spokenChunks[index],
|
|
210
|
-
fetchImpl: args.fetchImpl,
|
|
211
|
-
signal: args.signal,
|
|
212
|
-
maxAttempts: args.maxAttempts,
|
|
267
|
+
const pcmAudio = await synthesizeSpeechAudioSegment({
|
|
268
|
+
...args,
|
|
269
|
+
spokenText: args.spokenSegments[index],
|
|
213
270
|
});
|
|
214
271
|
totalPcmBytes += pcmAudio.length;
|
|
215
272
|
if (totalPcmBytes > MAX_SPEECH_PCM_BYTES) {
|
|
@@ -217,12 +274,12 @@ async function* synthesizeSpeechAudioChunksInOrder(args) {
|
|
|
217
274
|
}
|
|
218
275
|
results[index] = pcmAudio;
|
|
219
276
|
ready += 1;
|
|
220
|
-
await args.
|
|
277
|
+
await args.onSegmentProgress?.({ ready, total });
|
|
221
278
|
wake();
|
|
222
279
|
}
|
|
223
|
-
catch (
|
|
280
|
+
catch (error) {
|
|
224
281
|
if (failure === undefined) {
|
|
225
|
-
failure =
|
|
282
|
+
failure = error;
|
|
226
283
|
}
|
|
227
284
|
args.abortOnFailure?.();
|
|
228
285
|
wake();
|
|
@@ -252,40 +309,19 @@ async function* synthesizeSpeechAudioChunksInOrder(args) {
|
|
|
252
309
|
args.signal?.removeEventListener("abort", onAbort);
|
|
253
310
|
}
|
|
254
311
|
}
|
|
255
|
-
async function
|
|
312
|
+
async function synthesizeSpeechAudioSegment(args) {
|
|
256
313
|
const maxAttempts = Math.max(1, Math.trunc(args.maxAttempts));
|
|
257
314
|
let lastError;
|
|
258
|
-
for (let attempt = 1; attempt <= maxAttempts; attempt
|
|
315
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
259
316
|
try {
|
|
260
317
|
const payload = await requestGeminiGenerateContent({
|
|
261
318
|
apiKey: args.apiKey,
|
|
262
319
|
model: args.model,
|
|
263
320
|
fetchImpl: args.fetchImpl,
|
|
264
321
|
signal: args.signal,
|
|
265
|
-
body:
|
|
266
|
-
contents: [
|
|
267
|
-
{
|
|
268
|
-
parts: [
|
|
269
|
-
{
|
|
270
|
-
text: buildSpeechSynthesisPrompt(args.spokenText),
|
|
271
|
-
},
|
|
272
|
-
],
|
|
273
|
-
},
|
|
274
|
-
],
|
|
275
|
-
generationConfig: {
|
|
276
|
-
responseModalities: ["AUDIO"],
|
|
277
|
-
maxOutputTokens: TTS_MAX_OUTPUT_TOKENS,
|
|
278
|
-
speechConfig: {
|
|
279
|
-
voiceConfig: {
|
|
280
|
-
prebuiltVoiceConfig: {
|
|
281
|
-
voiceName: args.voiceName,
|
|
282
|
-
},
|
|
283
|
-
},
|
|
284
|
-
},
|
|
285
|
-
},
|
|
286
|
-
},
|
|
322
|
+
body: buildSpeechSynthesisRequest(args.spokenText, args.voiceName),
|
|
287
323
|
});
|
|
288
|
-
|
|
324
|
+
assertGeminiTtsCompleted(payload);
|
|
289
325
|
const audioData = extractGeminiInlineAudioData(payload);
|
|
290
326
|
if (!audioData) {
|
|
291
327
|
throw new GeminiTtsResponseError("Gemini TTS response did not include audio data");
|
|
@@ -302,8 +338,139 @@ async function synthesizeSpeechAudioChunk(args) {
|
|
|
302
338
|
}
|
|
303
339
|
throw lastError instanceof Error ? lastError : new Error("Gemini TTS request failed");
|
|
304
340
|
}
|
|
341
|
+
async function* streamSpeechSegmentWithRetries(args) {
|
|
342
|
+
const maxAttempts = Math.max(1, Math.trunc(args.maxAttempts));
|
|
343
|
+
let lastError;
|
|
344
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
345
|
+
const bufferedAudio = [];
|
|
346
|
+
let bufferedAudioBytes = 0;
|
|
347
|
+
let emittedAudio = false;
|
|
348
|
+
try {
|
|
349
|
+
for await (const audio of requestGeminiSpeechStream(args)) {
|
|
350
|
+
if (!emittedAudio && bufferedAudioBytes < args.initialBufferBytes) {
|
|
351
|
+
bufferedAudio.push(audio);
|
|
352
|
+
bufferedAudioBytes += audio.length;
|
|
353
|
+
if (bufferedAudioBytes < args.initialBufferBytes) {
|
|
354
|
+
continue;
|
|
355
|
+
}
|
|
356
|
+
emittedAudio = true;
|
|
357
|
+
yield Buffer.concat(bufferedAudio);
|
|
358
|
+
continue;
|
|
359
|
+
}
|
|
360
|
+
emittedAudio = true;
|
|
361
|
+
yield audio;
|
|
362
|
+
}
|
|
363
|
+
if (bufferedAudio.length > 0 && !emittedAudio) {
|
|
364
|
+
emittedAudio = true;
|
|
365
|
+
yield Buffer.concat(bufferedAudio);
|
|
366
|
+
}
|
|
367
|
+
return;
|
|
368
|
+
}
|
|
369
|
+
catch (error) {
|
|
370
|
+
lastError = error;
|
|
371
|
+
if (emittedAudio ||
|
|
372
|
+
args.signal?.aborted ||
|
|
373
|
+
!isRetryableTtsError(error) ||
|
|
374
|
+
attempt >= maxAttempts) {
|
|
375
|
+
throw error;
|
|
376
|
+
}
|
|
377
|
+
await waitForRetryDelay(attempt, args.signal);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
throw lastError instanceof Error ? lastError : new Error("Gemini TTS request failed");
|
|
381
|
+
}
|
|
382
|
+
async function collectStreamingSpeechSegment(args) {
|
|
383
|
+
const maxAttempts = Math.max(1, Math.trunc(args.maxAttempts));
|
|
384
|
+
let lastError;
|
|
385
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt += 1) {
|
|
386
|
+
const chunks = [];
|
|
387
|
+
try {
|
|
388
|
+
for await (const audio of requestGeminiSpeechStream(args)) {
|
|
389
|
+
chunks.push(audio);
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
catch (error) {
|
|
393
|
+
lastError = error;
|
|
394
|
+
if (args.signal?.aborted || !isRetryableTtsError(error) || attempt >= maxAttempts) {
|
|
395
|
+
throw error;
|
|
396
|
+
}
|
|
397
|
+
await waitForRetryDelay(attempt, args.signal);
|
|
398
|
+
continue;
|
|
399
|
+
}
|
|
400
|
+
const audio = Buffer.concat(chunks);
|
|
401
|
+
args.accountAudio(audio);
|
|
402
|
+
return audio;
|
|
403
|
+
}
|
|
404
|
+
throw lastError instanceof Error ? lastError : new Error("Gemini TTS request failed");
|
|
405
|
+
}
|
|
406
|
+
async function* requestGeminiSpeechStream(args) {
|
|
407
|
+
const response = await requestGeminiResponse({
|
|
408
|
+
apiKey: args.apiKey,
|
|
409
|
+
model: args.model,
|
|
410
|
+
method: "streamGenerateContent?alt=sse",
|
|
411
|
+
fetchImpl: args.fetchImpl,
|
|
412
|
+
signal: args.signal,
|
|
413
|
+
body: buildSpeechSynthesisRequest(args.spokenText, args.voiceName),
|
|
414
|
+
});
|
|
415
|
+
if (!response.body) {
|
|
416
|
+
throw new GeminiTtsResponseError("Gemini TTS streaming response did not include a body");
|
|
417
|
+
}
|
|
418
|
+
let receivedAudio = false;
|
|
419
|
+
let completed = false;
|
|
420
|
+
for await (const payload of parseGeminiSse(response.body)) {
|
|
421
|
+
const finishReason = getGeminiFinishReason(payload);
|
|
422
|
+
if (finishReason) {
|
|
423
|
+
assertGeminiTtsFinishReason(finishReason);
|
|
424
|
+
completed = finishReason === "STOP";
|
|
425
|
+
}
|
|
426
|
+
for (const audioData of extractGeminiInlineAudioDataParts(payload)) {
|
|
427
|
+
receivedAudio = true;
|
|
428
|
+
yield Buffer.from(audioData, "base64");
|
|
429
|
+
}
|
|
430
|
+
}
|
|
431
|
+
if (!receivedAudio) {
|
|
432
|
+
throw new GeminiTtsResponseError("Gemini TTS response did not include audio data");
|
|
433
|
+
}
|
|
434
|
+
if (!completed) {
|
|
435
|
+
throw new GeminiTtsResponseError("Gemini TTS stream ended without a stop response");
|
|
436
|
+
}
|
|
437
|
+
}
|
|
438
|
+
function buildSpeechSynthesisRequest(spokenText, voiceName) {
|
|
439
|
+
return {
|
|
440
|
+
contents: [
|
|
441
|
+
{
|
|
442
|
+
parts: [
|
|
443
|
+
{
|
|
444
|
+
text: buildSpeechSynthesisPrompt(spokenText),
|
|
445
|
+
},
|
|
446
|
+
],
|
|
447
|
+
},
|
|
448
|
+
],
|
|
449
|
+
generationConfig: {
|
|
450
|
+
responseModalities: ["AUDIO"],
|
|
451
|
+
maxOutputTokens: TTS_MAX_OUTPUT_TOKENS,
|
|
452
|
+
speechConfig: {
|
|
453
|
+
voiceConfig: {
|
|
454
|
+
prebuiltVoiceConfig: {
|
|
455
|
+
voiceName,
|
|
456
|
+
},
|
|
457
|
+
},
|
|
458
|
+
},
|
|
459
|
+
},
|
|
460
|
+
};
|
|
461
|
+
}
|
|
305
462
|
async function requestGeminiGenerateContent(args) {
|
|
306
|
-
const response = await
|
|
463
|
+
const response = await requestGeminiResponse({ ...args, method: "generateContent" });
|
|
464
|
+
const responseText = await response.text();
|
|
465
|
+
try {
|
|
466
|
+
return responseText ? JSON.parse(responseText) : undefined;
|
|
467
|
+
}
|
|
468
|
+
catch {
|
|
469
|
+
return undefined;
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
async function requestGeminiResponse(args) {
|
|
473
|
+
const response = await args.fetchImpl(`${GEMINI_GENERATE_CONTENT_BASE_URL}/${encodeURIComponent(args.model)}:${args.method}`, {
|
|
307
474
|
method: "POST",
|
|
308
475
|
headers: {
|
|
309
476
|
"Content-Type": "application/json",
|
|
@@ -312,6 +479,9 @@ async function requestGeminiGenerateContent(args) {
|
|
|
312
479
|
body: JSON.stringify(args.body),
|
|
313
480
|
signal: args.signal,
|
|
314
481
|
});
|
|
482
|
+
if (response.ok) {
|
|
483
|
+
return response;
|
|
484
|
+
}
|
|
315
485
|
const responseText = await response.text();
|
|
316
486
|
let payload;
|
|
317
487
|
try {
|
|
@@ -320,15 +490,69 @@ async function requestGeminiGenerateContent(args) {
|
|
|
320
490
|
catch {
|
|
321
491
|
payload = undefined;
|
|
322
492
|
}
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
493
|
+
const parsed = errorPayloadSchema.safeParse(payload);
|
|
494
|
+
const fallbackMessage = responseText.trim() || `HTTP ${response.status}`;
|
|
495
|
+
const message = parsed.success
|
|
496
|
+
? (parsed.data.error?.message ?? fallbackMessage)
|
|
497
|
+
: fallbackMessage;
|
|
498
|
+
throw new GeminiApiError(message, response.status);
|
|
499
|
+
}
|
|
500
|
+
async function* parseGeminiSse(body) {
|
|
501
|
+
const reader = body.getReader();
|
|
502
|
+
const decoder = new TextDecoder();
|
|
503
|
+
let buffered = "";
|
|
504
|
+
let completed = false;
|
|
505
|
+
try {
|
|
506
|
+
while (true) {
|
|
507
|
+
const { done, value } = await reader.read();
|
|
508
|
+
buffered += decoder.decode(value, { stream: !done });
|
|
509
|
+
let boundary = findSseEventBoundary(buffered);
|
|
510
|
+
while (boundary) {
|
|
511
|
+
const event = buffered.slice(0, boundary.index);
|
|
512
|
+
buffered = buffered.slice(boundary.index + boundary.length);
|
|
513
|
+
const payload = parseSseEvent(event);
|
|
514
|
+
if (payload !== undefined) {
|
|
515
|
+
yield payload;
|
|
516
|
+
}
|
|
517
|
+
boundary = findSseEventBoundary(buffered);
|
|
518
|
+
}
|
|
519
|
+
if (done) {
|
|
520
|
+
const payload = parseSseEvent(buffered);
|
|
521
|
+
if (payload !== undefined) {
|
|
522
|
+
yield payload;
|
|
523
|
+
}
|
|
524
|
+
completed = true;
|
|
525
|
+
return;
|
|
526
|
+
}
|
|
527
|
+
}
|
|
528
|
+
}
|
|
529
|
+
finally {
|
|
530
|
+
if (!completed) {
|
|
531
|
+
await reader.cancel().catch(() => { });
|
|
532
|
+
}
|
|
533
|
+
reader.releaseLock();
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
function findSseEventBoundary(text) {
|
|
537
|
+
const match = /\r?\n\r?\n/u.exec(text);
|
|
538
|
+
return match ? { index: match.index, length: match[0].length } : undefined;
|
|
539
|
+
}
|
|
540
|
+
function parseSseEvent(event) {
|
|
541
|
+
const data = event
|
|
542
|
+
.split(/\r?\n/u)
|
|
543
|
+
.filter((line) => line.startsWith("data:"))
|
|
544
|
+
.map((line) => line.slice(5).trimStart())
|
|
545
|
+
.join("\n")
|
|
546
|
+
.trim();
|
|
547
|
+
if (!data || data === "[DONE]") {
|
|
548
|
+
return undefined;
|
|
549
|
+
}
|
|
550
|
+
try {
|
|
551
|
+
return JSON.parse(data);
|
|
552
|
+
}
|
|
553
|
+
catch {
|
|
554
|
+
throw new GeminiTtsResponseError("Gemini TTS returned malformed streaming data");
|
|
330
555
|
}
|
|
331
|
-
return payload;
|
|
332
556
|
}
|
|
333
557
|
function extractGeminiText(payload) {
|
|
334
558
|
if (!isObject(payload) || !Array.isArray(payload.candidates)) {
|
|
@@ -350,18 +574,39 @@ function extractGeminiText(payload) {
|
|
|
350
574
|
}
|
|
351
575
|
return "";
|
|
352
576
|
}
|
|
353
|
-
function
|
|
354
|
-
|
|
355
|
-
|
|
577
|
+
function assertGeminiTtsCompleted(payload) {
|
|
578
|
+
const finishReason = getGeminiFinishReason(payload);
|
|
579
|
+
if (finishReason) {
|
|
580
|
+
assertGeminiTtsFinishReason(finishReason);
|
|
356
581
|
}
|
|
357
|
-
|
|
582
|
+
}
|
|
583
|
+
function assertGeminiTtsFinishReason(finishReason) {
|
|
584
|
+
if (finishReason === "MAX_TOKENS") {
|
|
358
585
|
throw new GeminiTtsOutputLimitError();
|
|
359
586
|
}
|
|
587
|
+
if (finishReason !== "STOP") {
|
|
588
|
+
throw new GeminiTtsResponseError(`Gemini TTS stopped with finish reason '${finishReason}'`);
|
|
589
|
+
}
|
|
360
590
|
}
|
|
361
|
-
function
|
|
591
|
+
function getGeminiFinishReason(payload) {
|
|
362
592
|
if (!isObject(payload) || !Array.isArray(payload.candidates)) {
|
|
363
593
|
return undefined;
|
|
364
594
|
}
|
|
595
|
+
for (const candidate of payload.candidates) {
|
|
596
|
+
if (isObject(candidate) && typeof candidate.finishReason === "string") {
|
|
597
|
+
return candidate.finishReason;
|
|
598
|
+
}
|
|
599
|
+
}
|
|
600
|
+
return undefined;
|
|
601
|
+
}
|
|
602
|
+
function extractGeminiInlineAudioData(payload) {
|
|
603
|
+
return extractGeminiInlineAudioDataParts(payload)[0];
|
|
604
|
+
}
|
|
605
|
+
function extractGeminiInlineAudioDataParts(payload) {
|
|
606
|
+
if (!isObject(payload) || !Array.isArray(payload.candidates)) {
|
|
607
|
+
return [];
|
|
608
|
+
}
|
|
609
|
+
const audioData = [];
|
|
365
610
|
for (const candidate of payload.candidates) {
|
|
366
611
|
if (!isObject(candidate) ||
|
|
367
612
|
!isObject(candidate.content) ||
|
|
@@ -376,11 +621,11 @@ function extractGeminiInlineAudioData(payload) {
|
|
|
376
621
|
}
|
|
377
622
|
const data = part.inlineData.data.trim();
|
|
378
623
|
if (data) {
|
|
379
|
-
|
|
624
|
+
audioData.push(data);
|
|
380
625
|
}
|
|
381
626
|
}
|
|
382
627
|
}
|
|
383
|
-
return
|
|
628
|
+
return audioData;
|
|
384
629
|
}
|
|
385
630
|
function buildSpeechRewritePrompt(sourceText) {
|
|
386
631
|
return [
|
|
@@ -422,46 +667,86 @@ function exceedsUnicodeCharacterLimit(text, limit) {
|
|
|
422
667
|
}
|
|
423
668
|
return false;
|
|
424
669
|
}
|
|
425
|
-
function
|
|
670
|
+
function splitSpeechSegments(spokenText) {
|
|
426
671
|
const normalizedText = spokenText
|
|
427
672
|
.trim()
|
|
428
673
|
.split(/\n\s*\n/gu)
|
|
429
674
|
.map((paragraph) => paragraph.replace(/\s+/gu, " ").trim())
|
|
430
675
|
.filter(Boolean)
|
|
431
676
|
.join("\n\n");
|
|
432
|
-
const
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
677
|
+
const characters = Array.from(normalizedText);
|
|
678
|
+
const cumulativeWeights = [0];
|
|
679
|
+
for (const character of characters) {
|
|
680
|
+
cumulativeWeights.push(cumulativeWeights.at(-1) + speechCharacterWeight(character));
|
|
681
|
+
}
|
|
682
|
+
const totalWeight = cumulativeWeights.at(-1);
|
|
683
|
+
const minimumSegmentCounts = minimumSpeechSegmentCounts(cumulativeWeights);
|
|
684
|
+
const segmentCount = minimumSegmentCounts[0];
|
|
685
|
+
if (segmentCount === 1) {
|
|
686
|
+
return [normalizedText];
|
|
687
|
+
}
|
|
688
|
+
const boundaries = [0];
|
|
689
|
+
for (let segment = 1; segment < segmentCount; segment += 1) {
|
|
690
|
+
const previousBoundary = boundaries.at(-1);
|
|
691
|
+
const previousWeight = cumulativeWeights[previousBoundary];
|
|
692
|
+
const remainingSegments = segmentCount - segment;
|
|
693
|
+
const idealWeight = (totalWeight * segment) / segmentCount;
|
|
694
|
+
const maximumBoundary = characters.length - remainingSegments;
|
|
695
|
+
const candidates = Array.from({ length: maximumBoundary - previousBoundary }, (_, offset) => previousBoundary + offset + 1).filter((index) => cumulativeWeights[index] - previousWeight <= MAX_SPEECH_SEGMENT_WEIGHT &&
|
|
696
|
+
minimumSegmentCounts[index] <= remainingSegments);
|
|
697
|
+
const naturalCandidates = candidates.filter((index) => isNaturalSpeechBoundary(characters, index));
|
|
698
|
+
const wordCandidates = candidates.filter((index) => /\s/u.test(characters[index] ?? ""));
|
|
699
|
+
boundaries.push(selectSpeechBoundary(naturalCandidates, wordCandidates.length > 0 ? wordCandidates : candidates, cumulativeWeights, idealWeight));
|
|
700
|
+
}
|
|
701
|
+
boundaries.push(characters.length);
|
|
702
|
+
return boundaries
|
|
703
|
+
.slice(1)
|
|
704
|
+
.map((boundary, index) => characters.slice(boundaries[index], boundary).join("").trim());
|
|
446
705
|
}
|
|
447
|
-
function
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
706
|
+
function speechCharacterWeight(character) {
|
|
707
|
+
return /[\p{Script=Han}\p{Script=Hiragana}\p{Script=Katakana}\p{Script=Hangul}]/u.test(character)
|
|
708
|
+
? 3
|
|
709
|
+
: 1;
|
|
710
|
+
}
|
|
711
|
+
function minimumSpeechSegmentCounts(cumulativeWeights) {
|
|
712
|
+
const counts = new Array(cumulativeWeights.length).fill(0);
|
|
713
|
+
let boundary = cumulativeWeights.length - 1;
|
|
714
|
+
for (let index = cumulativeWeights.length - 2; index >= 0; index -= 1) {
|
|
715
|
+
while (cumulativeWeights[boundary] - cumulativeWeights[index] > MAX_SPEECH_SEGMENT_WEIGHT) {
|
|
716
|
+
boundary -= 1;
|
|
457
717
|
}
|
|
718
|
+
counts[index] = counts[boundary] + 1;
|
|
458
719
|
}
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
720
|
+
return counts;
|
|
721
|
+
}
|
|
722
|
+
function isNaturalSpeechBoundary(characters, index) {
|
|
723
|
+
const previous = characters[index - 1] ?? "";
|
|
724
|
+
const next = characters[index] ?? "";
|
|
725
|
+
return (previous === "\n" && next === "\n") || isSpeechSentenceBoundary(previous, next);
|
|
726
|
+
}
|
|
727
|
+
function selectSpeechBoundary(naturalCandidates, fallbackCandidates, cumulativeWeights, idealWeight) {
|
|
728
|
+
const fallback = closestSpeechBoundary(fallbackCandidates, cumulativeWeights, idealWeight);
|
|
729
|
+
if (naturalCandidates.length === 0) {
|
|
730
|
+
return fallback;
|
|
731
|
+
}
|
|
732
|
+
const natural = closestSpeechBoundary(naturalCandidates, cumulativeWeights, idealWeight);
|
|
733
|
+
const naturalDistance = Math.abs(cumulativeWeights[natural] - idealWeight);
|
|
734
|
+
const fallbackDistance = Math.abs(cumulativeWeights[fallback] - idealWeight);
|
|
735
|
+
return naturalDistance <= fallbackDistance + MAX_SPEECH_SEGMENT_WEIGHT * 0.05
|
|
736
|
+
? natural
|
|
737
|
+
: fallback;
|
|
738
|
+
}
|
|
739
|
+
function closestSpeechBoundary(candidates, cumulativeWeights, idealWeight) {
|
|
740
|
+
let closest = candidates[0];
|
|
741
|
+
let closestDistance = Math.abs(cumulativeWeights[closest] - idealWeight);
|
|
742
|
+
for (const candidate of candidates.slice(1)) {
|
|
743
|
+
const distance = Math.abs(cumulativeWeights[candidate] - idealWeight);
|
|
744
|
+
if (distance < closestDistance) {
|
|
745
|
+
closest = candidate;
|
|
746
|
+
closestDistance = distance;
|
|
462
747
|
}
|
|
463
748
|
}
|
|
464
|
-
return
|
|
749
|
+
return closest;
|
|
465
750
|
}
|
|
466
751
|
function isSpeechSentenceBoundary(previous, next) {
|
|
467
752
|
return /[。!?]/u.test(previous) || (/[.!?;:]/u.test(previous) && /\s/u.test(next));
|