@framers/agentos-ext-voice-synthesis 2.0.1 → 2.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +96 -21
- package/README.md +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/tools/speechToText.d.ts.map +1 -1
- package/dist/tools/speechToText.js +1 -1
- package/dist/tools/speechToText.js.map +1 -1
- package/dist/tools/textToSpeech.d.ts.map +1 -1
- package/dist/tools/textToSpeech.js +1 -0
- package/dist/tools/textToSpeech.js.map +1 -1
- package/manifest.json +3 -3
- package/package.json +17 -9
- package/src/index.ts +0 -92
- package/src/tools/speechToText.ts +0 -650
- package/src/tools/textToSpeech.ts +0 -287
- package/test/textToSpeech.spec.ts +0 -398
- package/tsconfig.json +0 -22
- package/vitest.config.ts +0 -10
|
@@ -1,650 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Multi-provider STT Tool — speech-to-text transcription.
|
|
3
|
-
*
|
|
4
|
-
* Supports: OpenAI Whisper, Deepgram, and Whisper-local/OpenAI-compatible
|
|
5
|
-
* local runtimes behind one stable tool contract.
|
|
6
|
-
*/
|
|
7
|
-
|
|
8
|
-
import type {
|
|
9
|
-
ITool,
|
|
10
|
-
JSONSchemaObject,
|
|
11
|
-
ToolExecutionContext,
|
|
12
|
-
ToolExecutionResult,
|
|
13
|
-
} from '@framers/agentos';
|
|
14
|
-
|
|
15
|
-
export type STTProvider = 'openai' | 'deepgram' | 'whisper-local' | 'auto';
|
|
16
|
-
export type STTResponseFormat = 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt';
|
|
17
|
-
|
|
18
|
-
export interface STTInput {
|
|
19
|
-
audioBase64?: string;
|
|
20
|
-
audioUrl?: string;
|
|
21
|
-
mimeType?: string;
|
|
22
|
-
fileName?: string;
|
|
23
|
-
format?: string;
|
|
24
|
-
language?: string;
|
|
25
|
-
prompt?: string;
|
|
26
|
-
temperature?: number;
|
|
27
|
-
responseFormat?: STTResponseFormat;
|
|
28
|
-
provider?: STTProvider;
|
|
29
|
-
model?: string;
|
|
30
|
-
diarize?: boolean;
|
|
31
|
-
utterances?: boolean;
|
|
32
|
-
smartFormat?: boolean;
|
|
33
|
-
detectLanguage?: boolean;
|
|
34
|
-
providerOptions?: Record<string, unknown>;
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
export interface STTWord {
|
|
38
|
-
word: string;
|
|
39
|
-
start: number;
|
|
40
|
-
end: number;
|
|
41
|
-
confidence?: number;
|
|
42
|
-
speaker?: string | number;
|
|
43
|
-
}
|
|
44
|
-
|
|
45
|
-
export interface STTSegment {
|
|
46
|
-
text: string;
|
|
47
|
-
start: number;
|
|
48
|
-
end: number;
|
|
49
|
-
confidence?: number;
|
|
50
|
-
speaker?: string | number;
|
|
51
|
-
words?: STTWord[];
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
export interface STTOutput {
|
|
55
|
-
text: string;
|
|
56
|
-
provider: Exclude<STTProvider, 'auto'>;
|
|
57
|
-
model: string;
|
|
58
|
-
language?: string;
|
|
59
|
-
confidence?: number;
|
|
60
|
-
durationSeconds?: number;
|
|
61
|
-
segments?: STTSegment[];
|
|
62
|
-
providerResponse?: unknown;
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
export interface STTConfig {
|
|
66
|
-
openaiApiKey?: string;
|
|
67
|
-
openaiBaseUrl?: string;
|
|
68
|
-
deepgramApiKey?: string;
|
|
69
|
-
deepgramBaseUrl?: string;
|
|
70
|
-
whisperLocalBaseUrl?: string;
|
|
71
|
-
defaultProvider?: STTProvider;
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
interface PreparedAudio {
|
|
75
|
-
data: Buffer;
|
|
76
|
-
mimeType: string;
|
|
77
|
-
format: string;
|
|
78
|
-
fileName: string;
|
|
79
|
-
}
|
|
80
|
-
|
|
81
|
-
interface SttBackend {
|
|
82
|
-
readonly provider: Exclude<STTProvider, 'auto'>;
|
|
83
|
-
isConfigured(config: ResolvedSttConfig): boolean;
|
|
84
|
-
transcribe(
|
|
85
|
-
audio: PreparedAudio,
|
|
86
|
-
input: STTInput,
|
|
87
|
-
config: ResolvedSttConfig,
|
|
88
|
-
): Promise<STTOutput>;
|
|
89
|
-
}
|
|
90
|
-
|
|
91
|
-
interface ResolvedSttConfig {
|
|
92
|
-
openaiApiKey?: string;
|
|
93
|
-
openaiBaseUrl: string;
|
|
94
|
-
deepgramApiKey?: string;
|
|
95
|
-
deepgramBaseUrl: string;
|
|
96
|
-
whisperLocalBaseUrl?: string;
|
|
97
|
-
defaultProvider: STTProvider;
|
|
98
|
-
}
|
|
99
|
-
|
|
100
|
-
const DEFAULT_OPENAI_BASE_URL = 'https://api.openai.com/v1';
|
|
101
|
-
const DEFAULT_DEEPGRAM_BASE_URL = 'https://api.deepgram.com/v1';
|
|
102
|
-
const DEFAULT_WHISPER_LOCAL_BASE_URL = 'http://127.0.0.1:8080/v1';
|
|
103
|
-
|
|
104
|
-
function stripTrailingSlash(value: string): string {
|
|
105
|
-
return value.replace(/\/+$/, '');
|
|
106
|
-
}
|
|
107
|
-
|
|
108
|
-
function extensionFromMimeType(mimeType: string | undefined): string {
|
|
109
|
-
switch (mimeType) {
|
|
110
|
-
case 'audio/mpeg':
|
|
111
|
-
return 'mp3';
|
|
112
|
-
case 'audio/mp4':
|
|
113
|
-
case 'audio/m4a':
|
|
114
|
-
return 'm4a';
|
|
115
|
-
case 'audio/webm':
|
|
116
|
-
return 'webm';
|
|
117
|
-
case 'audio/ogg':
|
|
118
|
-
case 'audio/opus':
|
|
119
|
-
return 'ogg';
|
|
120
|
-
case 'audio/flac':
|
|
121
|
-
return 'flac';
|
|
122
|
-
case 'audio/wav':
|
|
123
|
-
case 'audio/x-wav':
|
|
124
|
-
default:
|
|
125
|
-
return 'wav';
|
|
126
|
-
}
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
function normalizeBase64Input(input: string): { data: Buffer; mimeType?: string } {
|
|
130
|
-
const trimmed = input.trim();
|
|
131
|
-
const dataUrlMatch = trimmed.match(/^data:([^;]+);base64,(.+)$/i);
|
|
132
|
-
if (dataUrlMatch) {
|
|
133
|
-
return {
|
|
134
|
-
mimeType: dataUrlMatch[1],
|
|
135
|
-
data: Buffer.from(dataUrlMatch[2], 'base64'),
|
|
136
|
-
};
|
|
137
|
-
}
|
|
138
|
-
|
|
139
|
-
return {
|
|
140
|
-
data: Buffer.from(trimmed, 'base64'),
|
|
141
|
-
};
|
|
142
|
-
}
|
|
143
|
-
|
|
144
|
-
function normalizeOpenAiSegments(input: unknown): STTSegment[] | undefined {
|
|
145
|
-
if (!Array.isArray(input)) return undefined;
|
|
146
|
-
const segments = input
|
|
147
|
-
.filter((segment) => typeof segment === 'object' && segment !== null)
|
|
148
|
-
.map((segment) => {
|
|
149
|
-
const value = segment as Record<string, unknown>;
|
|
150
|
-
const words = Array.isArray(value.words)
|
|
151
|
-
? value.words
|
|
152
|
-
.filter((word) => typeof word === 'object' && word !== null)
|
|
153
|
-
.map((word) => {
|
|
154
|
-
const item = word as Record<string, unknown>;
|
|
155
|
-
return {
|
|
156
|
-
word:
|
|
157
|
-
typeof item.word === 'string'
|
|
158
|
-
? item.word
|
|
159
|
-
: typeof item.text === 'string'
|
|
160
|
-
? item.text
|
|
161
|
-
: '',
|
|
162
|
-
start: typeof item.start === 'number' ? item.start : 0,
|
|
163
|
-
end: typeof item.end === 'number' ? item.end : 0,
|
|
164
|
-
confidence: typeof item.confidence === 'number' ? item.confidence : undefined,
|
|
165
|
-
speaker:
|
|
166
|
-
typeof item.speaker === 'string' || typeof item.speaker === 'number'
|
|
167
|
-
? item.speaker
|
|
168
|
-
: undefined,
|
|
169
|
-
};
|
|
170
|
-
})
|
|
171
|
-
: undefined;
|
|
172
|
-
|
|
173
|
-
return {
|
|
174
|
-
text: typeof value.text === 'string' ? value.text : '',
|
|
175
|
-
start: typeof value.start === 'number' ? value.start : 0,
|
|
176
|
-
end: typeof value.end === 'number' ? value.end : 0,
|
|
177
|
-
confidence: typeof value.confidence === 'number' ? value.confidence : undefined,
|
|
178
|
-
speaker:
|
|
179
|
-
typeof value.speaker === 'string' || typeof value.speaker === 'number'
|
|
180
|
-
? value.speaker
|
|
181
|
-
: undefined,
|
|
182
|
-
words,
|
|
183
|
-
};
|
|
184
|
-
})
|
|
185
|
-
.filter((segment) => segment.text || segment.end > segment.start);
|
|
186
|
-
|
|
187
|
-
return segments.length > 0 ? segments : undefined;
|
|
188
|
-
}
|
|
189
|
-
|
|
190
|
-
function buildOpenAiFormData(audio: PreparedAudio, input: STTInput, model: string): FormData {
|
|
191
|
-
const form = new FormData();
|
|
192
|
-
form.append(
|
|
193
|
-
'file',
|
|
194
|
-
new Blob([Uint8Array.from(audio.data)], { type: audio.mimeType }),
|
|
195
|
-
audio.fileName,
|
|
196
|
-
);
|
|
197
|
-
form.append('model', model);
|
|
198
|
-
form.append('response_format', input.responseFormat ?? 'verbose_json');
|
|
199
|
-
if (input.language) form.append('language', input.language);
|
|
200
|
-
if (input.prompt) form.append('prompt', input.prompt);
|
|
201
|
-
if (typeof input.temperature === 'number') {
|
|
202
|
-
form.append('temperature', String(input.temperature));
|
|
203
|
-
}
|
|
204
|
-
return form;
|
|
205
|
-
}
|
|
206
|
-
|
|
207
|
-
async function parseOpenAiLikeResponse(
|
|
208
|
-
response: Response,
|
|
209
|
-
input: STTInput,
|
|
210
|
-
provider: 'openai' | 'whisper-local',
|
|
211
|
-
model: string,
|
|
212
|
-
): Promise<STTOutput> {
|
|
213
|
-
const responseFormat = input.responseFormat ?? 'verbose_json';
|
|
214
|
-
const contentType = response.headers.get('content-type') ?? '';
|
|
215
|
-
|
|
216
|
-
if (responseFormat === 'text' || contentType.includes('text/plain')) {
|
|
217
|
-
const text = await response.text();
|
|
218
|
-
return {
|
|
219
|
-
text,
|
|
220
|
-
provider,
|
|
221
|
-
model,
|
|
222
|
-
language: input.language,
|
|
223
|
-
};
|
|
224
|
-
}
|
|
225
|
-
|
|
226
|
-
const payload = (await response.json()) as Record<string, unknown>;
|
|
227
|
-
return {
|
|
228
|
-
text: typeof payload.text === 'string' ? payload.text : '',
|
|
229
|
-
provider,
|
|
230
|
-
model,
|
|
231
|
-
language: typeof payload.language === 'string' ? payload.language : input.language,
|
|
232
|
-
durationSeconds:
|
|
233
|
-
typeof payload.duration === 'number'
|
|
234
|
-
? payload.duration
|
|
235
|
-
: typeof payload.duration_seconds === 'number'
|
|
236
|
-
? payload.duration_seconds
|
|
237
|
-
: undefined,
|
|
238
|
-
segments: normalizeOpenAiSegments(payload.segments),
|
|
239
|
-
providerResponse: payload,
|
|
240
|
-
};
|
|
241
|
-
}
|
|
242
|
-
|
|
243
|
-
function normalizeDeepgramWords(input: unknown): STTWord[] | undefined {
|
|
244
|
-
if (!Array.isArray(input)) return undefined;
|
|
245
|
-
const words = input
|
|
246
|
-
.filter((word) => typeof word === 'object' && word !== null)
|
|
247
|
-
.map((word) => {
|
|
248
|
-
const value = word as Record<string, unknown>;
|
|
249
|
-
return {
|
|
250
|
-
word:
|
|
251
|
-
typeof value.punctuated_word === 'string'
|
|
252
|
-
? value.punctuated_word
|
|
253
|
-
: typeof value.word === 'string'
|
|
254
|
-
? value.word
|
|
255
|
-
: '',
|
|
256
|
-
start: typeof value.start === 'number' ? value.start : 0,
|
|
257
|
-
end: typeof value.end === 'number' ? value.end : 0,
|
|
258
|
-
confidence: typeof value.confidence === 'number' ? value.confidence : undefined,
|
|
259
|
-
speaker:
|
|
260
|
-
typeof value.speaker === 'string' || typeof value.speaker === 'number'
|
|
261
|
-
? value.speaker
|
|
262
|
-
: undefined,
|
|
263
|
-
};
|
|
264
|
-
})
|
|
265
|
-
.filter((word) => word.word.length > 0);
|
|
266
|
-
|
|
267
|
-
return words.length > 0 ? words : undefined;
|
|
268
|
-
}
|
|
269
|
-
|
|
270
|
-
function normalizeDeepgramSegments(payload: Record<string, unknown>): STTSegment[] | undefined {
|
|
271
|
-
const results =
|
|
272
|
-
typeof payload.results === 'object' && payload.results !== null
|
|
273
|
-
? (payload.results as Record<string, unknown>)
|
|
274
|
-
: undefined;
|
|
275
|
-
|
|
276
|
-
if (results && Array.isArray(results.utterances) && results.utterances.length > 0) {
|
|
277
|
-
const utterances = results.utterances
|
|
278
|
-
.filter((utterance) => typeof utterance === 'object' && utterance !== null)
|
|
279
|
-
.map((utterance) => {
|
|
280
|
-
const value = utterance as Record<string, unknown>;
|
|
281
|
-
return {
|
|
282
|
-
text: typeof value.transcript === 'string' ? value.transcript : '',
|
|
283
|
-
start: typeof value.start === 'number' ? value.start : 0,
|
|
284
|
-
end: typeof value.end === 'number' ? value.end : 0,
|
|
285
|
-
confidence: typeof value.confidence === 'number' ? value.confidence : undefined,
|
|
286
|
-
speaker:
|
|
287
|
-
typeof value.speaker === 'string' || typeof value.speaker === 'number'
|
|
288
|
-
? value.speaker
|
|
289
|
-
: undefined,
|
|
290
|
-
words: normalizeDeepgramWords(value.words),
|
|
291
|
-
};
|
|
292
|
-
})
|
|
293
|
-
.filter((segment) => segment.text.length > 0);
|
|
294
|
-
|
|
295
|
-
return utterances.length > 0 ? utterances : undefined;
|
|
296
|
-
}
|
|
297
|
-
|
|
298
|
-
const channels = results?.channels;
|
|
299
|
-
if (!Array.isArray(channels) || channels.length === 0) return undefined;
|
|
300
|
-
const firstChannel = channels[0];
|
|
301
|
-
if (typeof firstChannel !== 'object' || firstChannel === null) return undefined;
|
|
302
|
-
const alternatives = (firstChannel as Record<string, unknown>).alternatives;
|
|
303
|
-
if (!Array.isArray(alternatives) || alternatives.length === 0) return undefined;
|
|
304
|
-
const firstAlt = alternatives[0];
|
|
305
|
-
if (typeof firstAlt !== 'object' || firstAlt === null) return undefined;
|
|
306
|
-
|
|
307
|
-
const alt = firstAlt as Record<string, unknown>;
|
|
308
|
-
const words = normalizeDeepgramWords(alt.words);
|
|
309
|
-
const start = words?.[0]?.start ?? 0;
|
|
310
|
-
const end = words?.[words.length - 1]?.end ?? 0;
|
|
311
|
-
const text = typeof alt.transcript === 'string' ? alt.transcript : '';
|
|
312
|
-
if (!text) return undefined;
|
|
313
|
-
|
|
314
|
-
return [
|
|
315
|
-
{
|
|
316
|
-
text,
|
|
317
|
-
start,
|
|
318
|
-
end,
|
|
319
|
-
confidence: typeof alt.confidence === 'number' ? alt.confidence : undefined,
|
|
320
|
-
words,
|
|
321
|
-
},
|
|
322
|
-
];
|
|
323
|
-
}
|
|
324
|
-
|
|
325
|
-
class OpenAiBackend implements SttBackend {
|
|
326
|
-
readonly provider = 'openai' as const;
|
|
327
|
-
|
|
328
|
-
isConfigured(config: ResolvedSttConfig): boolean {
|
|
329
|
-
return Boolean(config.openaiApiKey);
|
|
330
|
-
}
|
|
331
|
-
|
|
332
|
-
async transcribe(audio: PreparedAudio, input: STTInput, config: ResolvedSttConfig): Promise<STTOutput> {
|
|
333
|
-
const model = input.model || 'whisper-1';
|
|
334
|
-
const response = await fetch(
|
|
335
|
-
`${stripTrailingSlash(config.openaiBaseUrl)}/audio/transcriptions`,
|
|
336
|
-
{
|
|
337
|
-
method: 'POST',
|
|
338
|
-
headers: {
|
|
339
|
-
Authorization: `Bearer ${config.openaiApiKey!}`,
|
|
340
|
-
},
|
|
341
|
-
body: buildOpenAiFormData(audio, input, model),
|
|
342
|
-
},
|
|
343
|
-
);
|
|
344
|
-
|
|
345
|
-
if (!response.ok) {
|
|
346
|
-
const message = await response.text();
|
|
347
|
-
throw new Error(`OpenAI Whisper transcription failed (${response.status}): ${message}`);
|
|
348
|
-
}
|
|
349
|
-
|
|
350
|
-
return parseOpenAiLikeResponse(response, input, 'openai', model);
|
|
351
|
-
}
|
|
352
|
-
}
|
|
353
|
-
|
|
354
|
-
class WhisperLocalBackend implements SttBackend {
|
|
355
|
-
readonly provider = 'whisper-local' as const;
|
|
356
|
-
|
|
357
|
-
isConfigured(config: ResolvedSttConfig): boolean {
|
|
358
|
-
return Boolean(config.whisperLocalBaseUrl);
|
|
359
|
-
}
|
|
360
|
-
|
|
361
|
-
async transcribe(audio: PreparedAudio, input: STTInput, config: ResolvedSttConfig): Promise<STTOutput> {
|
|
362
|
-
const baseUrl = config.whisperLocalBaseUrl || DEFAULT_WHISPER_LOCAL_BASE_URL;
|
|
363
|
-
const model = input.model || 'base';
|
|
364
|
-
const response = await fetch(
|
|
365
|
-
`${stripTrailingSlash(baseUrl)}/audio/transcriptions`,
|
|
366
|
-
{
|
|
367
|
-
method: 'POST',
|
|
368
|
-
body: buildOpenAiFormData(audio, input, model),
|
|
369
|
-
},
|
|
370
|
-
);
|
|
371
|
-
|
|
372
|
-
if (!response.ok) {
|
|
373
|
-
const message = await response.text();
|
|
374
|
-
throw new Error(
|
|
375
|
-
`Whisper-local transcription failed (${response.status}): ${message}. ` +
|
|
376
|
-
'Ensure your local STT server exposes an OpenAI-compatible /audio/transcriptions endpoint.',
|
|
377
|
-
);
|
|
378
|
-
}
|
|
379
|
-
|
|
380
|
-
return parseOpenAiLikeResponse(response, input, 'whisper-local', model);
|
|
381
|
-
}
|
|
382
|
-
}
|
|
383
|
-
|
|
384
|
-
class DeepgramBackend implements SttBackend {
|
|
385
|
-
readonly provider = 'deepgram' as const;
|
|
386
|
-
|
|
387
|
-
isConfigured(config: ResolvedSttConfig): boolean {
|
|
388
|
-
return Boolean(config.deepgramApiKey);
|
|
389
|
-
}
|
|
390
|
-
|
|
391
|
-
async transcribe(audio: PreparedAudio, input: STTInput, config: ResolvedSttConfig): Promise<STTOutput> {
|
|
392
|
-
const model = input.model || 'nova-2';
|
|
393
|
-
const params = new URLSearchParams({
|
|
394
|
-
model,
|
|
395
|
-
smart_format: String(input.smartFormat ?? true),
|
|
396
|
-
punctuate: 'true',
|
|
397
|
-
utterances: String(input.utterances ?? true),
|
|
398
|
-
diarize: String(input.diarize ?? false),
|
|
399
|
-
});
|
|
400
|
-
|
|
401
|
-
if (input.language) params.set('language', input.language);
|
|
402
|
-
if (input.detectLanguage) params.set('detect_language', 'true');
|
|
403
|
-
if (input.prompt) params.set('keywords', input.prompt);
|
|
404
|
-
|
|
405
|
-
const response = await fetch(
|
|
406
|
-
`${stripTrailingSlash(config.deepgramBaseUrl)}/listen?${params.toString()}`,
|
|
407
|
-
{
|
|
408
|
-
method: 'POST',
|
|
409
|
-
headers: {
|
|
410
|
-
Authorization: `Token ${config.deepgramApiKey!}`,
|
|
411
|
-
'Content-Type': audio.mimeType,
|
|
412
|
-
},
|
|
413
|
-
body: audio.data,
|
|
414
|
-
},
|
|
415
|
-
);
|
|
416
|
-
|
|
417
|
-
if (!response.ok) {
|
|
418
|
-
const message = await response.text();
|
|
419
|
-
throw new Error(`Deepgram transcription failed (${response.status}): ${message}`);
|
|
420
|
-
}
|
|
421
|
-
|
|
422
|
-
const payload = (await response.json()) as Record<string, unknown>;
|
|
423
|
-
const results =
|
|
424
|
-
typeof payload.results === 'object' && payload.results !== null
|
|
425
|
-
? (payload.results as Record<string, unknown>)
|
|
426
|
-
: {};
|
|
427
|
-
const channels = Array.isArray(results.channels) ? results.channels : [];
|
|
428
|
-
const firstChannel =
|
|
429
|
-
channels.length > 0 && typeof channels[0] === 'object' && channels[0] !== null
|
|
430
|
-
? (channels[0] as Record<string, unknown>)
|
|
431
|
-
: undefined;
|
|
432
|
-
const alternatives = Array.isArray(firstChannel?.alternatives) ? firstChannel!.alternatives : [];
|
|
433
|
-
const firstAlt =
|
|
434
|
-
alternatives.length > 0 && typeof alternatives[0] === 'object' && alternatives[0] !== null
|
|
435
|
-
? (alternatives[0] as Record<string, unknown>)
|
|
436
|
-
: undefined;
|
|
437
|
-
const transcript = typeof firstAlt?.transcript === 'string' ? firstAlt.transcript : '';
|
|
438
|
-
|
|
439
|
-
const metadata =
|
|
440
|
-
typeof payload.metadata === 'object' && payload.metadata !== null
|
|
441
|
-
? (payload.metadata as Record<string, unknown>)
|
|
442
|
-
: {};
|
|
443
|
-
const language =
|
|
444
|
-
typeof firstAlt?.detected_language === 'string'
|
|
445
|
-
? firstAlt.detected_language
|
|
446
|
-
: typeof firstChannel?.detected_language === 'string'
|
|
447
|
-
? firstChannel.detected_language
|
|
448
|
-
: input.language;
|
|
449
|
-
|
|
450
|
-
return {
|
|
451
|
-
text: transcript,
|
|
452
|
-
provider: 'deepgram',
|
|
453
|
-
model,
|
|
454
|
-
language,
|
|
455
|
-
confidence: typeof firstAlt?.confidence === 'number' ? firstAlt.confidence : undefined,
|
|
456
|
-
durationSeconds:
|
|
457
|
-
typeof metadata.duration === 'number' ? metadata.duration : undefined,
|
|
458
|
-
segments: normalizeDeepgramSegments(payload),
|
|
459
|
-
providerResponse: payload,
|
|
460
|
-
};
|
|
461
|
-
}
|
|
462
|
-
}
|
|
463
|
-
|
|
464
|
-
const STT_BACKENDS: Record<Exclude<STTProvider, 'auto'>, SttBackend> = {
|
|
465
|
-
openai: new OpenAiBackend(),
|
|
466
|
-
deepgram: new DeepgramBackend(),
|
|
467
|
-
'whisper-local': new WhisperLocalBackend(),
|
|
468
|
-
};
|
|
469
|
-
|
|
470
|
-
export class SpeechToTextTool implements ITool<STTInput, STTOutput> {
|
|
471
|
-
readonly id = 'stt-multi-provider-v1';
|
|
472
|
-
readonly name = 'speech_to_text';
|
|
473
|
-
readonly displayName = 'Speech to Text';
|
|
474
|
-
readonly description =
|
|
475
|
-
'Transcribe audio into text. Supports OpenAI Whisper, Deepgram, and Whisper-local/OpenAI-compatible local STT runtimes. ' +
|
|
476
|
-
'Accepts either base64 audio or a fetchable audio URL.';
|
|
477
|
-
readonly category = 'media';
|
|
478
|
-
readonly version = '2.0.0';
|
|
479
|
-
readonly hasSideEffects = false;
|
|
480
|
-
readonly requiredCapabilities = ['capability:stt'];
|
|
481
|
-
|
|
482
|
-
readonly inputSchema: JSONSchemaObject = {
|
|
483
|
-
type: 'object',
|
|
484
|
-
properties: {
|
|
485
|
-
audioBase64: {
|
|
486
|
-
type: 'string',
|
|
487
|
-
description: 'Base64 audio payload. May be raw base64 or a data URL.',
|
|
488
|
-
},
|
|
489
|
-
audioUrl: {
|
|
490
|
-
type: 'string',
|
|
491
|
-
description: 'Fetchable remote audio URL. Used when audio is not provided inline.',
|
|
492
|
-
},
|
|
493
|
-
mimeType: {
|
|
494
|
-
type: 'string',
|
|
495
|
-
description: 'Optional MIME type override, such as audio/wav or audio/mpeg.',
|
|
496
|
-
},
|
|
497
|
-
fileName: {
|
|
498
|
-
type: 'string',
|
|
499
|
-
description: 'Optional filename sent to transcription providers.',
|
|
500
|
-
},
|
|
501
|
-
format: {
|
|
502
|
-
type: 'string',
|
|
503
|
-
description: 'Optional audio format hint, such as wav, mp3, m4a, or webm.',
|
|
504
|
-
},
|
|
505
|
-
language: {
|
|
506
|
-
type: 'string',
|
|
507
|
-
description: 'Optional ISO language hint, for example en or es.',
|
|
508
|
-
},
|
|
509
|
-
prompt: {
|
|
510
|
-
type: 'string',
|
|
511
|
-
description: 'Optional context prompt to bias the transcript.',
|
|
512
|
-
},
|
|
513
|
-
temperature: {
|
|
514
|
-
type: 'number',
|
|
515
|
-
minimum: 0,
|
|
516
|
-
maximum: 1,
|
|
517
|
-
description: 'Temperature override for Whisper-style providers.',
|
|
518
|
-
},
|
|
519
|
-
responseFormat: {
|
|
520
|
-
type: 'string',
|
|
521
|
-
enum: ['json', 'text', 'srt', 'verbose_json', 'vtt'],
|
|
522
|
-
description: 'Response format for OpenAI-compatible providers.',
|
|
523
|
-
},
|
|
524
|
-
provider: {
|
|
525
|
-
type: 'string',
|
|
526
|
-
enum: ['auto', 'openai', 'deepgram', 'whisper-local'],
|
|
527
|
-
description: 'STT provider selection. Default: auto.',
|
|
528
|
-
},
|
|
529
|
-
model: {
|
|
530
|
-
type: 'string',
|
|
531
|
-
description: 'Provider model override. Examples: whisper-1, nova-2, base.',
|
|
532
|
-
},
|
|
533
|
-
diarize: {
|
|
534
|
-
type: 'boolean',
|
|
535
|
-
description: 'Enable speaker diarization when the provider supports it.',
|
|
536
|
-
},
|
|
537
|
-
utterances: {
|
|
538
|
-
type: 'boolean',
|
|
539
|
-
description: 'Request utterance segmentation when the provider supports it.',
|
|
540
|
-
},
|
|
541
|
-
smartFormat: {
|
|
542
|
-
type: 'boolean',
|
|
543
|
-
description: 'Enable provider-side smart formatting where supported.',
|
|
544
|
-
},
|
|
545
|
-
detectLanguage: {
|
|
546
|
-
type: 'boolean',
|
|
547
|
-
description: 'Enable provider-side language detection where supported.',
|
|
548
|
-
},
|
|
549
|
-
providerOptions: {
|
|
550
|
-
type: 'object',
|
|
551
|
-
description: 'Optional provider-specific passthrough options for future-compatible callers.',
|
|
552
|
-
additionalProperties: true,
|
|
553
|
-
},
|
|
554
|
-
},
|
|
555
|
-
required: [],
|
|
556
|
-
};
|
|
557
|
-
|
|
558
|
-
private readonly config: ResolvedSttConfig;
|
|
559
|
-
|
|
560
|
-
constructor(config?: STTConfig) {
|
|
561
|
-
this.config = {
|
|
562
|
-
openaiApiKey: config?.openaiApiKey || process.env.OPENAI_API_KEY || undefined,
|
|
563
|
-
openaiBaseUrl: stripTrailingSlash(
|
|
564
|
-
config?.openaiBaseUrl || process.env.OPENAI_BASE_URL || DEFAULT_OPENAI_BASE_URL,
|
|
565
|
-
),
|
|
566
|
-
deepgramApiKey: config?.deepgramApiKey || process.env.DEEPGRAM_API_KEY || undefined,
|
|
567
|
-
deepgramBaseUrl: stripTrailingSlash(
|
|
568
|
-
config?.deepgramBaseUrl || process.env.DEEPGRAM_BASE_URL || DEFAULT_DEEPGRAM_BASE_URL,
|
|
569
|
-
),
|
|
570
|
-
whisperLocalBaseUrl: config?.whisperLocalBaseUrl || process.env.WHISPER_LOCAL_BASE_URL || undefined,
|
|
571
|
-
defaultProvider:
|
|
572
|
-
config?.defaultProvider || (process.env.STT_PROVIDER as STTProvider) || 'auto',
|
|
573
|
-
};
|
|
574
|
-
}
|
|
575
|
-
|
|
576
|
-
private resolveProvider(requested?: STTProvider): Exclude<STTProvider, 'auto'> | null {
|
|
577
|
-
const preferred = requested || this.config.defaultProvider || 'auto';
|
|
578
|
-
|
|
579
|
-
if (preferred !== 'auto') {
|
|
580
|
-
if (preferred === 'whisper-local') return 'whisper-local';
|
|
581
|
-
return STT_BACKENDS[preferred].isConfigured(this.config) ? preferred : null;
|
|
582
|
-
}
|
|
583
|
-
|
|
584
|
-
if (STT_BACKENDS.openai.isConfigured(this.config)) return 'openai';
|
|
585
|
-
if (STT_BACKENDS.deepgram.isConfigured(this.config)) return 'deepgram';
|
|
586
|
-
if (STT_BACKENDS['whisper-local'].isConfigured(this.config)) return 'whisper-local';
|
|
587
|
-
return null;
|
|
588
|
-
}
|
|
589
|
-
|
|
590
|
-
private async prepareAudio(input: STTInput): Promise<PreparedAudio> {
|
|
591
|
-
if (typeof input.audioBase64 === 'string' && input.audioBase64.trim()) {
|
|
592
|
-
const normalized = normalizeBase64Input(input.audioBase64);
|
|
593
|
-
const mimeType = input.mimeType || normalized.mimeType || 'audio/wav';
|
|
594
|
-
const format = input.format || extensionFromMimeType(mimeType);
|
|
595
|
-
return {
|
|
596
|
-
data: normalized.data,
|
|
597
|
-
mimeType,
|
|
598
|
-
format,
|
|
599
|
-
fileName: input.fileName || `audio.${format}`,
|
|
600
|
-
};
|
|
601
|
-
}
|
|
602
|
-
|
|
603
|
-
if (typeof input.audioUrl === 'string' && input.audioUrl.trim()) {
|
|
604
|
-
const response = await fetch(input.audioUrl);
|
|
605
|
-
if (!response.ok) {
|
|
606
|
-
throw new Error(`Audio download failed (${response.status})`);
|
|
607
|
-
}
|
|
608
|
-
|
|
609
|
-
const mimeType = input.mimeType || response.headers.get('content-type') || 'audio/wav';
|
|
610
|
-
const format = input.format || extensionFromMimeType(mimeType);
|
|
611
|
-
return {
|
|
612
|
-
data: Buffer.from(await response.arrayBuffer()),
|
|
613
|
-
mimeType,
|
|
614
|
-
format,
|
|
615
|
-
fileName: input.fileName || `audio.${format}`,
|
|
616
|
-
};
|
|
617
|
-
}
|
|
618
|
-
|
|
619
|
-
throw new Error('Provide either audioBase64 or audioUrl.');
|
|
620
|
-
}
|
|
621
|
-
|
|
622
|
-
async execute(
|
|
623
|
-
args: STTInput,
|
|
624
|
-
_context: ToolExecutionContext,
|
|
625
|
-
): Promise<ToolExecutionResult<STTOutput>> {
|
|
626
|
-
try {
|
|
627
|
-
const provider = this.resolveProvider(args.provider);
|
|
628
|
-
if (!provider) {
|
|
629
|
-
return {
|
|
630
|
-
success: false,
|
|
631
|
-
error:
|
|
632
|
-
'No STT provider available. Configure OPENAI_API_KEY, DEEPGRAM_API_KEY, or WHISPER_LOCAL_BASE_URL. ' +
|
|
633
|
-
'You can also explicitly set provider to "whisper-local" to target a local OpenAI-compatible STT server.',
|
|
634
|
-
};
|
|
635
|
-
}
|
|
636
|
-
|
|
637
|
-
const audio = await this.prepareAudio(args);
|
|
638
|
-
const output = await STT_BACKENDS[provider].transcribe(audio, args, this.config);
|
|
639
|
-
return {
|
|
640
|
-
success: true,
|
|
641
|
-
output,
|
|
642
|
-
};
|
|
643
|
-
} catch (error) {
|
|
644
|
-
return {
|
|
645
|
-
success: false,
|
|
646
|
-
error: error instanceof Error ? error.message : String(error),
|
|
647
|
-
};
|
|
648
|
-
}
|
|
649
|
-
}
|
|
650
|
-
}
|