@andreprado/agentkit 0.1.0-alpha.15 → 0.1.0-alpha.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/guides/add-channel.md +63 -0
- package/docs/guides/channel-security.md +32 -0
- package/docs/guides/connect-telegram.md +58 -0
- package/docs/guides/connect-whatsapp-zapster.md +65 -0
- package/package.json +1 -1
- package/src/cli/commands/channels.ts +2 -0
- package/src/index.ts +86 -0
- package/src/runtime/channel-test-harness.ts +2 -0
- package/src/runtime/channels/telegram.ts +326 -10
- package/src/runtime/channels/whatsapp-zapster.ts +319 -0
- package/src/runtime/channels.ts +47 -1
- package/src/runtime/config.ts +85 -4
- package/src/runtime/core/manifest.ts +32 -0
- package/src/runtime/dev-server.ts +72 -4
- package/src/runtime/inspect.ts +34 -0
- package/src/runtime/targets/cloudflare/build.ts +2 -0
- package/src/runtime/transcription.ts +483 -0
- package/src/templates/skills/agentkit-channels/SKILL.md +34 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +13 -0
- package/src/templates/skills/agentkit-channels/references/telegram.md +32 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +29 -0
|
@@ -7,7 +7,7 @@ import { Readable } from "node:stream";
|
|
|
7
7
|
|
|
8
8
|
import { openCapsuleStore, type ConversationRecord } from "../storage/sqlite";
|
|
9
9
|
import type { AgentChannel, ChannelProvider, ChannelType } from "../index";
|
|
10
|
-
import type { ChannelAdapter, RawWebhookEvent } from "./channels";
|
|
10
|
+
import type { ChannelAdapter, NormalizedChannelMessage, RawWebhookEvent } from "./channels";
|
|
11
11
|
import { telegramChannelAdapter } from "./channels/telegram";
|
|
12
12
|
import { websiteChannelAdapter } from "./channels/website";
|
|
13
13
|
import { metaWhatsappChannelAdapter } from "./channels/whatsapp-meta";
|
|
@@ -20,6 +20,7 @@ import { loadCapsuleEnv } from "./env";
|
|
|
20
20
|
import { buildInspectState } from "./inspect";
|
|
21
21
|
import { syncConfiguredKnowledgeSources } from "./knowledge/ingest";
|
|
22
22
|
import { runToolFromCwd } from "./tool-runner";
|
|
23
|
+
import { resolveTranscriptionConfig, transcribeAudio } from "./transcription";
|
|
23
24
|
|
|
24
25
|
export type AgentDevServerOptions = {
|
|
25
26
|
port?: number;
|
|
@@ -515,7 +516,9 @@ async function handlePortableChannel(
|
|
|
515
516
|
|
|
516
517
|
channelRuntime.dedupeKeys.add(event.dedupeKey);
|
|
517
518
|
|
|
518
|
-
|
|
519
|
+
const processedMessage = await processDevChannelMessage(capsule, adapter, channel, event.message, secrets);
|
|
520
|
+
|
|
521
|
+
if (event.kind !== "message" || !processedMessage) {
|
|
519
522
|
deliveries.push({ state: "skipped", dedupeKey: event.dedupeKey, unsupportedReason: event.unsupportedReason });
|
|
520
523
|
continue;
|
|
521
524
|
}
|
|
@@ -531,7 +534,7 @@ async function handlePortableChannel(
|
|
|
531
534
|
bufferDevChannelMessage(capsule, channelRuntime, {
|
|
532
535
|
bufferKey: `${channel.name}:${event.externalIdentity.key}`,
|
|
533
536
|
conversationId,
|
|
534
|
-
content:
|
|
537
|
+
content: processedMessage.content,
|
|
535
538
|
receivedAt: event.receivedAt,
|
|
536
539
|
buffer,
|
|
537
540
|
});
|
|
@@ -545,7 +548,7 @@ async function handlePortableChannel(
|
|
|
545
548
|
}
|
|
546
549
|
|
|
547
550
|
const result = await runAgentMessage(capsule, {
|
|
548
|
-
message:
|
|
551
|
+
message: processedMessage.content,
|
|
549
552
|
conversationId,
|
|
550
553
|
signal: request.signal,
|
|
551
554
|
});
|
|
@@ -560,6 +563,71 @@ async function handlePortableChannel(
|
|
|
560
563
|
return jsonResponse({ deliveries });
|
|
561
564
|
}
|
|
562
565
|
|
|
566
|
+
async function processDevChannelMessage(
|
|
567
|
+
capsule: LoadedAgentCapsule,
|
|
568
|
+
adapter: ChannelAdapter,
|
|
569
|
+
channel: AgentChannel,
|
|
570
|
+
message: NormalizedChannelMessage | undefined,
|
|
571
|
+
secrets: Record<string, string>,
|
|
572
|
+
): Promise<NormalizedChannelMessage | null> {
|
|
573
|
+
if (!message) {
|
|
574
|
+
return null;
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
if (message.contentType === "text" && message.content) {
|
|
578
|
+
return message;
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
if (message.contentType !== "audio" || !message.audio || channel.audio?.mode !== "transcribe") {
|
|
582
|
+
return null;
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
if (!adapter.downloadMedia) {
|
|
586
|
+
return null;
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
const env = await loadCapsuleEnv(capsule.root);
|
|
590
|
+
const transcriptionConfig = resolveTranscriptionConfig(capsule.config.transcription);
|
|
591
|
+
const secretName = transcriptionConfig.secret;
|
|
592
|
+
const download = await adapter.downloadMedia({
|
|
593
|
+
type: channel.type,
|
|
594
|
+
provider: channel.provider,
|
|
595
|
+
channelId: channel.name,
|
|
596
|
+
message,
|
|
597
|
+
secrets: {
|
|
598
|
+
...secrets,
|
|
599
|
+
...(secretName ? { [secretName]: env[secretName] ?? "" } : {}),
|
|
600
|
+
...(env.AGENTKIT_TELEGRAM_API_BASE_URL ? { AGENTKIT_TELEGRAM_API_BASE_URL: env.AGENTKIT_TELEGRAM_API_BASE_URL } : {}),
|
|
601
|
+
},
|
|
602
|
+
});
|
|
603
|
+
|
|
604
|
+
if (!download.ok) {
|
|
605
|
+
return null;
|
|
606
|
+
}
|
|
607
|
+
|
|
608
|
+
const transcription = await transcribeAudio({
|
|
609
|
+
config: transcriptionConfig,
|
|
610
|
+
secrets: {
|
|
611
|
+
...(secretName ? { [secretName]: env[secretName] ?? "" } : {}),
|
|
612
|
+
},
|
|
613
|
+
audio: download.audio,
|
|
614
|
+
filename: download.filename,
|
|
615
|
+
...(download.mimeType ? { mimeType: download.mimeType } : {}),
|
|
616
|
+
...(download.durationSeconds !== undefined ? { durationSeconds: download.durationSeconds } : {}),
|
|
617
|
+
});
|
|
618
|
+
|
|
619
|
+
if (!transcription.ok) {
|
|
620
|
+
return null;
|
|
621
|
+
}
|
|
622
|
+
|
|
623
|
+
return {
|
|
624
|
+
role: "user",
|
|
625
|
+
contentType: "text",
|
|
626
|
+
content: `The user sent an audio message.\n\nTranscript:\n${transcription.text}`,
|
|
627
|
+
providerMessageId: message.providerMessageId,
|
|
628
|
+
};
|
|
629
|
+
}
|
|
630
|
+
|
|
563
631
|
function bufferDevChannelMessage(
|
|
564
632
|
capsule: LoadedAgentCapsule,
|
|
565
633
|
runtime: DevChannelRuntimeState,
|
package/src/runtime/inspect.ts
CHANGED
|
@@ -7,6 +7,7 @@ import { loadCapsuleEnv } from "./env";
|
|
|
7
7
|
import { resolveKnowledgeConfig } from "./knowledge/config";
|
|
8
8
|
import { KNOWLEDGE_SEARCH_TOOL_NAME } from "./knowledge/tool";
|
|
9
9
|
import { resolveRuntimeTimeZone } from "./prompt-context";
|
|
10
|
+
import { resolveTranscriptionConfig } from "./transcription";
|
|
10
11
|
|
|
11
12
|
export type SecretState = "set" | "missing";
|
|
12
13
|
|
|
@@ -21,6 +22,19 @@ export type AgentInspectState = {
|
|
|
21
22
|
prompt: string;
|
|
22
23
|
tools: string[];
|
|
23
24
|
channels: AgentInspectChannel[];
|
|
25
|
+
transcription: {
|
|
26
|
+
enabled: boolean;
|
|
27
|
+
provider: string | null;
|
|
28
|
+
model: string | null;
|
|
29
|
+
secret: string | null;
|
|
30
|
+
language: string | null;
|
|
31
|
+
prompt: string | null;
|
|
32
|
+
limits: {
|
|
33
|
+
maxDurationSeconds: number | null;
|
|
34
|
+
maxBytes: number | null;
|
|
35
|
+
};
|
|
36
|
+
rawAudioTtlSeconds: number | null;
|
|
37
|
+
};
|
|
24
38
|
knowledge: {
|
|
25
39
|
enabled: boolean;
|
|
26
40
|
sources: string[];
|
|
@@ -80,6 +94,7 @@ export type AgentInspectChannel = {
|
|
|
80
94
|
provider: AgentChannel["provider"];
|
|
81
95
|
secrets: string[];
|
|
82
96
|
buffer?: AgentChannel["buffer"];
|
|
97
|
+
audio?: AgentChannel["audio"];
|
|
83
98
|
};
|
|
84
99
|
|
|
85
100
|
export async function inspectAgentCapsule(
|
|
@@ -96,6 +111,7 @@ export function buildInspectState(
|
|
|
96
111
|
): AgentInspectState {
|
|
97
112
|
const declaredSecrets = new Set(capsule.config.secrets);
|
|
98
113
|
const knowledge = resolveKnowledgeConfig(capsule.config);
|
|
114
|
+
const transcription = resolveTranscriptionConfig(capsule.config.transcription);
|
|
99
115
|
|
|
100
116
|
for (const tool of capsule.config.tools ?? []) {
|
|
101
117
|
for (const secret of tool.secrets ?? []) {
|
|
@@ -113,6 +129,10 @@ export function buildInspectState(
|
|
|
113
129
|
declaredSecrets.add(knowledge.embedding.secret);
|
|
114
130
|
}
|
|
115
131
|
|
|
132
|
+
if (transcription.secret) {
|
|
133
|
+
declaredSecrets.add(transcription.secret);
|
|
134
|
+
}
|
|
135
|
+
|
|
116
136
|
const managedSecrets = managedCloudSecrets(capsule);
|
|
117
137
|
const userSecrets = Array.from(declaredSecrets).sort();
|
|
118
138
|
const secrets: Record<string, SecretState> = {};
|
|
@@ -142,7 +162,21 @@ export function buildInspectState(
|
|
|
142
162
|
provider: channel.provider,
|
|
143
163
|
secrets: [...channel.secrets],
|
|
144
164
|
...(channel.buffer ? { buffer: channel.buffer } : {}),
|
|
165
|
+
...(channel.audio ? { audio: channel.audio } : {}),
|
|
145
166
|
})),
|
|
167
|
+
transcription: {
|
|
168
|
+
enabled: transcription.enabled,
|
|
169
|
+
provider: transcription.enabled ? transcription.provider : null,
|
|
170
|
+
model: transcription.enabled ? transcription.model : null,
|
|
171
|
+
secret: transcription.secret,
|
|
172
|
+
language: transcription.language,
|
|
173
|
+
prompt: transcription.prompt,
|
|
174
|
+
limits: {
|
|
175
|
+
maxDurationSeconds: transcription.limits.maxDurationSeconds ?? null,
|
|
176
|
+
maxBytes: transcription.limits.maxBytes ?? null,
|
|
177
|
+
},
|
|
178
|
+
rawAudioTtlSeconds: transcription.rawAudioTtlSeconds,
|
|
179
|
+
},
|
|
146
180
|
knowledge: {
|
|
147
181
|
enabled: knowledge.enabled,
|
|
148
182
|
sources: knowledge.sources.map((source) => source.value),
|
|
@@ -37,6 +37,7 @@ export type AgentManifest = {
|
|
|
37
37
|
secrets: SharedAgentManifest["secrets"];
|
|
38
38
|
tools: SharedAgentManifest["tools"];
|
|
39
39
|
channels: SharedAgentManifest["channels"];
|
|
40
|
+
transcription: SharedAgentManifest["transcription"];
|
|
40
41
|
knowledge: SharedAgentManifest["knowledge"];
|
|
41
42
|
access: SharedAgentManifest["access"];
|
|
42
43
|
storage: {
|
|
@@ -198,6 +199,7 @@ function buildManifest(
|
|
|
198
199
|
secrets: shared.secrets,
|
|
199
200
|
tools: shared.tools,
|
|
200
201
|
channels: shared.channels,
|
|
202
|
+
transcription: shared.transcription,
|
|
201
203
|
knowledge: shared.knowledge,
|
|
202
204
|
access: shared.access,
|
|
203
205
|
storage: {
|
|
@@ -0,0 +1,483 @@
|
|
|
1
|
+
import type { AgentTranscriptionConfig, TranscriptionProviderName } from "../index";
|
|
2
|
+
|
|
3
|
+
export type ResolvedTranscriptionConfig = {
|
|
4
|
+
enabled: boolean;
|
|
5
|
+
provider: TranscriptionProviderName;
|
|
6
|
+
model: string;
|
|
7
|
+
secret: string | null;
|
|
8
|
+
language: string | null;
|
|
9
|
+
prompt: string | null;
|
|
10
|
+
limits: {
|
|
11
|
+
maxDurationSeconds?: number;
|
|
12
|
+
maxBytes?: number;
|
|
13
|
+
};
|
|
14
|
+
rawAudioTtlSeconds: number | null;
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
export type TranscriptionInput = {
|
|
18
|
+
config: ResolvedTranscriptionConfig;
|
|
19
|
+
secrets: Record<string, string>;
|
|
20
|
+
audio: Uint8Array;
|
|
21
|
+
filename: string;
|
|
22
|
+
mimeType?: string;
|
|
23
|
+
durationSeconds?: number;
|
|
24
|
+
providerMetadata?: Record<string, unknown>;
|
|
25
|
+
};
|
|
26
|
+
|
|
27
|
+
export type TranscriptionFetch = (input: RequestInfo | URL, init?: RequestInit) => Promise<Response>;
|
|
28
|
+
|
|
29
|
+
export type TranscriptionResult =
|
|
30
|
+
| {
|
|
31
|
+
ok: true;
|
|
32
|
+
text: string;
|
|
33
|
+
provider: TranscriptionProviderName;
|
|
34
|
+
model: string;
|
|
35
|
+
language?: string;
|
|
36
|
+
durationSeconds?: number;
|
|
37
|
+
providerMetadata?: Record<string, unknown>;
|
|
38
|
+
}
|
|
39
|
+
| {
|
|
40
|
+
ok: false;
|
|
41
|
+
retryable: boolean;
|
|
42
|
+
code:
|
|
43
|
+
| "transcription_disabled"
|
|
44
|
+
| "transcription_secret_missing"
|
|
45
|
+
| "transcription_model_unsupported"
|
|
46
|
+
| "transcription_audio_too_large"
|
|
47
|
+
| "transcription_audio_too_long"
|
|
48
|
+
| "transcription_audio_format_unsupported"
|
|
49
|
+
| "transcription_provider_unavailable"
|
|
50
|
+
| "transcription_failed";
|
|
51
|
+
message: string;
|
|
52
|
+
provider?: TranscriptionProviderName;
|
|
53
|
+
model?: string;
|
|
54
|
+
providerMetadata?: Record<string, unknown>;
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
export type TranscriptionAdapter = {
|
|
58
|
+
provider: TranscriptionProviderName;
|
|
59
|
+
models: string[];
|
|
60
|
+
supportedMimeTypes: string[];
|
|
61
|
+
supportedExtensions: string[];
|
|
62
|
+
maxBytes: number;
|
|
63
|
+
requiredSecret(config: ResolvedTranscriptionConfig): string | null;
|
|
64
|
+
transcribe(input: TranscriptionInput, fetcher?: TranscriptionFetch): Promise<TranscriptionResult>;
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
const DEFAULT_MAX_BYTES = 25_000_000;
|
|
68
|
+
const OPENAI_MODELS = ["gpt-4o-mini-transcribe", "gpt-4o-transcribe", "whisper-1"];
|
|
69
|
+
const GROQ_MODELS = ["whisper-large-v3-turbo", "whisper-large-v3", "distil-whisper-large-v3-en"];
|
|
70
|
+
const OPENAI_EXTENSIONS = ["mp3", "mp4", "mpeg", "mpga", "m4a", "wav", "webm"];
|
|
71
|
+
const GROQ_EXTENSIONS = ["flac", "mp3", "mp4", "mpeg", "mpga", "m4a", "ogg", "wav", "webm"];
|
|
72
|
+
const OPENAI_MIME_TYPES = [
|
|
73
|
+
"audio/mpeg",
|
|
74
|
+
"audio/mp3",
|
|
75
|
+
"audio/mp4",
|
|
76
|
+
"audio/mpga",
|
|
77
|
+
"audio/m4a",
|
|
78
|
+
"audio/wav",
|
|
79
|
+
"audio/webm",
|
|
80
|
+
"video/mp4",
|
|
81
|
+
];
|
|
82
|
+
const GROQ_MIME_TYPES = [
|
|
83
|
+
...OPENAI_MIME_TYPES,
|
|
84
|
+
"audio/flac",
|
|
85
|
+
"audio/ogg",
|
|
86
|
+
"audio/opus",
|
|
87
|
+
"application/ogg",
|
|
88
|
+
];
|
|
89
|
+
|
|
90
|
+
export function resolveTranscriptionConfig(config: AgentTranscriptionConfig | undefined): ResolvedTranscriptionConfig {
|
|
91
|
+
if (!config) {
|
|
92
|
+
return {
|
|
93
|
+
enabled: false,
|
|
94
|
+
provider: "test",
|
|
95
|
+
model: "fake",
|
|
96
|
+
secret: null,
|
|
97
|
+
language: null,
|
|
98
|
+
prompt: null,
|
|
99
|
+
limits: {},
|
|
100
|
+
rawAudioTtlSeconds: null,
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
return {
|
|
105
|
+
enabled: true,
|
|
106
|
+
provider: config.provider,
|
|
107
|
+
model: config.model,
|
|
108
|
+
secret: config.secret ?? defaultTranscriptionSecret(config.provider),
|
|
109
|
+
language: config.language ?? null,
|
|
110
|
+
prompt: config.prompt ?? null,
|
|
111
|
+
limits: {
|
|
112
|
+
...(config.limits?.maxDurationSeconds !== undefined
|
|
113
|
+
? { maxDurationSeconds: config.limits.maxDurationSeconds }
|
|
114
|
+
: {}),
|
|
115
|
+
...(config.limits?.maxBytes !== undefined ? { maxBytes: config.limits.maxBytes } : {}),
|
|
116
|
+
},
|
|
117
|
+
rawAudioTtlSeconds: config.rawAudioTtlSeconds ?? 3600,
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export async function transcribeAudio(
|
|
122
|
+
input: TranscriptionInput,
|
|
123
|
+
fetcher: TranscriptionFetch = fetch,
|
|
124
|
+
): Promise<TranscriptionResult> {
|
|
125
|
+
if (!input.config.enabled) {
|
|
126
|
+
return {
|
|
127
|
+
ok: false,
|
|
128
|
+
retryable: false,
|
|
129
|
+
code: "transcription_disabled",
|
|
130
|
+
message: "Audio transcription is not configured for this Agent Capsule.",
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
const adapter = transcriptionAdapterFor(input.config.provider);
|
|
135
|
+
|
|
136
|
+
if (!adapter) {
|
|
137
|
+
return {
|
|
138
|
+
ok: false,
|
|
139
|
+
retryable: false,
|
|
140
|
+
code: "transcription_model_unsupported",
|
|
141
|
+
message: `No transcription adapter is registered for ${input.config.provider}.`,
|
|
142
|
+
provider: input.config.provider,
|
|
143
|
+
model: input.config.model,
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
const validation = validateTranscriptionInput(adapter, input);
|
|
148
|
+
|
|
149
|
+
if (!validation.ok) {
|
|
150
|
+
return validation;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
return adapter.transcribe(input, fetcher);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
export function transcriptionAdapterFor(provider: TranscriptionProviderName): TranscriptionAdapter | null {
|
|
157
|
+
if (provider === "test") {
|
|
158
|
+
return testTranscriptionAdapter;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
if (provider === "openai") {
|
|
162
|
+
return openaiTranscriptionAdapter;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
if (provider === "groq") {
|
|
166
|
+
return groqTranscriptionAdapter;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
return null;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
export function defaultTranscriptionSecret(provider: TranscriptionProviderName): string | null {
|
|
173
|
+
if (provider === "openai") {
|
|
174
|
+
return "OPENAI_API_KEY";
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
if (provider === "groq") {
|
|
178
|
+
return "GROQ_API_KEY";
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
return null;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
function validateTranscriptionInput(adapter: TranscriptionAdapter, input: TranscriptionInput): TranscriptionResult {
|
|
185
|
+
if (!adapter.models.includes(input.config.model)) {
|
|
186
|
+
return {
|
|
187
|
+
ok: false,
|
|
188
|
+
retryable: false,
|
|
189
|
+
code: "transcription_model_unsupported",
|
|
190
|
+
message: `${input.config.provider} transcription model ${input.config.model} is not supported by AgentKit.`,
|
|
191
|
+
provider: input.config.provider,
|
|
192
|
+
model: input.config.model,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const maxBytes = Math.min(input.config.limits.maxBytes ?? adapter.maxBytes, adapter.maxBytes);
|
|
197
|
+
if (input.audio.byteLength > maxBytes) {
|
|
198
|
+
return {
|
|
199
|
+
ok: false,
|
|
200
|
+
retryable: false,
|
|
201
|
+
code: "transcription_audio_too_large",
|
|
202
|
+
message: `Audio file is ${input.audio.byteLength} bytes, which exceeds the configured ${maxBytes} byte limit.`,
|
|
203
|
+
provider: input.config.provider,
|
|
204
|
+
model: input.config.model,
|
|
205
|
+
};
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
if (
|
|
209
|
+
input.durationSeconds !== undefined &&
|
|
210
|
+
input.config.limits.maxDurationSeconds !== undefined &&
|
|
211
|
+
input.durationSeconds > input.config.limits.maxDurationSeconds
|
|
212
|
+
) {
|
|
213
|
+
return {
|
|
214
|
+
ok: false,
|
|
215
|
+
retryable: false,
|
|
216
|
+
code: "transcription_audio_too_long",
|
|
217
|
+
message: `Audio is ${input.durationSeconds}s, which exceeds the configured ${input.config.limits.maxDurationSeconds}s limit.`,
|
|
218
|
+
provider: input.config.provider,
|
|
219
|
+
model: input.config.model,
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
if (!isSupportedAudioFormat(adapter, input.filename, input.mimeType)) {
|
|
224
|
+
return {
|
|
225
|
+
ok: false,
|
|
226
|
+
retryable: false,
|
|
227
|
+
code: "transcription_audio_format_unsupported",
|
|
228
|
+
message: `${input.config.provider} does not support audio format ${input.mimeType ?? extensionFor(input.filename) ?? "unknown"}.`,
|
|
229
|
+
provider: input.config.provider,
|
|
230
|
+
model: input.config.model,
|
|
231
|
+
};
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
const secret = adapter.requiredSecret(input.config);
|
|
235
|
+
if (secret && !input.secrets[secret]) {
|
|
236
|
+
return {
|
|
237
|
+
ok: false,
|
|
238
|
+
retryable: false,
|
|
239
|
+
code: "transcription_secret_missing",
|
|
240
|
+
message: `Secret ${secret} is not set for audio transcription.`,
|
|
241
|
+
provider: input.config.provider,
|
|
242
|
+
model: input.config.model,
|
|
243
|
+
};
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
return {
|
|
247
|
+
ok: true,
|
|
248
|
+
text: "",
|
|
249
|
+
provider: input.config.provider,
|
|
250
|
+
model: input.config.model,
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
const testTranscriptionAdapter: TranscriptionAdapter = {
|
|
255
|
+
provider: "test",
|
|
256
|
+
models: ["fake"],
|
|
257
|
+
supportedMimeTypes: ["audio/wav", "audio/ogg", "audio/mpeg", "audio/webm", "application/octet-stream"],
|
|
258
|
+
supportedExtensions: ["wav", "ogg", "mp3", "webm", "bin"],
|
|
259
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
260
|
+
requiredSecret() {
|
|
261
|
+
return null;
|
|
262
|
+
},
|
|
263
|
+
async transcribe(input) {
|
|
264
|
+
const text = decodeFixtureTranscript(input.audio) ?? "fake audio transcript";
|
|
265
|
+
|
|
266
|
+
return {
|
|
267
|
+
ok: true,
|
|
268
|
+
text,
|
|
269
|
+
provider: "test",
|
|
270
|
+
model: input.config.model,
|
|
271
|
+
...(input.config.language ? { language: input.config.language } : {}),
|
|
272
|
+
...(input.durationSeconds !== undefined ? { durationSeconds: input.durationSeconds } : {}),
|
|
273
|
+
};
|
|
274
|
+
},
|
|
275
|
+
};
|
|
276
|
+
|
|
277
|
+
const openaiTranscriptionAdapter: TranscriptionAdapter = {
|
|
278
|
+
provider: "openai",
|
|
279
|
+
models: OPENAI_MODELS,
|
|
280
|
+
supportedMimeTypes: OPENAI_MIME_TYPES,
|
|
281
|
+
supportedExtensions: OPENAI_EXTENSIONS,
|
|
282
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
283
|
+
requiredSecret(config) {
|
|
284
|
+
return config.secret;
|
|
285
|
+
},
|
|
286
|
+
async transcribe(input, fetcher = fetch) {
|
|
287
|
+
return transcribeViaOpenAiCompatibleEndpoint({
|
|
288
|
+
input,
|
|
289
|
+
fetcher,
|
|
290
|
+
url: "https://api.openai.com/v1/audio/transcriptions",
|
|
291
|
+
apiKey: input.secrets[input.config.secret ?? ""],
|
|
292
|
+
provider: "openai",
|
|
293
|
+
});
|
|
294
|
+
},
|
|
295
|
+
};
|
|
296
|
+
|
|
297
|
+
const groqTranscriptionAdapter: TranscriptionAdapter = {
|
|
298
|
+
provider: "groq",
|
|
299
|
+
models: GROQ_MODELS,
|
|
300
|
+
supportedMimeTypes: GROQ_MIME_TYPES,
|
|
301
|
+
supportedExtensions: GROQ_EXTENSIONS,
|
|
302
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
303
|
+
requiredSecret(config) {
|
|
304
|
+
return config.secret;
|
|
305
|
+
},
|
|
306
|
+
async transcribe(input, fetcher = fetch) {
|
|
307
|
+
return transcribeViaOpenAiCompatibleEndpoint({
|
|
308
|
+
input,
|
|
309
|
+
fetcher,
|
|
310
|
+
url: "https://api.groq.com/openai/v1/audio/transcriptions",
|
|
311
|
+
apiKey: input.secrets[input.config.secret ?? ""],
|
|
312
|
+
provider: "groq",
|
|
313
|
+
});
|
|
314
|
+
},
|
|
315
|
+
};
|
|
316
|
+
|
|
317
|
+
async function transcribeViaOpenAiCompatibleEndpoint(input: {
|
|
318
|
+
input: TranscriptionInput;
|
|
319
|
+
fetcher: TranscriptionFetch;
|
|
320
|
+
url: string;
|
|
321
|
+
apiKey: string | undefined;
|
|
322
|
+
provider: TranscriptionProviderName;
|
|
323
|
+
}): Promise<TranscriptionResult> {
|
|
324
|
+
if (!input.apiKey) {
|
|
325
|
+
const secret = input.input.config.secret ?? defaultTranscriptionSecret(input.provider) ?? "TRANSCRIPTION_API_KEY";
|
|
326
|
+
|
|
327
|
+
return {
|
|
328
|
+
ok: false,
|
|
329
|
+
retryable: false,
|
|
330
|
+
code: "transcription_secret_missing",
|
|
331
|
+
message: `Secret ${secret} is not set for audio transcription.`,
|
|
332
|
+
provider: input.provider,
|
|
333
|
+
model: input.input.config.model,
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
const form = new FormData();
|
|
338
|
+
form.set("model", input.input.config.model);
|
|
339
|
+
form.set(
|
|
340
|
+
"file",
|
|
341
|
+
new Blob([arrayBufferForBlob(input.input.audio)], { type: input.input.mimeType ?? "application/octet-stream" }),
|
|
342
|
+
input.input.filename,
|
|
343
|
+
);
|
|
344
|
+
form.set("response_format", "json");
|
|
345
|
+
|
|
346
|
+
if (input.input.config.language) {
|
|
347
|
+
form.set("language", input.input.config.language);
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
if (input.input.config.prompt) {
|
|
351
|
+
form.set("prompt", input.input.config.prompt);
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
let response: Response;
|
|
355
|
+
|
|
356
|
+
try {
|
|
357
|
+
response = await input.fetcher(input.url, {
|
|
358
|
+
method: "POST",
|
|
359
|
+
headers: {
|
|
360
|
+
Authorization: `Bearer ${input.apiKey}`,
|
|
361
|
+
},
|
|
362
|
+
body: form,
|
|
363
|
+
});
|
|
364
|
+
} catch (error) {
|
|
365
|
+
return {
|
|
366
|
+
ok: false,
|
|
367
|
+
retryable: true,
|
|
368
|
+
code: "transcription_provider_unavailable",
|
|
369
|
+
message: `Transcription provider request failed before a response: ${redactSecret(
|
|
370
|
+
error instanceof Error ? error.message : String(error),
|
|
371
|
+
input.apiKey,
|
|
372
|
+
)}`,
|
|
373
|
+
provider: input.provider,
|
|
374
|
+
model: input.input.config.model,
|
|
375
|
+
};
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
const payload = await response.json().catch(() => null);
|
|
379
|
+
|
|
380
|
+
if (!response.ok) {
|
|
381
|
+
const message = readProviderError(payload) ?? `Transcription provider returned HTTP ${response.status}.`;
|
|
382
|
+
|
|
383
|
+
return {
|
|
384
|
+
ok: false,
|
|
385
|
+
retryable: response.status === 429 || response.status >= 500,
|
|
386
|
+
code: response.status === 429 || response.status >= 500 ? "transcription_provider_unavailable" : "transcription_failed",
|
|
387
|
+
message: redactSecret(message, input.apiKey),
|
|
388
|
+
provider: input.provider,
|
|
389
|
+
model: input.input.config.model,
|
|
390
|
+
providerMetadata: {
|
|
391
|
+
status: response.status,
|
|
392
|
+
},
|
|
393
|
+
};
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
const text = isRecord(payload) && typeof payload.text === "string" ? payload.text.trim() : "";
|
|
397
|
+
|
|
398
|
+
if (!text) {
|
|
399
|
+
return {
|
|
400
|
+
ok: false,
|
|
401
|
+
retryable: false,
|
|
402
|
+
code: "transcription_failed",
|
|
403
|
+
message: "Transcription provider returned an empty transcript.",
|
|
404
|
+
provider: input.provider,
|
|
405
|
+
model: input.input.config.model,
|
|
406
|
+
providerMetadata: {
|
|
407
|
+
status: response.status,
|
|
408
|
+
},
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
return {
|
|
413
|
+
ok: true,
|
|
414
|
+
text,
|
|
415
|
+
provider: input.provider,
|
|
416
|
+
model: input.input.config.model,
|
|
417
|
+
...(input.input.config.language ? { language: input.input.config.language } : {}),
|
|
418
|
+
...(input.input.durationSeconds !== undefined ? { durationSeconds: input.input.durationSeconds } : {}),
|
|
419
|
+
providerMetadata: {
|
|
420
|
+
status: response.status,
|
|
421
|
+
},
|
|
422
|
+
};
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
function isSupportedAudioFormat(adapter: TranscriptionAdapter, filename: string, mimeType?: string): boolean {
|
|
426
|
+
const normalizedMimeType = mimeType?.toLowerCase();
|
|
427
|
+
|
|
428
|
+
if (normalizedMimeType && adapter.supportedMimeTypes.includes(normalizedMimeType)) {
|
|
429
|
+
return true;
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
const extension = extensionFor(filename);
|
|
433
|
+
|
|
434
|
+
return Boolean(extension && adapter.supportedExtensions.includes(extension));
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
function extensionFor(filename: string): string | null {
|
|
438
|
+
const match = /\.([a-z0-9]+)$/i.exec(filename);
|
|
439
|
+
return match ? match[1].toLowerCase() : null;
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
function decodeFixtureTranscript(audio: Uint8Array): string | null {
|
|
443
|
+
try {
|
|
444
|
+
const text = new TextDecoder().decode(audio).trim();
|
|
445
|
+
return text.length > 0 && /^[\t\n\r -~\u00a0-\uffff]+$/.test(text) ? text : null;
|
|
446
|
+
} catch {
|
|
447
|
+
return null;
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
function arrayBufferForBlob(audio: Uint8Array): ArrayBuffer {
|
|
452
|
+
const copy = new Uint8Array(audio.byteLength);
|
|
453
|
+
copy.set(audio);
|
|
454
|
+
return copy.buffer;
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
function readProviderError(payload: unknown): string | null {
|
|
458
|
+
if (!isRecord(payload)) {
|
|
459
|
+
return null;
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
if (typeof payload.error === "string") {
|
|
463
|
+
return payload.error;
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
if (isRecord(payload.error) && typeof payload.error.message === "string") {
|
|
467
|
+
return payload.error.message;
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
if (typeof payload.message === "string") {
|
|
471
|
+
return payload.message;
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
return null;
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
function redactSecret(value: string, secret: string): string {
|
|
478
|
+
return secret ? value.replaceAll(secret, "<redacted>") : value;
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
482
|
+
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
|
483
|
+
}
|