@andreprado/agentkit 0.1.0-alpha.15 → 0.1.0-alpha.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -0
- package/docs/guides/add-channel.md +65 -0
- package/docs/guides/add-managed-composio.md +137 -0
- package/docs/guides/channel-security.md +32 -0
- package/docs/guides/channels-production-handoff.md +1 -1
- package/docs/guides/connect-telegram.md +60 -0
- package/docs/guides/connect-whatsapp-zapster.md +65 -0
- package/docs/guides/prepare-deploy.md +3 -1
- package/docs/llms-full.txt +13 -1
- package/docs/llms.txt +7 -0
- package/package.json +1 -1
- package/src/cli/commands/channels.ts +121 -7
- package/src/cli/commands/transcribe.ts +171 -0
- package/src/cli/deploy-readiness.ts +59 -3
- package/src/cli/help.ts +15 -0
- package/src/cli/index.ts +130 -0
- package/src/create-project.ts +4 -32
- package/src/index.ts +231 -0
- package/src/runtime/channel-test-harness.ts +2 -0
- package/src/runtime/channels/telegram.ts +326 -10
- package/src/runtime/channels/whatsapp-zapster.ts +319 -0
- package/src/runtime/channels.ts +47 -1
- package/src/runtime/chat.ts +2 -1
- package/src/runtime/config.ts +214 -4
- package/src/runtime/core/manifest.ts +61 -2
- package/src/runtime/deploy-readiness.ts +31 -1
- package/src/runtime/dev-server.ts +72 -4
- package/src/runtime/inspect.ts +142 -4
- package/src/runtime/integrations/composio.ts +257 -0
- package/src/runtime/skills.ts +95 -0
- package/src/runtime/targets/cloudflare/build.ts +162 -5
- package/src/runtime/tool-runner.ts +2 -1
- package/src/runtime/transcription.ts +483 -0
- package/src/templates/skills/agentkit-capsule/SKILL.md +1 -0
- package/src/templates/skills/agentkit-channels/SKILL.md +36 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +16 -1
- package/src/templates/skills/agentkit-channels/references/telegram.md +34 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +29 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +1 -1
- package/src/templates/skills/agentkit-integrations/SKILL.md +66 -0
- package/src/templates/skills/agentkit-troubleshooting/SKILL.md +3 -2
|
@@ -0,0 +1,483 @@
|
|
|
1
|
+
import type { AgentTranscriptionConfig, TranscriptionProviderName } from "../index";
|
|
2
|
+
|
|
3
|
+
export type ResolvedTranscriptionConfig = {
|
|
4
|
+
enabled: boolean;
|
|
5
|
+
provider: TranscriptionProviderName;
|
|
6
|
+
model: string;
|
|
7
|
+
secret: string | null;
|
|
8
|
+
language: string | null;
|
|
9
|
+
prompt: string | null;
|
|
10
|
+
limits: {
|
|
11
|
+
maxDurationSeconds?: number;
|
|
12
|
+
maxBytes?: number;
|
|
13
|
+
};
|
|
14
|
+
rawAudioTtlSeconds: number | null;
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
export type TranscriptionInput = {
|
|
18
|
+
config: ResolvedTranscriptionConfig;
|
|
19
|
+
secrets: Record<string, string>;
|
|
20
|
+
audio: Uint8Array;
|
|
21
|
+
filename: string;
|
|
22
|
+
mimeType?: string;
|
|
23
|
+
durationSeconds?: number;
|
|
24
|
+
providerMetadata?: Record<string, unknown>;
|
|
25
|
+
};
|
|
26
|
+
|
|
27
|
+
export type TranscriptionFetch = (input: RequestInfo | URL, init?: RequestInit) => Promise<Response>;
|
|
28
|
+
|
|
29
|
+
export type TranscriptionResult =
|
|
30
|
+
| {
|
|
31
|
+
ok: true;
|
|
32
|
+
text: string;
|
|
33
|
+
provider: TranscriptionProviderName;
|
|
34
|
+
model: string;
|
|
35
|
+
language?: string;
|
|
36
|
+
durationSeconds?: number;
|
|
37
|
+
providerMetadata?: Record<string, unknown>;
|
|
38
|
+
}
|
|
39
|
+
| {
|
|
40
|
+
ok: false;
|
|
41
|
+
retryable: boolean;
|
|
42
|
+
code:
|
|
43
|
+
| "transcription_disabled"
|
|
44
|
+
| "transcription_secret_missing"
|
|
45
|
+
| "transcription_model_unsupported"
|
|
46
|
+
| "transcription_audio_too_large"
|
|
47
|
+
| "transcription_audio_too_long"
|
|
48
|
+
| "transcription_audio_format_unsupported"
|
|
49
|
+
| "transcription_provider_unavailable"
|
|
50
|
+
| "transcription_failed";
|
|
51
|
+
message: string;
|
|
52
|
+
provider?: TranscriptionProviderName;
|
|
53
|
+
model?: string;
|
|
54
|
+
providerMetadata?: Record<string, unknown>;
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
export type TranscriptionAdapter = {
|
|
58
|
+
provider: TranscriptionProviderName;
|
|
59
|
+
models: string[];
|
|
60
|
+
supportedMimeTypes: string[];
|
|
61
|
+
supportedExtensions: string[];
|
|
62
|
+
maxBytes: number;
|
|
63
|
+
requiredSecret(config: ResolvedTranscriptionConfig): string | null;
|
|
64
|
+
transcribe(input: TranscriptionInput, fetcher?: TranscriptionFetch): Promise<TranscriptionResult>;
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
const DEFAULT_MAX_BYTES = 25_000_000;
|
|
68
|
+
const OPENAI_MODELS = ["gpt-4o-mini-transcribe", "gpt-4o-transcribe", "whisper-1"];
|
|
69
|
+
const GROQ_MODELS = ["whisper-large-v3-turbo", "whisper-large-v3", "distil-whisper-large-v3-en"];
|
|
70
|
+
const OPENAI_EXTENSIONS = ["mp3", "mp4", "mpeg", "mpga", "m4a", "wav", "webm"];
|
|
71
|
+
const GROQ_EXTENSIONS = ["flac", "mp3", "mp4", "mpeg", "mpga", "m4a", "ogg", "wav", "webm"];
|
|
72
|
+
const OPENAI_MIME_TYPES = [
|
|
73
|
+
"audio/mpeg",
|
|
74
|
+
"audio/mp3",
|
|
75
|
+
"audio/mp4",
|
|
76
|
+
"audio/mpga",
|
|
77
|
+
"audio/m4a",
|
|
78
|
+
"audio/wav",
|
|
79
|
+
"audio/webm",
|
|
80
|
+
"video/mp4",
|
|
81
|
+
];
|
|
82
|
+
const GROQ_MIME_TYPES = [
|
|
83
|
+
...OPENAI_MIME_TYPES,
|
|
84
|
+
"audio/flac",
|
|
85
|
+
"audio/ogg",
|
|
86
|
+
"audio/opus",
|
|
87
|
+
"application/ogg",
|
|
88
|
+
];
|
|
89
|
+
|
|
90
|
+
export function resolveTranscriptionConfig(config: AgentTranscriptionConfig | undefined): ResolvedTranscriptionConfig {
|
|
91
|
+
if (!config) {
|
|
92
|
+
return {
|
|
93
|
+
enabled: false,
|
|
94
|
+
provider: "test",
|
|
95
|
+
model: "fake",
|
|
96
|
+
secret: null,
|
|
97
|
+
language: null,
|
|
98
|
+
prompt: null,
|
|
99
|
+
limits: {},
|
|
100
|
+
rawAudioTtlSeconds: null,
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
return {
|
|
105
|
+
enabled: true,
|
|
106
|
+
provider: config.provider,
|
|
107
|
+
model: config.model,
|
|
108
|
+
secret: config.secret ?? defaultTranscriptionSecret(config.provider),
|
|
109
|
+
language: config.language ?? null,
|
|
110
|
+
prompt: config.prompt ?? null,
|
|
111
|
+
limits: {
|
|
112
|
+
...(config.limits?.maxDurationSeconds !== undefined
|
|
113
|
+
? { maxDurationSeconds: config.limits.maxDurationSeconds }
|
|
114
|
+
: {}),
|
|
115
|
+
...(config.limits?.maxBytes !== undefined ? { maxBytes: config.limits.maxBytes } : {}),
|
|
116
|
+
},
|
|
117
|
+
rawAudioTtlSeconds: config.rawAudioTtlSeconds ?? 3600,
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export async function transcribeAudio(
|
|
122
|
+
input: TranscriptionInput,
|
|
123
|
+
fetcher: TranscriptionFetch = fetch,
|
|
124
|
+
): Promise<TranscriptionResult> {
|
|
125
|
+
if (!input.config.enabled) {
|
|
126
|
+
return {
|
|
127
|
+
ok: false,
|
|
128
|
+
retryable: false,
|
|
129
|
+
code: "transcription_disabled",
|
|
130
|
+
message: "Audio transcription is not configured for this Agent Capsule.",
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
const adapter = transcriptionAdapterFor(input.config.provider);
|
|
135
|
+
|
|
136
|
+
if (!adapter) {
|
|
137
|
+
return {
|
|
138
|
+
ok: false,
|
|
139
|
+
retryable: false,
|
|
140
|
+
code: "transcription_model_unsupported",
|
|
141
|
+
message: `No transcription adapter is registered for ${input.config.provider}.`,
|
|
142
|
+
provider: input.config.provider,
|
|
143
|
+
model: input.config.model,
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
const validation = validateTranscriptionInput(adapter, input);
|
|
148
|
+
|
|
149
|
+
if (!validation.ok) {
|
|
150
|
+
return validation;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
return adapter.transcribe(input, fetcher);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
export function transcriptionAdapterFor(provider: TranscriptionProviderName): TranscriptionAdapter | null {
|
|
157
|
+
if (provider === "test") {
|
|
158
|
+
return testTranscriptionAdapter;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
if (provider === "openai") {
|
|
162
|
+
return openaiTranscriptionAdapter;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
if (provider === "groq") {
|
|
166
|
+
return groqTranscriptionAdapter;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
return null;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
export function defaultTranscriptionSecret(provider: TranscriptionProviderName): string | null {
|
|
173
|
+
if (provider === "openai") {
|
|
174
|
+
return "OPENAI_API_KEY";
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
if (provider === "groq") {
|
|
178
|
+
return "GROQ_API_KEY";
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
return null;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
function validateTranscriptionInput(adapter: TranscriptionAdapter, input: TranscriptionInput): TranscriptionResult {
|
|
185
|
+
if (!adapter.models.includes(input.config.model)) {
|
|
186
|
+
return {
|
|
187
|
+
ok: false,
|
|
188
|
+
retryable: false,
|
|
189
|
+
code: "transcription_model_unsupported",
|
|
190
|
+
message: `${input.config.provider} transcription model ${input.config.model} is not supported by AgentKit.`,
|
|
191
|
+
provider: input.config.provider,
|
|
192
|
+
model: input.config.model,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const maxBytes = Math.min(input.config.limits.maxBytes ?? adapter.maxBytes, adapter.maxBytes);
|
|
197
|
+
if (input.audio.byteLength > maxBytes) {
|
|
198
|
+
return {
|
|
199
|
+
ok: false,
|
|
200
|
+
retryable: false,
|
|
201
|
+
code: "transcription_audio_too_large",
|
|
202
|
+
message: `Audio file is ${input.audio.byteLength} bytes, which exceeds the configured ${maxBytes} byte limit.`,
|
|
203
|
+
provider: input.config.provider,
|
|
204
|
+
model: input.config.model,
|
|
205
|
+
};
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
if (
|
|
209
|
+
input.durationSeconds !== undefined &&
|
|
210
|
+
input.config.limits.maxDurationSeconds !== undefined &&
|
|
211
|
+
input.durationSeconds > input.config.limits.maxDurationSeconds
|
|
212
|
+
) {
|
|
213
|
+
return {
|
|
214
|
+
ok: false,
|
|
215
|
+
retryable: false,
|
|
216
|
+
code: "transcription_audio_too_long",
|
|
217
|
+
message: `Audio is ${input.durationSeconds}s, which exceeds the configured ${input.config.limits.maxDurationSeconds}s limit.`,
|
|
218
|
+
provider: input.config.provider,
|
|
219
|
+
model: input.config.model,
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
if (!isSupportedAudioFormat(adapter, input.filename, input.mimeType)) {
|
|
224
|
+
return {
|
|
225
|
+
ok: false,
|
|
226
|
+
retryable: false,
|
|
227
|
+
code: "transcription_audio_format_unsupported",
|
|
228
|
+
message: `${input.config.provider} does not support audio format ${input.mimeType ?? extensionFor(input.filename) ?? "unknown"}.`,
|
|
229
|
+
provider: input.config.provider,
|
|
230
|
+
model: input.config.model,
|
|
231
|
+
};
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
const secret = adapter.requiredSecret(input.config);
|
|
235
|
+
if (secret && !input.secrets[secret]) {
|
|
236
|
+
return {
|
|
237
|
+
ok: false,
|
|
238
|
+
retryable: false,
|
|
239
|
+
code: "transcription_secret_missing",
|
|
240
|
+
message: `Secret ${secret} is not set for audio transcription.`,
|
|
241
|
+
provider: input.config.provider,
|
|
242
|
+
model: input.config.model,
|
|
243
|
+
};
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
return {
|
|
247
|
+
ok: true,
|
|
248
|
+
text: "",
|
|
249
|
+
provider: input.config.provider,
|
|
250
|
+
model: input.config.model,
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
const testTranscriptionAdapter: TranscriptionAdapter = {
|
|
255
|
+
provider: "test",
|
|
256
|
+
models: ["fake"],
|
|
257
|
+
supportedMimeTypes: ["audio/wav", "audio/ogg", "audio/mpeg", "audio/webm", "application/octet-stream"],
|
|
258
|
+
supportedExtensions: ["wav", "ogg", "mp3", "webm", "bin"],
|
|
259
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
260
|
+
requiredSecret() {
|
|
261
|
+
return null;
|
|
262
|
+
},
|
|
263
|
+
async transcribe(input) {
|
|
264
|
+
const text = decodeFixtureTranscript(input.audio) ?? "fake audio transcript";
|
|
265
|
+
|
|
266
|
+
return {
|
|
267
|
+
ok: true,
|
|
268
|
+
text,
|
|
269
|
+
provider: "test",
|
|
270
|
+
model: input.config.model,
|
|
271
|
+
...(input.config.language ? { language: input.config.language } : {}),
|
|
272
|
+
...(input.durationSeconds !== undefined ? { durationSeconds: input.durationSeconds } : {}),
|
|
273
|
+
};
|
|
274
|
+
},
|
|
275
|
+
};
|
|
276
|
+
|
|
277
|
+
const openaiTranscriptionAdapter: TranscriptionAdapter = {
|
|
278
|
+
provider: "openai",
|
|
279
|
+
models: OPENAI_MODELS,
|
|
280
|
+
supportedMimeTypes: OPENAI_MIME_TYPES,
|
|
281
|
+
supportedExtensions: OPENAI_EXTENSIONS,
|
|
282
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
283
|
+
requiredSecret(config) {
|
|
284
|
+
return config.secret;
|
|
285
|
+
},
|
|
286
|
+
async transcribe(input, fetcher = fetch) {
|
|
287
|
+
return transcribeViaOpenAiCompatibleEndpoint({
|
|
288
|
+
input,
|
|
289
|
+
fetcher,
|
|
290
|
+
url: "https://api.openai.com/v1/audio/transcriptions",
|
|
291
|
+
apiKey: input.secrets[input.config.secret ?? ""],
|
|
292
|
+
provider: "openai",
|
|
293
|
+
});
|
|
294
|
+
},
|
|
295
|
+
};
|
|
296
|
+
|
|
297
|
+
const groqTranscriptionAdapter: TranscriptionAdapter = {
|
|
298
|
+
provider: "groq",
|
|
299
|
+
models: GROQ_MODELS,
|
|
300
|
+
supportedMimeTypes: GROQ_MIME_TYPES,
|
|
301
|
+
supportedExtensions: GROQ_EXTENSIONS,
|
|
302
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
303
|
+
requiredSecret(config) {
|
|
304
|
+
return config.secret;
|
|
305
|
+
},
|
|
306
|
+
async transcribe(input, fetcher = fetch) {
|
|
307
|
+
return transcribeViaOpenAiCompatibleEndpoint({
|
|
308
|
+
input,
|
|
309
|
+
fetcher,
|
|
310
|
+
url: "https://api.groq.com/openai/v1/audio/transcriptions",
|
|
311
|
+
apiKey: input.secrets[input.config.secret ?? ""],
|
|
312
|
+
provider: "groq",
|
|
313
|
+
});
|
|
314
|
+
},
|
|
315
|
+
};
|
|
316
|
+
|
|
317
|
+
async function transcribeViaOpenAiCompatibleEndpoint(input: {
|
|
318
|
+
input: TranscriptionInput;
|
|
319
|
+
fetcher: TranscriptionFetch;
|
|
320
|
+
url: string;
|
|
321
|
+
apiKey: string | undefined;
|
|
322
|
+
provider: TranscriptionProviderName;
|
|
323
|
+
}): Promise<TranscriptionResult> {
|
|
324
|
+
if (!input.apiKey) {
|
|
325
|
+
const secret = input.input.config.secret ?? defaultTranscriptionSecret(input.provider) ?? "TRANSCRIPTION_API_KEY";
|
|
326
|
+
|
|
327
|
+
return {
|
|
328
|
+
ok: false,
|
|
329
|
+
retryable: false,
|
|
330
|
+
code: "transcription_secret_missing",
|
|
331
|
+
message: `Secret ${secret} is not set for audio transcription.`,
|
|
332
|
+
provider: input.provider,
|
|
333
|
+
model: input.input.config.model,
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
const form = new FormData();
|
|
338
|
+
form.set("model", input.input.config.model);
|
|
339
|
+
form.set(
|
|
340
|
+
"file",
|
|
341
|
+
new Blob([arrayBufferForBlob(input.input.audio)], { type: input.input.mimeType ?? "application/octet-stream" }),
|
|
342
|
+
input.input.filename,
|
|
343
|
+
);
|
|
344
|
+
form.set("response_format", "json");
|
|
345
|
+
|
|
346
|
+
if (input.input.config.language) {
|
|
347
|
+
form.set("language", input.input.config.language);
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
if (input.input.config.prompt) {
|
|
351
|
+
form.set("prompt", input.input.config.prompt);
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
let response: Response;
|
|
355
|
+
|
|
356
|
+
try {
|
|
357
|
+
response = await input.fetcher(input.url, {
|
|
358
|
+
method: "POST",
|
|
359
|
+
headers: {
|
|
360
|
+
Authorization: `Bearer ${input.apiKey}`,
|
|
361
|
+
},
|
|
362
|
+
body: form,
|
|
363
|
+
});
|
|
364
|
+
} catch (error) {
|
|
365
|
+
return {
|
|
366
|
+
ok: false,
|
|
367
|
+
retryable: true,
|
|
368
|
+
code: "transcription_provider_unavailable",
|
|
369
|
+
message: `Transcription provider request failed before a response: ${redactSecret(
|
|
370
|
+
error instanceof Error ? error.message : String(error),
|
|
371
|
+
input.apiKey,
|
|
372
|
+
)}`,
|
|
373
|
+
provider: input.provider,
|
|
374
|
+
model: input.input.config.model,
|
|
375
|
+
};
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
const payload = await response.json().catch(() => null);
|
|
379
|
+
|
|
380
|
+
if (!response.ok) {
|
|
381
|
+
const message = readProviderError(payload) ?? `Transcription provider returned HTTP ${response.status}.`;
|
|
382
|
+
|
|
383
|
+
return {
|
|
384
|
+
ok: false,
|
|
385
|
+
retryable: response.status === 429 || response.status >= 500,
|
|
386
|
+
code: response.status === 429 || response.status >= 500 ? "transcription_provider_unavailable" : "transcription_failed",
|
|
387
|
+
message: redactSecret(message, input.apiKey),
|
|
388
|
+
provider: input.provider,
|
|
389
|
+
model: input.input.config.model,
|
|
390
|
+
providerMetadata: {
|
|
391
|
+
status: response.status,
|
|
392
|
+
},
|
|
393
|
+
};
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
const text = isRecord(payload) && typeof payload.text === "string" ? payload.text.trim() : "";
|
|
397
|
+
|
|
398
|
+
if (!text) {
|
|
399
|
+
return {
|
|
400
|
+
ok: false,
|
|
401
|
+
retryable: false,
|
|
402
|
+
code: "transcription_failed",
|
|
403
|
+
message: "Transcription provider returned an empty transcript.",
|
|
404
|
+
provider: input.provider,
|
|
405
|
+
model: input.input.config.model,
|
|
406
|
+
providerMetadata: {
|
|
407
|
+
status: response.status,
|
|
408
|
+
},
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
return {
|
|
413
|
+
ok: true,
|
|
414
|
+
text,
|
|
415
|
+
provider: input.provider,
|
|
416
|
+
model: input.input.config.model,
|
|
417
|
+
...(input.input.config.language ? { language: input.input.config.language } : {}),
|
|
418
|
+
...(input.input.durationSeconds !== undefined ? { durationSeconds: input.input.durationSeconds } : {}),
|
|
419
|
+
providerMetadata: {
|
|
420
|
+
status: response.status,
|
|
421
|
+
},
|
|
422
|
+
};
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
function isSupportedAudioFormat(adapter: TranscriptionAdapter, filename: string, mimeType?: string): boolean {
|
|
426
|
+
const normalizedMimeType = mimeType?.toLowerCase();
|
|
427
|
+
|
|
428
|
+
if (normalizedMimeType && adapter.supportedMimeTypes.includes(normalizedMimeType)) {
|
|
429
|
+
return true;
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
const extension = extensionFor(filename);
|
|
433
|
+
|
|
434
|
+
return Boolean(extension && adapter.supportedExtensions.includes(extension));
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
function extensionFor(filename: string): string | null {
|
|
438
|
+
const match = /\.([a-z0-9]+)$/i.exec(filename);
|
|
439
|
+
return match ? match[1].toLowerCase() : null;
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
function decodeFixtureTranscript(audio: Uint8Array): string | null {
|
|
443
|
+
try {
|
|
444
|
+
const text = new TextDecoder().decode(audio).trim();
|
|
445
|
+
return text.length > 0 && /^[\t\n\r -~\u00a0-\uffff]+$/.test(text) ? text : null;
|
|
446
|
+
} catch {
|
|
447
|
+
return null;
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
function arrayBufferForBlob(audio: Uint8Array): ArrayBuffer {
|
|
452
|
+
const copy = new Uint8Array(audio.byteLength);
|
|
453
|
+
copy.set(audio);
|
|
454
|
+
return copy.buffer;
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
function readProviderError(payload: unknown): string | null {
|
|
458
|
+
if (!isRecord(payload)) {
|
|
459
|
+
return null;
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
if (typeof payload.error === "string") {
|
|
463
|
+
return payload.error;
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
if (isRecord(payload.error) && typeof payload.error.message === "string") {
|
|
467
|
+
return payload.error.message;
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
if (typeof payload.message === "string") {
|
|
471
|
+
return payload.message;
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
return null;
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
function redactSecret(value: string, secret: string): string {
|
|
478
|
+
return secret ? value.replaceAll(secret, "<redacted>") : value;
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
482
|
+
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
|
483
|
+
}
|
|
@@ -19,6 +19,7 @@ Use this first inside an AgentKit Agent Capsule.
|
|
|
19
19
|
- Build or reshape the agent from the owner's brief: `skills/agentkit-build-agent/SKILL.md`
|
|
20
20
|
- Edit prompts: `skills/agentkit-prompts/SKILL.md`
|
|
21
21
|
- Add actions or external data: `skills/agentkit-tools/SKILL.md`
|
|
22
|
+
- Add AgentKit-managed integrations such as managed Composio: `skills/agentkit-integrations/SKILL.md`
|
|
22
23
|
- Add database tables or database-backed tools: `skills/agentkit-database/SKILL.md`
|
|
23
24
|
- Add docs, FAQs, prices, policies, or CSV facts: `skills/agentkit-knowledge/SKILL.md`
|
|
24
25
|
- Switch from `test/fake` to a real model provider: `skills/agentkit-provider/SKILL.md`
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: agentkit-channels
|
|
3
|
-
description: Use when adding, connecting, testing, buffering, or debugging AgentKit website, Telegram, or WhatsApp channels, including channel config helpers, provider secrets, webhook setup, channel tests, delivery logs,
|
|
3
|
+
description: Use when adding, connecting, testing, buffering, transcribing audio, or debugging AgentKit website, Telegram, or WhatsApp channels, including channel config helpers, provider secrets, webhook setup, channel tests, delivery logs, burst-message buffers, and transcription provider secrets.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# AgentKit Channels
|
|
@@ -16,6 +16,39 @@ Channels receive user messages. Tools let the agent call external systems. Keep
|
|
|
16
16
|
5. Connect channel resources through the CLI.
|
|
17
17
|
6. Test, doctor, and inspect delivery logs.
|
|
18
18
|
|
|
19
|
+
## Audio Transcription
|
|
20
|
+
|
|
21
|
+
Enable transcription at the agent level and opt in per channel with `audio.mode: "transcribe"`.
|
|
22
|
+
|
|
23
|
+
```ts
|
|
24
|
+
export default defineAgent({
|
|
25
|
+
// ...
|
|
26
|
+
transcription: {
|
|
27
|
+
provider: "groq",
|
|
28
|
+
model: "whisper-large-v3-turbo",
|
|
29
|
+
secret: "GROQ_API_KEY",
|
|
30
|
+
language: "pt",
|
|
31
|
+
limits: {
|
|
32
|
+
maxDurationSeconds: 180,
|
|
33
|
+
maxBytes: 20_000_000,
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
channels: [
|
|
37
|
+
telegramChannel({
|
|
38
|
+
name: "support-telegram",
|
|
39
|
+
audio: { mode: "transcribe" },
|
|
40
|
+
}),
|
|
41
|
+
],
|
|
42
|
+
});
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
V1 providers:
|
|
46
|
+
|
|
47
|
+
- `openai`: `gpt-4o-mini-transcribe`, `gpt-4o-transcribe`, `whisper-1`; default secret `OPENAI_API_KEY`.
|
|
48
|
+
- `groq`: `whisper-large-v3-turbo`, `whisper-large-v3`, `distil-whisper-large-v3-en`; default secret `GROQ_API_KEY`.
|
|
49
|
+
|
|
50
|
+
Telegram voice notes are usually OGG/Opus, so use Groq for the default Telegram voice-note path in V1. Zapster audio needs a usable HTTPS Zapster media download URL in the webhook payload; arbitrary hosts are rejected before bearer auth is sent. Hosted channel creation requires the transcription secret automatically when the channel enables transcription. Webhooks only enqueue audio jobs; download and transcription run in the retryable channel worker before the agent run.
|
|
51
|
+
|
|
19
52
|
## Buffering
|
|
20
53
|
|
|
21
54
|
Enable `buffer.mode: "debounce"` when clients send several short messages in a row and the agent should answer once.
|
|
@@ -47,6 +80,8 @@ npm run agentkit -- channels connect telegram support-telegram
|
|
|
47
80
|
npm run agentkit -- channels add whatsapp support-whatsapp --provider zapster
|
|
48
81
|
npm run agentkit -- channels doctor support-telegram
|
|
49
82
|
npm run agentkit -- channels test support-telegram --message "hello"
|
|
83
|
+
npm run agentkit -- channels test-audio support-telegram --fixture voice-note
|
|
84
|
+
npm run agentkit -- transcribe smoke --provider groq
|
|
50
85
|
npm run agentkit -- channels deliveries list support-telegram
|
|
51
86
|
```
|
|
52
87
|
|
|
@@ -7,6 +7,8 @@ agentkit channels list
|
|
|
7
7
|
agentkit channels status <name>
|
|
8
8
|
agentkit channels doctor <name>
|
|
9
9
|
agentkit channels test <name> --message "hello"
|
|
10
|
+
agentkit channels test-audio <name> --fixture voice-note
|
|
11
|
+
agentkit transcribe smoke --provider groq
|
|
10
12
|
agentkit channels deliveries list <name> --since 24h
|
|
11
13
|
agentkit channels deliveries show <delivery-id>
|
|
12
14
|
agentkit channels buffers list <name>
|
|
@@ -22,6 +24,10 @@ Common states:
|
|
|
22
24
|
webhook_received
|
|
23
25
|
validated
|
|
24
26
|
duplicate
|
|
27
|
+
audio_received
|
|
28
|
+
audio_downloaded
|
|
29
|
+
transcribing
|
|
30
|
+
transcribed
|
|
25
31
|
buffered
|
|
26
32
|
queued
|
|
27
33
|
running
|
|
@@ -38,11 +44,20 @@ skipped
|
|
|
38
44
|
|
|
39
45
|
Common errors:
|
|
40
46
|
|
|
41
|
-
- `channel_not_found`: webhook URL points to an unknown channel.
|
|
47
|
+
- `channel_not_found`: webhook URL points to an unknown channel. `channels test` is the official synthetic smoke and should resolve the current `.agentkit/deploy.json` channel.
|
|
42
48
|
- `channel_secret_missing`: required hosted secret is not set.
|
|
43
49
|
- `channel_signature_invalid`: webhook secret, token, or origin header mismatch.
|
|
44
50
|
- `channel_payload_invalid`: malformed or unsupported provider payload.
|
|
45
51
|
- `channel_event_duplicate`: provider retry; do not create a second run.
|
|
52
|
+
- `audio_received`: audio message was accepted and normalized.
|
|
53
|
+
- `audio_downloaded`: retryable channel worker downloaded provider media into memory.
|
|
54
|
+
- `transcribing`: AgentKit is calling the configured transcription provider.
|
|
55
|
+
- `transcribed`: transcript text was queued for the agent.
|
|
56
|
+
- `channel_audio_download_unavailable`: provider audio payload did not include a usable download URL, or Zapster sent a non-HTTPS/non-Zapster media host.
|
|
57
|
+
- `transcription_secret_missing`: managed transcription secret is missing.
|
|
58
|
+
- `transcription_audio_too_large` or `transcription_audio_too_long`: audio exceeded configured limits.
|
|
59
|
+
- `transcription_audio_format_unsupported`: provider does not accept this audio MIME type or extension.
|
|
60
|
+
- `transcription_provider_unavailable`: retryable transcription provider failure.
|
|
46
61
|
- `channel_limit_exceeded`: backpressure skipped the message.
|
|
47
62
|
- `synthetic_expected_failure`: a synthetic test reached AgentKit, but the provider correctly rejected a fake test recipient.
|
|
48
63
|
- `buffered` delivery state: message is waiting for the channel quiet window or max wait before one coalesced agent run is queued.
|
|
@@ -7,6 +7,12 @@ TELEGRAM_BOT_TOKEN
|
|
|
7
7
|
TELEGRAM_WEBHOOK_SECRET
|
|
8
8
|
```
|
|
9
9
|
|
|
10
|
+
Audio transcription also needs the configured transcription secret, usually:
|
|
11
|
+
|
|
12
|
+
```txt
|
|
13
|
+
GROQ_API_KEY
|
|
14
|
+
```
|
|
15
|
+
|
|
10
16
|
Commands:
|
|
11
17
|
|
|
12
18
|
```sh
|
|
@@ -17,6 +23,8 @@ agentkit channels connect telegram support-telegram
|
|
|
17
23
|
agentkit channels doctor support-telegram
|
|
18
24
|
agentkit channels status support-telegram
|
|
19
25
|
agentkit channels test support-telegram --message "hello"
|
|
26
|
+
agentkit channels test-audio support-telegram --fixture voice-note
|
|
27
|
+
agentkit transcribe smoke --provider groq
|
|
20
28
|
agentkit channels deliveries list support-telegram
|
|
21
29
|
```
|
|
22
30
|
|
|
@@ -36,3 +44,29 @@ telegramChannel({
|
|
|
36
44
|
},
|
|
37
45
|
})
|
|
38
46
|
```
|
|
47
|
+
|
|
48
|
+
Transcribe Telegram voice notes:
|
|
49
|
+
|
|
50
|
+
```ts
|
|
51
|
+
export default defineAgent({
|
|
52
|
+
// ...
|
|
53
|
+
transcription: {
|
|
54
|
+
provider: "groq",
|
|
55
|
+
model: "whisper-large-v3-turbo",
|
|
56
|
+
secret: "GROQ_API_KEY",
|
|
57
|
+
language: "pt",
|
|
58
|
+
limits: {
|
|
59
|
+
maxDurationSeconds: 180,
|
|
60
|
+
maxBytes: 20_000_000,
|
|
61
|
+
},
|
|
62
|
+
},
|
|
63
|
+
channels: [
|
|
64
|
+
telegramChannel({
|
|
65
|
+
name: "support-telegram",
|
|
66
|
+
audio: { mode: "transcribe" },
|
|
67
|
+
}),
|
|
68
|
+
],
|
|
69
|
+
});
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
AgentKit validates the Telegram webhook, normalizes `voice` and `audio` payloads, enqueues an audio job, then the retryable channel worker calls Telegram `getFile`, downloads the media with `TELEGRAM_BOT_TOKEN`, sends the bytes to the configured transcription provider, and runs the agent with transcript text. Telegram voice notes are usually OGG/Opus; use Groq in V1 for that path. `test-audio` validates the audio channel ingress path; `transcribe smoke` validates the transcription provider separately.
|
|
@@ -14,6 +14,8 @@ Optional hardening secret:
|
|
|
14
14
|
ZAPSTER_WEBHOOK_TOKEN
|
|
15
15
|
```
|
|
16
16
|
|
|
17
|
+
Audio transcription also needs the configured transcription secret, usually `OPENAI_API_KEY` or `GROQ_API_KEY`.
|
|
18
|
+
|
|
17
19
|
Commands:
|
|
18
20
|
|
|
19
21
|
```sh
|
|
@@ -46,3 +48,30 @@ whatsappChannel({
|
|
|
46
48
|
},
|
|
47
49
|
})
|
|
48
50
|
```
|
|
51
|
+
|
|
52
|
+
Transcribe WhatsApp audio:
|
|
53
|
+
|
|
54
|
+
```ts
|
|
55
|
+
export default defineAgent({
|
|
56
|
+
// ...
|
|
57
|
+
transcription: {
|
|
58
|
+
provider: "openai",
|
|
59
|
+
model: "gpt-4o-mini-transcribe",
|
|
60
|
+
secret: "OPENAI_API_KEY",
|
|
61
|
+
language: "pt",
|
|
62
|
+
limits: {
|
|
63
|
+
maxDurationSeconds: 180,
|
|
64
|
+
maxBytes: 20_000_000,
|
|
65
|
+
},
|
|
66
|
+
},
|
|
67
|
+
channels: [
|
|
68
|
+
whatsappChannel({
|
|
69
|
+
name: "support-whatsapp",
|
|
70
|
+
provider: "zapster",
|
|
71
|
+
audio: { mode: "transcribe" },
|
|
72
|
+
}),
|
|
73
|
+
],
|
|
74
|
+
});
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Zapster audio payloads must include a usable HTTPS Zapster media download URL such as `audio.downloadUrl`, `audio.url`, `audio.mediaUrl`, or the snake_case equivalents. AgentKit rejects arbitrary hosts before sending `ZAPSTER_API_KEY`. The retryable channel worker downloads the media, transcribes it through the configured provider secret, and runs the agent with transcript text. If Zapster sends only a media ID in V1, AgentKit records `channel_audio_download_unavailable`.
|
|
@@ -18,6 +18,7 @@ Use this when the owner asks to prepare, test, or run hosted deploy.
|
|
|
18
18
|
|
|
19
19
|
```sh
|
|
20
20
|
npm run typecheck
|
|
21
|
+
npm run agentkit -- skills status
|
|
21
22
|
npm run agentkit -- inspect
|
|
22
23
|
npm run agentkit -- db migrate
|
|
23
24
|
npm run chat -- --message "hello"
|
|
@@ -41,4 +42,3 @@ Open the printed `Chat:` URL and report it to the owner.
|
|
|
41
42
|
## Production Handoff
|
|
42
43
|
|
|
43
44
|
Report changed files, required env/secret names, database schema changes, deploy order, smoke checks, rollback concerns, and whether the provider was still `test/fake`.
|
|
44
|
-
|