@andreprado/agentkit 0.1.0-alpha.15 → 0.1.0-alpha.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +8 -0
  2. package/docs/guides/add-channel.md +65 -0
  3. package/docs/guides/add-managed-composio.md +137 -0
  4. package/docs/guides/channel-security.md +32 -0
  5. package/docs/guides/channels-production-handoff.md +1 -1
  6. package/docs/guides/connect-telegram.md +60 -0
  7. package/docs/guides/connect-whatsapp-zapster.md +65 -0
  8. package/docs/guides/prepare-deploy.md +3 -1
  9. package/docs/llms-full.txt +13 -1
  10. package/docs/llms.txt +7 -0
  11. package/package.json +1 -1
  12. package/src/cli/commands/channels.ts +121 -7
  13. package/src/cli/commands/transcribe.ts +171 -0
  14. package/src/cli/deploy-readiness.ts +59 -3
  15. package/src/cli/help.ts +15 -0
  16. package/src/cli/index.ts +130 -0
  17. package/src/create-project.ts +4 -32
  18. package/src/index.ts +231 -0
  19. package/src/runtime/channel-test-harness.ts +2 -0
  20. package/src/runtime/channels/telegram.ts +326 -10
  21. package/src/runtime/channels/whatsapp-zapster.ts +319 -0
  22. package/src/runtime/channels.ts +47 -1
  23. package/src/runtime/chat.ts +2 -1
  24. package/src/runtime/config.ts +214 -4
  25. package/src/runtime/core/manifest.ts +61 -2
  26. package/src/runtime/deploy-readiness.ts +31 -1
  27. package/src/runtime/dev-server.ts +72 -4
  28. package/src/runtime/inspect.ts +142 -4
  29. package/src/runtime/integrations/composio.ts +257 -0
  30. package/src/runtime/skills.ts +95 -0
  31. package/src/runtime/targets/cloudflare/build.ts +162 -5
  32. package/src/runtime/tool-runner.ts +2 -1
  33. package/src/runtime/transcription.ts +483 -0
  34. package/src/templates/skills/agentkit-capsule/SKILL.md +1 -0
  35. package/src/templates/skills/agentkit-channels/SKILL.md +36 -1
  36. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +16 -1
  37. package/src/templates/skills/agentkit-channels/references/telegram.md +34 -0
  38. package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +29 -0
  39. package/src/templates/skills/agentkit-deploy/SKILL.md +1 -1
  40. package/src/templates/skills/agentkit-integrations/SKILL.md +66 -0
  41. package/src/templates/skills/agentkit-troubleshooting/SKILL.md +3 -2
@@ -0,0 +1,483 @@
1
+ import type { AgentTranscriptionConfig, TranscriptionProviderName } from "../index";
2
+
3
+ export type ResolvedTranscriptionConfig = {
4
+ enabled: boolean;
5
+ provider: TranscriptionProviderName;
6
+ model: string;
7
+ secret: string | null;
8
+ language: string | null;
9
+ prompt: string | null;
10
+ limits: {
11
+ maxDurationSeconds?: number;
12
+ maxBytes?: number;
13
+ };
14
+ rawAudioTtlSeconds: number | null;
15
+ };
16
+
17
+ export type TranscriptionInput = {
18
+ config: ResolvedTranscriptionConfig;
19
+ secrets: Record<string, string>;
20
+ audio: Uint8Array;
21
+ filename: string;
22
+ mimeType?: string;
23
+ durationSeconds?: number;
24
+ providerMetadata?: Record<string, unknown>;
25
+ };
26
+
27
+ export type TranscriptionFetch = (input: RequestInfo | URL, init?: RequestInit) => Promise<Response>;
28
+
29
+ export type TranscriptionResult =
30
+ | {
31
+ ok: true;
32
+ text: string;
33
+ provider: TranscriptionProviderName;
34
+ model: string;
35
+ language?: string;
36
+ durationSeconds?: number;
37
+ providerMetadata?: Record<string, unknown>;
38
+ }
39
+ | {
40
+ ok: false;
41
+ retryable: boolean;
42
+ code:
43
+ | "transcription_disabled"
44
+ | "transcription_secret_missing"
45
+ | "transcription_model_unsupported"
46
+ | "transcription_audio_too_large"
47
+ | "transcription_audio_too_long"
48
+ | "transcription_audio_format_unsupported"
49
+ | "transcription_provider_unavailable"
50
+ | "transcription_failed";
51
+ message: string;
52
+ provider?: TranscriptionProviderName;
53
+ model?: string;
54
+ providerMetadata?: Record<string, unknown>;
55
+ };
56
+
57
+ export type TranscriptionAdapter = {
58
+ provider: TranscriptionProviderName;
59
+ models: string[];
60
+ supportedMimeTypes: string[];
61
+ supportedExtensions: string[];
62
+ maxBytes: number;
63
+ requiredSecret(config: ResolvedTranscriptionConfig): string | null;
64
+ transcribe(input: TranscriptionInput, fetcher?: TranscriptionFetch): Promise<TranscriptionResult>;
65
+ };
66
+
67
+ const DEFAULT_MAX_BYTES = 25_000_000;
68
+ const OPENAI_MODELS = ["gpt-4o-mini-transcribe", "gpt-4o-transcribe", "whisper-1"];
69
+ const GROQ_MODELS = ["whisper-large-v3-turbo", "whisper-large-v3", "distil-whisper-large-v3-en"];
70
+ const OPENAI_EXTENSIONS = ["mp3", "mp4", "mpeg", "mpga", "m4a", "wav", "webm"];
71
+ const GROQ_EXTENSIONS = ["flac", "mp3", "mp4", "mpeg", "mpga", "m4a", "ogg", "wav", "webm"];
72
+ const OPENAI_MIME_TYPES = [
73
+ "audio/mpeg",
74
+ "audio/mp3",
75
+ "audio/mp4",
76
+ "audio/mpga",
77
+ "audio/m4a",
78
+ "audio/wav",
79
+ "audio/webm",
80
+ "video/mp4",
81
+ ];
82
+ const GROQ_MIME_TYPES = [
83
+ ...OPENAI_MIME_TYPES,
84
+ "audio/flac",
85
+ "audio/ogg",
86
+ "audio/opus",
87
+ "application/ogg",
88
+ ];
89
+
90
+ export function resolveTranscriptionConfig(config: AgentTranscriptionConfig | undefined): ResolvedTranscriptionConfig {
91
+ if (!config) {
92
+ return {
93
+ enabled: false,
94
+ provider: "test",
95
+ model: "fake",
96
+ secret: null,
97
+ language: null,
98
+ prompt: null,
99
+ limits: {},
100
+ rawAudioTtlSeconds: null,
101
+ };
102
+ }
103
+
104
+ return {
105
+ enabled: true,
106
+ provider: config.provider,
107
+ model: config.model,
108
+ secret: config.secret ?? defaultTranscriptionSecret(config.provider),
109
+ language: config.language ?? null,
110
+ prompt: config.prompt ?? null,
111
+ limits: {
112
+ ...(config.limits?.maxDurationSeconds !== undefined
113
+ ? { maxDurationSeconds: config.limits.maxDurationSeconds }
114
+ : {}),
115
+ ...(config.limits?.maxBytes !== undefined ? { maxBytes: config.limits.maxBytes } : {}),
116
+ },
117
+ rawAudioTtlSeconds: config.rawAudioTtlSeconds ?? 3600,
118
+ };
119
+ }
120
+
121
+ export async function transcribeAudio(
122
+ input: TranscriptionInput,
123
+ fetcher: TranscriptionFetch = fetch,
124
+ ): Promise<TranscriptionResult> {
125
+ if (!input.config.enabled) {
126
+ return {
127
+ ok: false,
128
+ retryable: false,
129
+ code: "transcription_disabled",
130
+ message: "Audio transcription is not configured for this Agent Capsule.",
131
+ };
132
+ }
133
+
134
+ const adapter = transcriptionAdapterFor(input.config.provider);
135
+
136
+ if (!adapter) {
137
+ return {
138
+ ok: false,
139
+ retryable: false,
140
+ code: "transcription_model_unsupported",
141
+ message: `No transcription adapter is registered for ${input.config.provider}.`,
142
+ provider: input.config.provider,
143
+ model: input.config.model,
144
+ };
145
+ }
146
+
147
+ const validation = validateTranscriptionInput(adapter, input);
148
+
149
+ if (!validation.ok) {
150
+ return validation;
151
+ }
152
+
153
+ return adapter.transcribe(input, fetcher);
154
+ }
155
+
156
+ export function transcriptionAdapterFor(provider: TranscriptionProviderName): TranscriptionAdapter | null {
157
+ if (provider === "test") {
158
+ return testTranscriptionAdapter;
159
+ }
160
+
161
+ if (provider === "openai") {
162
+ return openaiTranscriptionAdapter;
163
+ }
164
+
165
+ if (provider === "groq") {
166
+ return groqTranscriptionAdapter;
167
+ }
168
+
169
+ return null;
170
+ }
171
+
172
+ export function defaultTranscriptionSecret(provider: TranscriptionProviderName): string | null {
173
+ if (provider === "openai") {
174
+ return "OPENAI_API_KEY";
175
+ }
176
+
177
+ if (provider === "groq") {
178
+ return "GROQ_API_KEY";
179
+ }
180
+
181
+ return null;
182
+ }
183
+
184
+ function validateTranscriptionInput(adapter: TranscriptionAdapter, input: TranscriptionInput): TranscriptionResult {
185
+ if (!adapter.models.includes(input.config.model)) {
186
+ return {
187
+ ok: false,
188
+ retryable: false,
189
+ code: "transcription_model_unsupported",
190
+ message: `${input.config.provider} transcription model ${input.config.model} is not supported by AgentKit.`,
191
+ provider: input.config.provider,
192
+ model: input.config.model,
193
+ };
194
+ }
195
+
196
+ const maxBytes = Math.min(input.config.limits.maxBytes ?? adapter.maxBytes, adapter.maxBytes);
197
+ if (input.audio.byteLength > maxBytes) {
198
+ return {
199
+ ok: false,
200
+ retryable: false,
201
+ code: "transcription_audio_too_large",
202
+ message: `Audio file is ${input.audio.byteLength} bytes, which exceeds the configured ${maxBytes} byte limit.`,
203
+ provider: input.config.provider,
204
+ model: input.config.model,
205
+ };
206
+ }
207
+
208
+ if (
209
+ input.durationSeconds !== undefined &&
210
+ input.config.limits.maxDurationSeconds !== undefined &&
211
+ input.durationSeconds > input.config.limits.maxDurationSeconds
212
+ ) {
213
+ return {
214
+ ok: false,
215
+ retryable: false,
216
+ code: "transcription_audio_too_long",
217
+ message: `Audio is ${input.durationSeconds}s, which exceeds the configured ${input.config.limits.maxDurationSeconds}s limit.`,
218
+ provider: input.config.provider,
219
+ model: input.config.model,
220
+ };
221
+ }
222
+
223
+ if (!isSupportedAudioFormat(adapter, input.filename, input.mimeType)) {
224
+ return {
225
+ ok: false,
226
+ retryable: false,
227
+ code: "transcription_audio_format_unsupported",
228
+ message: `${input.config.provider} does not support audio format ${input.mimeType ?? extensionFor(input.filename) ?? "unknown"}.`,
229
+ provider: input.config.provider,
230
+ model: input.config.model,
231
+ };
232
+ }
233
+
234
+ const secret = adapter.requiredSecret(input.config);
235
+ if (secret && !input.secrets[secret]) {
236
+ return {
237
+ ok: false,
238
+ retryable: false,
239
+ code: "transcription_secret_missing",
240
+ message: `Secret ${secret} is not set for audio transcription.`,
241
+ provider: input.config.provider,
242
+ model: input.config.model,
243
+ };
244
+ }
245
+
246
+ return {
247
+ ok: true,
248
+ text: "",
249
+ provider: input.config.provider,
250
+ model: input.config.model,
251
+ };
252
+ }
253
+
254
+ const testTranscriptionAdapter: TranscriptionAdapter = {
255
+ provider: "test",
256
+ models: ["fake"],
257
+ supportedMimeTypes: ["audio/wav", "audio/ogg", "audio/mpeg", "audio/webm", "application/octet-stream"],
258
+ supportedExtensions: ["wav", "ogg", "mp3", "webm", "bin"],
259
+ maxBytes: DEFAULT_MAX_BYTES,
260
+ requiredSecret() {
261
+ return null;
262
+ },
263
+ async transcribe(input) {
264
+ const text = decodeFixtureTranscript(input.audio) ?? "fake audio transcript";
265
+
266
+ return {
267
+ ok: true,
268
+ text,
269
+ provider: "test",
270
+ model: input.config.model,
271
+ ...(input.config.language ? { language: input.config.language } : {}),
272
+ ...(input.durationSeconds !== undefined ? { durationSeconds: input.durationSeconds } : {}),
273
+ };
274
+ },
275
+ };
276
+
277
+ const openaiTranscriptionAdapter: TranscriptionAdapter = {
278
+ provider: "openai",
279
+ models: OPENAI_MODELS,
280
+ supportedMimeTypes: OPENAI_MIME_TYPES,
281
+ supportedExtensions: OPENAI_EXTENSIONS,
282
+ maxBytes: DEFAULT_MAX_BYTES,
283
+ requiredSecret(config) {
284
+ return config.secret;
285
+ },
286
+ async transcribe(input, fetcher = fetch) {
287
+ return transcribeViaOpenAiCompatibleEndpoint({
288
+ input,
289
+ fetcher,
290
+ url: "https://api.openai.com/v1/audio/transcriptions",
291
+ apiKey: input.secrets[input.config.secret ?? ""],
292
+ provider: "openai",
293
+ });
294
+ },
295
+ };
296
+
297
+ const groqTranscriptionAdapter: TranscriptionAdapter = {
298
+ provider: "groq",
299
+ models: GROQ_MODELS,
300
+ supportedMimeTypes: GROQ_MIME_TYPES,
301
+ supportedExtensions: GROQ_EXTENSIONS,
302
+ maxBytes: DEFAULT_MAX_BYTES,
303
+ requiredSecret(config) {
304
+ return config.secret;
305
+ },
306
+ async transcribe(input, fetcher = fetch) {
307
+ return transcribeViaOpenAiCompatibleEndpoint({
308
+ input,
309
+ fetcher,
310
+ url: "https://api.groq.com/openai/v1/audio/transcriptions",
311
+ apiKey: input.secrets[input.config.secret ?? ""],
312
+ provider: "groq",
313
+ });
314
+ },
315
+ };
316
+
317
+ async function transcribeViaOpenAiCompatibleEndpoint(input: {
318
+ input: TranscriptionInput;
319
+ fetcher: TranscriptionFetch;
320
+ url: string;
321
+ apiKey: string | undefined;
322
+ provider: TranscriptionProviderName;
323
+ }): Promise<TranscriptionResult> {
324
+ if (!input.apiKey) {
325
+ const secret = input.input.config.secret ?? defaultTranscriptionSecret(input.provider) ?? "TRANSCRIPTION_API_KEY";
326
+
327
+ return {
328
+ ok: false,
329
+ retryable: false,
330
+ code: "transcription_secret_missing",
331
+ message: `Secret ${secret} is not set for audio transcription.`,
332
+ provider: input.provider,
333
+ model: input.input.config.model,
334
+ };
335
+ }
336
+
337
+ const form = new FormData();
338
+ form.set("model", input.input.config.model);
339
+ form.set(
340
+ "file",
341
+ new Blob([arrayBufferForBlob(input.input.audio)], { type: input.input.mimeType ?? "application/octet-stream" }),
342
+ input.input.filename,
343
+ );
344
+ form.set("response_format", "json");
345
+
346
+ if (input.input.config.language) {
347
+ form.set("language", input.input.config.language);
348
+ }
349
+
350
+ if (input.input.config.prompt) {
351
+ form.set("prompt", input.input.config.prompt);
352
+ }
353
+
354
+ let response: Response;
355
+
356
+ try {
357
+ response = await input.fetcher(input.url, {
358
+ method: "POST",
359
+ headers: {
360
+ Authorization: `Bearer ${input.apiKey}`,
361
+ },
362
+ body: form,
363
+ });
364
+ } catch (error) {
365
+ return {
366
+ ok: false,
367
+ retryable: true,
368
+ code: "transcription_provider_unavailable",
369
+ message: `Transcription provider request failed before a response: ${redactSecret(
370
+ error instanceof Error ? error.message : String(error),
371
+ input.apiKey,
372
+ )}`,
373
+ provider: input.provider,
374
+ model: input.input.config.model,
375
+ };
376
+ }
377
+
378
+ const payload = await response.json().catch(() => null);
379
+
380
+ if (!response.ok) {
381
+ const message = readProviderError(payload) ?? `Transcription provider returned HTTP ${response.status}.`;
382
+
383
+ return {
384
+ ok: false,
385
+ retryable: response.status === 429 || response.status >= 500,
386
+ code: response.status === 429 || response.status >= 500 ? "transcription_provider_unavailable" : "transcription_failed",
387
+ message: redactSecret(message, input.apiKey),
388
+ provider: input.provider,
389
+ model: input.input.config.model,
390
+ providerMetadata: {
391
+ status: response.status,
392
+ },
393
+ };
394
+ }
395
+
396
+ const text = isRecord(payload) && typeof payload.text === "string" ? payload.text.trim() : "";
397
+
398
+ if (!text) {
399
+ return {
400
+ ok: false,
401
+ retryable: false,
402
+ code: "transcription_failed",
403
+ message: "Transcription provider returned an empty transcript.",
404
+ provider: input.provider,
405
+ model: input.input.config.model,
406
+ providerMetadata: {
407
+ status: response.status,
408
+ },
409
+ };
410
+ }
411
+
412
+ return {
413
+ ok: true,
414
+ text,
415
+ provider: input.provider,
416
+ model: input.input.config.model,
417
+ ...(input.input.config.language ? { language: input.input.config.language } : {}),
418
+ ...(input.input.durationSeconds !== undefined ? { durationSeconds: input.input.durationSeconds } : {}),
419
+ providerMetadata: {
420
+ status: response.status,
421
+ },
422
+ };
423
+ }
424
+
425
+ function isSupportedAudioFormat(adapter: TranscriptionAdapter, filename: string, mimeType?: string): boolean {
426
+ const normalizedMimeType = mimeType?.toLowerCase();
427
+
428
+ if (normalizedMimeType && adapter.supportedMimeTypes.includes(normalizedMimeType)) {
429
+ return true;
430
+ }
431
+
432
+ const extension = extensionFor(filename);
433
+
434
+ return Boolean(extension && adapter.supportedExtensions.includes(extension));
435
+ }
436
+
437
+ function extensionFor(filename: string): string | null {
438
+ const match = /\.([a-z0-9]+)$/i.exec(filename);
439
+ return match ? match[1].toLowerCase() : null;
440
+ }
441
+
442
+ function decodeFixtureTranscript(audio: Uint8Array): string | null {
443
+ try {
444
+ const text = new TextDecoder().decode(audio).trim();
445
+ return text.length > 0 && /^[\t\n\r -~\u00a0-\uffff]+$/.test(text) ? text : null;
446
+ } catch {
447
+ return null;
448
+ }
449
+ }
450
+
451
+ function arrayBufferForBlob(audio: Uint8Array): ArrayBuffer {
452
+ const copy = new Uint8Array(audio.byteLength);
453
+ copy.set(audio);
454
+ return copy.buffer;
455
+ }
456
+
457
+ function readProviderError(payload: unknown): string | null {
458
+ if (!isRecord(payload)) {
459
+ return null;
460
+ }
461
+
462
+ if (typeof payload.error === "string") {
463
+ return payload.error;
464
+ }
465
+
466
+ if (isRecord(payload.error) && typeof payload.error.message === "string") {
467
+ return payload.error.message;
468
+ }
469
+
470
+ if (typeof payload.message === "string") {
471
+ return payload.message;
472
+ }
473
+
474
+ return null;
475
+ }
476
+
477
+ function redactSecret(value: string, secret: string): string {
478
+ return secret ? value.replaceAll(secret, "<redacted>") : value;
479
+ }
480
+
481
+ function isRecord(value: unknown): value is Record<string, unknown> {
482
+ return Boolean(value) && typeof value === "object" && !Array.isArray(value);
483
+ }
@@ -19,6 +19,7 @@ Use this first inside an AgentKit Agent Capsule.
19
19
  - Build or reshape the agent from the owner's brief: `skills/agentkit-build-agent/SKILL.md`
20
20
  - Edit prompts: `skills/agentkit-prompts/SKILL.md`
21
21
  - Add actions or external data: `skills/agentkit-tools/SKILL.md`
22
+ - Add AgentKit-managed integrations such as managed Composio: `skills/agentkit-integrations/SKILL.md`
22
23
  - Add database tables or database-backed tools: `skills/agentkit-database/SKILL.md`
23
24
  - Add docs, FAQs, prices, policies, or CSV facts: `skills/agentkit-knowledge/SKILL.md`
24
25
  - Switch from `test/fake` to a real model provider: `skills/agentkit-provider/SKILL.md`
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: agentkit-channels
3
- description: Use when adding, connecting, testing, buffering, or debugging AgentKit website, Telegram, or WhatsApp channels, including channel config helpers, provider secrets, webhook setup, channel tests, delivery logs, and burst-message buffers.
3
+ description: Use when adding, connecting, testing, buffering, transcribing audio, or debugging AgentKit website, Telegram, or WhatsApp channels, including channel config helpers, provider secrets, webhook setup, channel tests, delivery logs, burst-message buffers, and transcription provider secrets.
4
4
  ---
5
5
 
6
6
  # AgentKit Channels
@@ -16,6 +16,39 @@ Channels receive user messages. Tools let the agent call external systems. Keep
16
16
  5. Connect channel resources through the CLI.
17
17
  6. Test, doctor, and inspect delivery logs.
18
18
 
19
+ ## Audio Transcription
20
+
21
+ Enable transcription at the agent level and opt in per channel with `audio.mode: "transcribe"`.
22
+
23
+ ```ts
24
+ export default defineAgent({
25
+ // ...
26
+ transcription: {
27
+ provider: "groq",
28
+ model: "whisper-large-v3-turbo",
29
+ secret: "GROQ_API_KEY",
30
+ language: "pt",
31
+ limits: {
32
+ maxDurationSeconds: 180,
33
+ maxBytes: 20_000_000,
34
+ },
35
+ },
36
+ channels: [
37
+ telegramChannel({
38
+ name: "support-telegram",
39
+ audio: { mode: "transcribe" },
40
+ }),
41
+ ],
42
+ });
43
+ ```
44
+
45
+ V1 providers:
46
+
47
+ - `openai`: `gpt-4o-mini-transcribe`, `gpt-4o-transcribe`, `whisper-1`; default secret `OPENAI_API_KEY`.
48
+ - `groq`: `whisper-large-v3-turbo`, `whisper-large-v3`, `distil-whisper-large-v3-en`; default secret `GROQ_API_KEY`.
49
+
50
+ Telegram voice notes are usually OGG/Opus, so use Groq for the default Telegram voice-note path in V1. Zapster audio needs a usable HTTPS Zapster media download URL in the webhook payload; arbitrary hosts are rejected before bearer auth is sent. Hosted channel creation requires the transcription secret automatically when the channel enables transcription. Webhooks only enqueue audio jobs; download and transcription run in the retryable channel worker before the agent run.
51
+
19
52
  ## Buffering
20
53
 
21
54
  Enable `buffer.mode: "debounce"` when clients send several short messages in a row and the agent should answer once.
@@ -47,6 +80,8 @@ npm run agentkit -- channels connect telegram support-telegram
47
80
  npm run agentkit -- channels add whatsapp support-whatsapp --provider zapster
48
81
  npm run agentkit -- channels doctor support-telegram
49
82
  npm run agentkit -- channels test support-telegram --message "hello"
83
+ npm run agentkit -- channels test-audio support-telegram --fixture voice-note
84
+ npm run agentkit -- transcribe smoke --provider groq
50
85
  npm run agentkit -- channels deliveries list support-telegram
51
86
  ```
52
87
 
@@ -7,6 +7,8 @@ agentkit channels list
7
7
  agentkit channels status <name>
8
8
  agentkit channels doctor <name>
9
9
  agentkit channels test <name> --message "hello"
10
+ agentkit channels test-audio <name> --fixture voice-note
11
+ agentkit transcribe smoke --provider groq
10
12
  agentkit channels deliveries list <name> --since 24h
11
13
  agentkit channels deliveries show <delivery-id>
12
14
  agentkit channels buffers list <name>
@@ -22,6 +24,10 @@ Common states:
22
24
  webhook_received
23
25
  validated
24
26
  duplicate
27
+ audio_received
28
+ audio_downloaded
29
+ transcribing
30
+ transcribed
25
31
  buffered
26
32
  queued
27
33
  running
@@ -38,11 +44,20 @@ skipped
38
44
 
39
45
  Common errors:
40
46
 
41
- - `channel_not_found`: webhook URL points to an unknown channel.
47
+ - `channel_not_found`: webhook URL points to an unknown channel. `channels test` is the official synthetic smoke and should resolve the current `.agentkit/deploy.json` channel.
42
48
  - `channel_secret_missing`: required hosted secret is not set.
43
49
  - `channel_signature_invalid`: webhook secret, token, or origin header mismatch.
44
50
  - `channel_payload_invalid`: malformed or unsupported provider payload.
45
51
  - `channel_event_duplicate`: provider retry; do not create a second run.
52
+ - `audio_received`: audio message was accepted and normalized.
53
+ - `audio_downloaded`: retryable channel worker downloaded provider media into memory.
54
+ - `transcribing`: AgentKit is calling the configured transcription provider.
55
+ - `transcribed`: transcript text was queued for the agent.
56
+ - `channel_audio_download_unavailable`: provider audio payload did not include a usable download URL, or Zapster sent a non-HTTPS/non-Zapster media host.
57
+ - `transcription_secret_missing`: managed transcription secret is missing.
58
+ - `transcription_audio_too_large` or `transcription_audio_too_long`: audio exceeded configured limits.
59
+ - `transcription_audio_format_unsupported`: provider does not accept this audio MIME type or extension.
60
+ - `transcription_provider_unavailable`: retryable transcription provider failure.
46
61
  - `channel_limit_exceeded`: backpressure skipped the message.
47
62
  - `synthetic_expected_failure`: a synthetic test reached AgentKit, but the provider correctly rejected a fake test recipient.
48
63
  - `buffered` delivery state: message is waiting for the channel quiet window or max wait before one coalesced agent run is queued.
@@ -7,6 +7,12 @@ TELEGRAM_BOT_TOKEN
7
7
  TELEGRAM_WEBHOOK_SECRET
8
8
  ```
9
9
 
10
+ Audio transcription also needs the configured transcription secret, usually:
11
+
12
+ ```txt
13
+ GROQ_API_KEY
14
+ ```
15
+
10
16
  Commands:
11
17
 
12
18
  ```sh
@@ -17,6 +23,8 @@ agentkit channels connect telegram support-telegram
17
23
  agentkit channels doctor support-telegram
18
24
  agentkit channels status support-telegram
19
25
  agentkit channels test support-telegram --message "hello"
26
+ agentkit channels test-audio support-telegram --fixture voice-note
27
+ agentkit transcribe smoke --provider groq
20
28
  agentkit channels deliveries list support-telegram
21
29
  ```
22
30
 
@@ -36,3 +44,29 @@ telegramChannel({
36
44
  },
37
45
  })
38
46
  ```
47
+
48
+ Transcribe Telegram voice notes:
49
+
50
+ ```ts
51
+ export default defineAgent({
52
+ // ...
53
+ transcription: {
54
+ provider: "groq",
55
+ model: "whisper-large-v3-turbo",
56
+ secret: "GROQ_API_KEY",
57
+ language: "pt",
58
+ limits: {
59
+ maxDurationSeconds: 180,
60
+ maxBytes: 20_000_000,
61
+ },
62
+ },
63
+ channels: [
64
+ telegramChannel({
65
+ name: "support-telegram",
66
+ audio: { mode: "transcribe" },
67
+ }),
68
+ ],
69
+ });
70
+ ```
71
+
72
+ AgentKit validates the Telegram webhook, normalizes `voice` and `audio` payloads, enqueues an audio job, then the retryable channel worker calls Telegram `getFile`, downloads the media with `TELEGRAM_BOT_TOKEN`, sends the bytes to the configured transcription provider, and runs the agent with transcript text. Telegram voice notes are usually OGG/Opus; use Groq in V1 for that path. `test-audio` validates the audio channel ingress path; `transcribe smoke` validates the transcription provider separately.
@@ -14,6 +14,8 @@ Optional hardening secret:
14
14
  ZAPSTER_WEBHOOK_TOKEN
15
15
  ```
16
16
 
17
+ Audio transcription also needs the configured transcription secret, usually `OPENAI_API_KEY` or `GROQ_API_KEY`.
18
+
17
19
  Commands:
18
20
 
19
21
  ```sh
@@ -46,3 +48,30 @@ whatsappChannel({
46
48
  },
47
49
  })
48
50
  ```
51
+
52
+ Transcribe WhatsApp audio:
53
+
54
+ ```ts
55
+ export default defineAgent({
56
+ // ...
57
+ transcription: {
58
+ provider: "openai",
59
+ model: "gpt-4o-mini-transcribe",
60
+ secret: "OPENAI_API_KEY",
61
+ language: "pt",
62
+ limits: {
63
+ maxDurationSeconds: 180,
64
+ maxBytes: 20_000_000,
65
+ },
66
+ },
67
+ channels: [
68
+ whatsappChannel({
69
+ name: "support-whatsapp",
70
+ provider: "zapster",
71
+ audio: { mode: "transcribe" },
72
+ }),
73
+ ],
74
+ });
75
+ ```
76
+
77
+ Zapster audio payloads must include a usable HTTPS Zapster media download URL such as `audio.downloadUrl`, `audio.url`, `audio.mediaUrl`, or the snake_case equivalents. AgentKit rejects arbitrary hosts before sending `ZAPSTER_API_KEY`. The retryable channel worker downloads the media, transcribes it through the configured provider secret, and runs the agent with transcript text. If Zapster sends only a media ID in V1, AgentKit records `channel_audio_download_unavailable`.
@@ -18,6 +18,7 @@ Use this when the owner asks to prepare, test, or run hosted deploy.
18
18
 
19
19
  ```sh
20
20
  npm run typecheck
21
+ npm run agentkit -- skills status
21
22
  npm run agentkit -- inspect
22
23
  npm run agentkit -- db migrate
23
24
  npm run chat -- --message "hello"
@@ -41,4 +42,3 @@ Open the printed `Chat:` URL and report it to the owner.
41
42
  ## Production Handoff
42
43
 
43
44
  Report changed files, required env/secret names, database schema changes, deploy order, smoke checks, rollback concerns, and whether the provider was still `test/fake`.
44
-