@andreprado/agentkit 0.1.0-alpha.14 → 0.1.0-alpha.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/docs/guides/add-channel.md +63 -0
- package/docs/guides/channel-security.md +32 -0
- package/docs/guides/connect-telegram.md +58 -0
- package/docs/guides/connect-whatsapp-zapster.md +65 -0
- package/docs/guides/run-evals.md +73 -25
- package/docs/llms-full.txt +24 -9
- package/docs/llms.txt +2 -0
- package/package.json +1 -1
- package/src/cli/cloud-client.ts +30 -10
- package/src/cli/commands/channels.ts +2 -0
- package/src/cli/deploy-readiness.ts +32 -11
- package/src/cli/index.ts +20 -6
- package/src/cloud/client.ts +4 -3
- package/src/cloud/contracts.ts +1 -1
- package/src/create-project.ts +1 -1
- package/src/index.ts +110 -1
- package/src/providers/pi.ts +14 -1
- package/src/providers/test.ts +36 -0
- package/src/runtime/channel-test-harness.ts +2 -0
- package/src/runtime/channels/telegram.ts +326 -10
- package/src/runtime/channels/whatsapp-zapster.ts +319 -0
- package/src/runtime/channels.ts +47 -1
- package/src/runtime/chat.ts +59 -42
- package/src/runtime/config.ts +96 -4
- package/src/runtime/core/manifest.ts +35 -3
- package/src/runtime/deploy-readiness.ts +3 -3
- package/src/runtime/dev-server.ts +243 -17
- package/src/runtime/env.ts +8 -3
- package/src/runtime/evals.ts +404 -69
- package/src/runtime/inspect.ts +46 -0
- package/src/runtime/prompt-context.ts +141 -0
- package/src/runtime/runtime-contract.ts +17 -7
- package/src/runtime/targets/cloudflare/build.ts +25 -3
- package/src/runtime/targets/container/server.ts +1 -1
- package/src/runtime/targets/vps/deploy.ts +25 -8
- package/src/runtime/tool-runner.ts +7 -0
- package/src/runtime/tools.ts +8 -2
- package/src/runtime/transcription.ts +483 -0
- package/src/templates/blank.ts +8 -3
- package/src/templates/dentista.ts +18 -10
- package/src/templates/skills/agentkit-build-agent/SKILL.md +6 -5
- package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
- package/src/templates/skills/agentkit-capsule/SKILL.md +1 -1
- package/src/templates/skills/agentkit-channels/SKILL.md +34 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +13 -0
- package/src/templates/skills/agentkit-channels/references/telegram.md +32 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +29 -0
- package/src/templates/skills/agentkit-evals/SKILL.md +53 -13
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
- package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
- package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
- package/src/templates/skills/agentkit-tools/SKILL.md +2 -1
- package/src/templates/support.ts +8 -3
|
@@ -0,0 +1,483 @@
|
|
|
1
|
+
import type { AgentTranscriptionConfig, TranscriptionProviderName } from "../index";
|
|
2
|
+
|
|
3
|
+
export type ResolvedTranscriptionConfig = {
|
|
4
|
+
enabled: boolean;
|
|
5
|
+
provider: TranscriptionProviderName;
|
|
6
|
+
model: string;
|
|
7
|
+
secret: string | null;
|
|
8
|
+
language: string | null;
|
|
9
|
+
prompt: string | null;
|
|
10
|
+
limits: {
|
|
11
|
+
maxDurationSeconds?: number;
|
|
12
|
+
maxBytes?: number;
|
|
13
|
+
};
|
|
14
|
+
rawAudioTtlSeconds: number | null;
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
export type TranscriptionInput = {
|
|
18
|
+
config: ResolvedTranscriptionConfig;
|
|
19
|
+
secrets: Record<string, string>;
|
|
20
|
+
audio: Uint8Array;
|
|
21
|
+
filename: string;
|
|
22
|
+
mimeType?: string;
|
|
23
|
+
durationSeconds?: number;
|
|
24
|
+
providerMetadata?: Record<string, unknown>;
|
|
25
|
+
};
|
|
26
|
+
|
|
27
|
+
export type TranscriptionFetch = (input: RequestInfo | URL, init?: RequestInit) => Promise<Response>;
|
|
28
|
+
|
|
29
|
+
export type TranscriptionResult =
|
|
30
|
+
| {
|
|
31
|
+
ok: true;
|
|
32
|
+
text: string;
|
|
33
|
+
provider: TranscriptionProviderName;
|
|
34
|
+
model: string;
|
|
35
|
+
language?: string;
|
|
36
|
+
durationSeconds?: number;
|
|
37
|
+
providerMetadata?: Record<string, unknown>;
|
|
38
|
+
}
|
|
39
|
+
| {
|
|
40
|
+
ok: false;
|
|
41
|
+
retryable: boolean;
|
|
42
|
+
code:
|
|
43
|
+
| "transcription_disabled"
|
|
44
|
+
| "transcription_secret_missing"
|
|
45
|
+
| "transcription_model_unsupported"
|
|
46
|
+
| "transcription_audio_too_large"
|
|
47
|
+
| "transcription_audio_too_long"
|
|
48
|
+
| "transcription_audio_format_unsupported"
|
|
49
|
+
| "transcription_provider_unavailable"
|
|
50
|
+
| "transcription_failed";
|
|
51
|
+
message: string;
|
|
52
|
+
provider?: TranscriptionProviderName;
|
|
53
|
+
model?: string;
|
|
54
|
+
providerMetadata?: Record<string, unknown>;
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
export type TranscriptionAdapter = {
|
|
58
|
+
provider: TranscriptionProviderName;
|
|
59
|
+
models: string[];
|
|
60
|
+
supportedMimeTypes: string[];
|
|
61
|
+
supportedExtensions: string[];
|
|
62
|
+
maxBytes: number;
|
|
63
|
+
requiredSecret(config: ResolvedTranscriptionConfig): string | null;
|
|
64
|
+
transcribe(input: TranscriptionInput, fetcher?: TranscriptionFetch): Promise<TranscriptionResult>;
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
const DEFAULT_MAX_BYTES = 25_000_000;
|
|
68
|
+
const OPENAI_MODELS = ["gpt-4o-mini-transcribe", "gpt-4o-transcribe", "whisper-1"];
|
|
69
|
+
const GROQ_MODELS = ["whisper-large-v3-turbo", "whisper-large-v3", "distil-whisper-large-v3-en"];
|
|
70
|
+
const OPENAI_EXTENSIONS = ["mp3", "mp4", "mpeg", "mpga", "m4a", "wav", "webm"];
|
|
71
|
+
const GROQ_EXTENSIONS = ["flac", "mp3", "mp4", "mpeg", "mpga", "m4a", "ogg", "wav", "webm"];
|
|
72
|
+
const OPENAI_MIME_TYPES = [
|
|
73
|
+
"audio/mpeg",
|
|
74
|
+
"audio/mp3",
|
|
75
|
+
"audio/mp4",
|
|
76
|
+
"audio/mpga",
|
|
77
|
+
"audio/m4a",
|
|
78
|
+
"audio/wav",
|
|
79
|
+
"audio/webm",
|
|
80
|
+
"video/mp4",
|
|
81
|
+
];
|
|
82
|
+
const GROQ_MIME_TYPES = [
|
|
83
|
+
...OPENAI_MIME_TYPES,
|
|
84
|
+
"audio/flac",
|
|
85
|
+
"audio/ogg",
|
|
86
|
+
"audio/opus",
|
|
87
|
+
"application/ogg",
|
|
88
|
+
];
|
|
89
|
+
|
|
90
|
+
export function resolveTranscriptionConfig(config: AgentTranscriptionConfig | undefined): ResolvedTranscriptionConfig {
|
|
91
|
+
if (!config) {
|
|
92
|
+
return {
|
|
93
|
+
enabled: false,
|
|
94
|
+
provider: "test",
|
|
95
|
+
model: "fake",
|
|
96
|
+
secret: null,
|
|
97
|
+
language: null,
|
|
98
|
+
prompt: null,
|
|
99
|
+
limits: {},
|
|
100
|
+
rawAudioTtlSeconds: null,
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
return {
|
|
105
|
+
enabled: true,
|
|
106
|
+
provider: config.provider,
|
|
107
|
+
model: config.model,
|
|
108
|
+
secret: config.secret ?? defaultTranscriptionSecret(config.provider),
|
|
109
|
+
language: config.language ?? null,
|
|
110
|
+
prompt: config.prompt ?? null,
|
|
111
|
+
limits: {
|
|
112
|
+
...(config.limits?.maxDurationSeconds !== undefined
|
|
113
|
+
? { maxDurationSeconds: config.limits.maxDurationSeconds }
|
|
114
|
+
: {}),
|
|
115
|
+
...(config.limits?.maxBytes !== undefined ? { maxBytes: config.limits.maxBytes } : {}),
|
|
116
|
+
},
|
|
117
|
+
rawAudioTtlSeconds: config.rawAudioTtlSeconds ?? 3600,
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
export async function transcribeAudio(
|
|
122
|
+
input: TranscriptionInput,
|
|
123
|
+
fetcher: TranscriptionFetch = fetch,
|
|
124
|
+
): Promise<TranscriptionResult> {
|
|
125
|
+
if (!input.config.enabled) {
|
|
126
|
+
return {
|
|
127
|
+
ok: false,
|
|
128
|
+
retryable: false,
|
|
129
|
+
code: "transcription_disabled",
|
|
130
|
+
message: "Audio transcription is not configured for this Agent Capsule.",
|
|
131
|
+
};
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
const adapter = transcriptionAdapterFor(input.config.provider);
|
|
135
|
+
|
|
136
|
+
if (!adapter) {
|
|
137
|
+
return {
|
|
138
|
+
ok: false,
|
|
139
|
+
retryable: false,
|
|
140
|
+
code: "transcription_model_unsupported",
|
|
141
|
+
message: `No transcription adapter is registered for ${input.config.provider}.`,
|
|
142
|
+
provider: input.config.provider,
|
|
143
|
+
model: input.config.model,
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
const validation = validateTranscriptionInput(adapter, input);
|
|
148
|
+
|
|
149
|
+
if (!validation.ok) {
|
|
150
|
+
return validation;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
return adapter.transcribe(input, fetcher);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
export function transcriptionAdapterFor(provider: TranscriptionProviderName): TranscriptionAdapter | null {
|
|
157
|
+
if (provider === "test") {
|
|
158
|
+
return testTranscriptionAdapter;
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
if (provider === "openai") {
|
|
162
|
+
return openaiTranscriptionAdapter;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
if (provider === "groq") {
|
|
166
|
+
return groqTranscriptionAdapter;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
return null;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
export function defaultTranscriptionSecret(provider: TranscriptionProviderName): string | null {
|
|
173
|
+
if (provider === "openai") {
|
|
174
|
+
return "OPENAI_API_KEY";
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
if (provider === "groq") {
|
|
178
|
+
return "GROQ_API_KEY";
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
return null;
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
function validateTranscriptionInput(adapter: TranscriptionAdapter, input: TranscriptionInput): TranscriptionResult {
|
|
185
|
+
if (!adapter.models.includes(input.config.model)) {
|
|
186
|
+
return {
|
|
187
|
+
ok: false,
|
|
188
|
+
retryable: false,
|
|
189
|
+
code: "transcription_model_unsupported",
|
|
190
|
+
message: `${input.config.provider} transcription model ${input.config.model} is not supported by AgentKit.`,
|
|
191
|
+
provider: input.config.provider,
|
|
192
|
+
model: input.config.model,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const maxBytes = Math.min(input.config.limits.maxBytes ?? adapter.maxBytes, adapter.maxBytes);
|
|
197
|
+
if (input.audio.byteLength > maxBytes) {
|
|
198
|
+
return {
|
|
199
|
+
ok: false,
|
|
200
|
+
retryable: false,
|
|
201
|
+
code: "transcription_audio_too_large",
|
|
202
|
+
message: `Audio file is ${input.audio.byteLength} bytes, which exceeds the configured ${maxBytes} byte limit.`,
|
|
203
|
+
provider: input.config.provider,
|
|
204
|
+
model: input.config.model,
|
|
205
|
+
};
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
if (
|
|
209
|
+
input.durationSeconds !== undefined &&
|
|
210
|
+
input.config.limits.maxDurationSeconds !== undefined &&
|
|
211
|
+
input.durationSeconds > input.config.limits.maxDurationSeconds
|
|
212
|
+
) {
|
|
213
|
+
return {
|
|
214
|
+
ok: false,
|
|
215
|
+
retryable: false,
|
|
216
|
+
code: "transcription_audio_too_long",
|
|
217
|
+
message: `Audio is ${input.durationSeconds}s, which exceeds the configured ${input.config.limits.maxDurationSeconds}s limit.`,
|
|
218
|
+
provider: input.config.provider,
|
|
219
|
+
model: input.config.model,
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
if (!isSupportedAudioFormat(adapter, input.filename, input.mimeType)) {
|
|
224
|
+
return {
|
|
225
|
+
ok: false,
|
|
226
|
+
retryable: false,
|
|
227
|
+
code: "transcription_audio_format_unsupported",
|
|
228
|
+
message: `${input.config.provider} does not support audio format ${input.mimeType ?? extensionFor(input.filename) ?? "unknown"}.`,
|
|
229
|
+
provider: input.config.provider,
|
|
230
|
+
model: input.config.model,
|
|
231
|
+
};
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
const secret = adapter.requiredSecret(input.config);
|
|
235
|
+
if (secret && !input.secrets[secret]) {
|
|
236
|
+
return {
|
|
237
|
+
ok: false,
|
|
238
|
+
retryable: false,
|
|
239
|
+
code: "transcription_secret_missing",
|
|
240
|
+
message: `Secret ${secret} is not set for audio transcription.`,
|
|
241
|
+
provider: input.config.provider,
|
|
242
|
+
model: input.config.model,
|
|
243
|
+
};
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
return {
|
|
247
|
+
ok: true,
|
|
248
|
+
text: "",
|
|
249
|
+
provider: input.config.provider,
|
|
250
|
+
model: input.config.model,
|
|
251
|
+
};
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
const testTranscriptionAdapter: TranscriptionAdapter = {
|
|
255
|
+
provider: "test",
|
|
256
|
+
models: ["fake"],
|
|
257
|
+
supportedMimeTypes: ["audio/wav", "audio/ogg", "audio/mpeg", "audio/webm", "application/octet-stream"],
|
|
258
|
+
supportedExtensions: ["wav", "ogg", "mp3", "webm", "bin"],
|
|
259
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
260
|
+
requiredSecret() {
|
|
261
|
+
return null;
|
|
262
|
+
},
|
|
263
|
+
async transcribe(input) {
|
|
264
|
+
const text = decodeFixtureTranscript(input.audio) ?? "fake audio transcript";
|
|
265
|
+
|
|
266
|
+
return {
|
|
267
|
+
ok: true,
|
|
268
|
+
text,
|
|
269
|
+
provider: "test",
|
|
270
|
+
model: input.config.model,
|
|
271
|
+
...(input.config.language ? { language: input.config.language } : {}),
|
|
272
|
+
...(input.durationSeconds !== undefined ? { durationSeconds: input.durationSeconds } : {}),
|
|
273
|
+
};
|
|
274
|
+
},
|
|
275
|
+
};
|
|
276
|
+
|
|
277
|
+
const openaiTranscriptionAdapter: TranscriptionAdapter = {
|
|
278
|
+
provider: "openai",
|
|
279
|
+
models: OPENAI_MODELS,
|
|
280
|
+
supportedMimeTypes: OPENAI_MIME_TYPES,
|
|
281
|
+
supportedExtensions: OPENAI_EXTENSIONS,
|
|
282
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
283
|
+
requiredSecret(config) {
|
|
284
|
+
return config.secret;
|
|
285
|
+
},
|
|
286
|
+
async transcribe(input, fetcher = fetch) {
|
|
287
|
+
return transcribeViaOpenAiCompatibleEndpoint({
|
|
288
|
+
input,
|
|
289
|
+
fetcher,
|
|
290
|
+
url: "https://api.openai.com/v1/audio/transcriptions",
|
|
291
|
+
apiKey: input.secrets[input.config.secret ?? ""],
|
|
292
|
+
provider: "openai",
|
|
293
|
+
});
|
|
294
|
+
},
|
|
295
|
+
};
|
|
296
|
+
|
|
297
|
+
const groqTranscriptionAdapter: TranscriptionAdapter = {
|
|
298
|
+
provider: "groq",
|
|
299
|
+
models: GROQ_MODELS,
|
|
300
|
+
supportedMimeTypes: GROQ_MIME_TYPES,
|
|
301
|
+
supportedExtensions: GROQ_EXTENSIONS,
|
|
302
|
+
maxBytes: DEFAULT_MAX_BYTES,
|
|
303
|
+
requiredSecret(config) {
|
|
304
|
+
return config.secret;
|
|
305
|
+
},
|
|
306
|
+
async transcribe(input, fetcher = fetch) {
|
|
307
|
+
return transcribeViaOpenAiCompatibleEndpoint({
|
|
308
|
+
input,
|
|
309
|
+
fetcher,
|
|
310
|
+
url: "https://api.groq.com/openai/v1/audio/transcriptions",
|
|
311
|
+
apiKey: input.secrets[input.config.secret ?? ""],
|
|
312
|
+
provider: "groq",
|
|
313
|
+
});
|
|
314
|
+
},
|
|
315
|
+
};
|
|
316
|
+
|
|
317
|
+
async function transcribeViaOpenAiCompatibleEndpoint(input: {
|
|
318
|
+
input: TranscriptionInput;
|
|
319
|
+
fetcher: TranscriptionFetch;
|
|
320
|
+
url: string;
|
|
321
|
+
apiKey: string | undefined;
|
|
322
|
+
provider: TranscriptionProviderName;
|
|
323
|
+
}): Promise<TranscriptionResult> {
|
|
324
|
+
if (!input.apiKey) {
|
|
325
|
+
const secret = input.input.config.secret ?? defaultTranscriptionSecret(input.provider) ?? "TRANSCRIPTION_API_KEY";
|
|
326
|
+
|
|
327
|
+
return {
|
|
328
|
+
ok: false,
|
|
329
|
+
retryable: false,
|
|
330
|
+
code: "transcription_secret_missing",
|
|
331
|
+
message: `Secret ${secret} is not set for audio transcription.`,
|
|
332
|
+
provider: input.provider,
|
|
333
|
+
model: input.input.config.model,
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
const form = new FormData();
|
|
338
|
+
form.set("model", input.input.config.model);
|
|
339
|
+
form.set(
|
|
340
|
+
"file",
|
|
341
|
+
new Blob([arrayBufferForBlob(input.input.audio)], { type: input.input.mimeType ?? "application/octet-stream" }),
|
|
342
|
+
input.input.filename,
|
|
343
|
+
);
|
|
344
|
+
form.set("response_format", "json");
|
|
345
|
+
|
|
346
|
+
if (input.input.config.language) {
|
|
347
|
+
form.set("language", input.input.config.language);
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
if (input.input.config.prompt) {
|
|
351
|
+
form.set("prompt", input.input.config.prompt);
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
let response: Response;
|
|
355
|
+
|
|
356
|
+
try {
|
|
357
|
+
response = await input.fetcher(input.url, {
|
|
358
|
+
method: "POST",
|
|
359
|
+
headers: {
|
|
360
|
+
Authorization: `Bearer ${input.apiKey}`,
|
|
361
|
+
},
|
|
362
|
+
body: form,
|
|
363
|
+
});
|
|
364
|
+
} catch (error) {
|
|
365
|
+
return {
|
|
366
|
+
ok: false,
|
|
367
|
+
retryable: true,
|
|
368
|
+
code: "transcription_provider_unavailable",
|
|
369
|
+
message: `Transcription provider request failed before a response: ${redactSecret(
|
|
370
|
+
error instanceof Error ? error.message : String(error),
|
|
371
|
+
input.apiKey,
|
|
372
|
+
)}`,
|
|
373
|
+
provider: input.provider,
|
|
374
|
+
model: input.input.config.model,
|
|
375
|
+
};
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
const payload = await response.json().catch(() => null);
|
|
379
|
+
|
|
380
|
+
if (!response.ok) {
|
|
381
|
+
const message = readProviderError(payload) ?? `Transcription provider returned HTTP ${response.status}.`;
|
|
382
|
+
|
|
383
|
+
return {
|
|
384
|
+
ok: false,
|
|
385
|
+
retryable: response.status === 429 || response.status >= 500,
|
|
386
|
+
code: response.status === 429 || response.status >= 500 ? "transcription_provider_unavailable" : "transcription_failed",
|
|
387
|
+
message: redactSecret(message, input.apiKey),
|
|
388
|
+
provider: input.provider,
|
|
389
|
+
model: input.input.config.model,
|
|
390
|
+
providerMetadata: {
|
|
391
|
+
status: response.status,
|
|
392
|
+
},
|
|
393
|
+
};
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
const text = isRecord(payload) && typeof payload.text === "string" ? payload.text.trim() : "";
|
|
397
|
+
|
|
398
|
+
if (!text) {
|
|
399
|
+
return {
|
|
400
|
+
ok: false,
|
|
401
|
+
retryable: false,
|
|
402
|
+
code: "transcription_failed",
|
|
403
|
+
message: "Transcription provider returned an empty transcript.",
|
|
404
|
+
provider: input.provider,
|
|
405
|
+
model: input.input.config.model,
|
|
406
|
+
providerMetadata: {
|
|
407
|
+
status: response.status,
|
|
408
|
+
},
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
return {
|
|
413
|
+
ok: true,
|
|
414
|
+
text,
|
|
415
|
+
provider: input.provider,
|
|
416
|
+
model: input.input.config.model,
|
|
417
|
+
...(input.input.config.language ? { language: input.input.config.language } : {}),
|
|
418
|
+
...(input.input.durationSeconds !== undefined ? { durationSeconds: input.input.durationSeconds } : {}),
|
|
419
|
+
providerMetadata: {
|
|
420
|
+
status: response.status,
|
|
421
|
+
},
|
|
422
|
+
};
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
function isSupportedAudioFormat(adapter: TranscriptionAdapter, filename: string, mimeType?: string): boolean {
|
|
426
|
+
const normalizedMimeType = mimeType?.toLowerCase();
|
|
427
|
+
|
|
428
|
+
if (normalizedMimeType && adapter.supportedMimeTypes.includes(normalizedMimeType)) {
|
|
429
|
+
return true;
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
const extension = extensionFor(filename);
|
|
433
|
+
|
|
434
|
+
return Boolean(extension && adapter.supportedExtensions.includes(extension));
|
|
435
|
+
}
|
|
436
|
+
|
|
437
|
+
function extensionFor(filename: string): string | null {
|
|
438
|
+
const match = /\.([a-z0-9]+)$/i.exec(filename);
|
|
439
|
+
return match ? match[1].toLowerCase() : null;
|
|
440
|
+
}
|
|
441
|
+
|
|
442
|
+
function decodeFixtureTranscript(audio: Uint8Array): string | null {
|
|
443
|
+
try {
|
|
444
|
+
const text = new TextDecoder().decode(audio).trim();
|
|
445
|
+
return text.length > 0 && /^[\t\n\r -~\u00a0-\uffff]+$/.test(text) ? text : null;
|
|
446
|
+
} catch {
|
|
447
|
+
return null;
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
function arrayBufferForBlob(audio: Uint8Array): ArrayBuffer {
|
|
452
|
+
const copy = new Uint8Array(audio.byteLength);
|
|
453
|
+
copy.set(audio);
|
|
454
|
+
return copy.buffer;
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
function readProviderError(payload: unknown): string | null {
|
|
458
|
+
if (!isRecord(payload)) {
|
|
459
|
+
return null;
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
if (typeof payload.error === "string") {
|
|
463
|
+
return payload.error;
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
if (isRecord(payload.error) && typeof payload.error.message === "string") {
|
|
467
|
+
return payload.error.message;
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
if (typeof payload.message === "string") {
|
|
471
|
+
return payload.message;
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
return null;
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
function redactSecret(value: string, secret: string): string {
|
|
478
|
+
return secret ? value.replaceAll(secret, "<redacted>") : value;
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
function isRecord(value: unknown): value is Record<string, unknown> {
|
|
482
|
+
return Boolean(value) && typeof value === "object" && !Array.isArray(value);
|
|
483
|
+
}
|
package/src/templates/blank.ts
CHANGED
|
@@ -124,13 +124,18 @@ Answer clearly, ask for missing context when needed, and do not claim to have pe
|
|
|
124
124
|
},
|
|
125
125
|
{
|
|
126
126
|
path: "evals/smoke.eval.ts",
|
|
127
|
-
contents: `
|
|
127
|
+
contents: `import { defineEval } from "@andreprado/agentkit";
|
|
128
|
+
|
|
129
|
+
export default defineEval({
|
|
128
130
|
name: "smoke",
|
|
129
131
|
input: "Say hello in one short sentence.",
|
|
130
132
|
expect: {
|
|
131
|
-
|
|
133
|
+
response: {
|
|
134
|
+
caseInsensitiveContains: "hello",
|
|
135
|
+
maxLength: 160,
|
|
136
|
+
},
|
|
132
137
|
},
|
|
133
|
-
};
|
|
138
|
+
});
|
|
134
139
|
`,
|
|
135
140
|
},
|
|
136
141
|
{
|
|
@@ -83,6 +83,7 @@ export default defineAgent({
|
|
|
83
83
|
name: "test",
|
|
84
84
|
model: "fake",
|
|
85
85
|
},
|
|
86
|
+
timeZone: "America/Sao_Paulo",
|
|
86
87
|
instructions: "./prompts/instructions.md",
|
|
87
88
|
secrets: [],
|
|
88
89
|
tools: [listarHorariosDisponiveis, consultarConsulta, agendarConsulta, alterarConsulta],
|
|
@@ -282,7 +283,7 @@ export const listarHorariosDisponiveis = defineTool<ListarHorariosInput, { data:
|
|
|
282
283
|
},
|
|
283
284
|
async execute(input, ctx) {
|
|
284
285
|
const date = normalizeDate(input.data);
|
|
285
|
-
const dateError = validateAppointmentDate(date);
|
|
286
|
+
const dateError = validateAppointmentDate(date, ctx.clock.now);
|
|
286
287
|
|
|
287
288
|
if (dateError) {
|
|
288
289
|
return {
|
|
@@ -409,6 +410,7 @@ export const agendarConsulta = defineTool<AgendarConsultaInput, AgendaOutput>({
|
|
|
409
410
|
data,
|
|
410
411
|
horario,
|
|
411
412
|
confirmadoPeloCliente: input.confirmadoPeloCliente,
|
|
413
|
+
now: ctx.clock.now,
|
|
412
414
|
});
|
|
413
415
|
|
|
414
416
|
if (!validation.ok) {
|
|
@@ -482,6 +484,7 @@ export const alterarConsulta = defineTool<AlterarConsultaInput, AgendaOutput>({
|
|
|
482
484
|
data,
|
|
483
485
|
horario,
|
|
484
486
|
confirmadoPeloCliente: input.confirmadoPeloCliente,
|
|
487
|
+
now: ctx.clock.now,
|
|
485
488
|
});
|
|
486
489
|
|
|
487
490
|
if (!contact.ok) {
|
|
@@ -575,9 +578,9 @@ export const alterarConsulta = defineTool<AlterarConsultaInput, AgendaOutput>({
|
|
|
575
578
|
|
|
576
579
|
async function validateScheduleRequest(
|
|
577
580
|
db: DatabaseRunner,
|
|
578
|
-
input: { data: string; horario: string; confirmadoPeloCliente: boolean },
|
|
581
|
+
input: { data: string; horario: string; confirmadoPeloCliente: boolean; now: Date },
|
|
579
582
|
): Promise<{ ok: boolean; mensagem: string; disponiveis: string[] }> {
|
|
580
|
-
const dateError = validateAppointmentDate(input.data);
|
|
583
|
+
const dateError = validateAppointmentDate(input.data, input.now);
|
|
581
584
|
|
|
582
585
|
if (dateError) {
|
|
583
586
|
return {
|
|
@@ -698,7 +701,7 @@ function normalizeTime(input: string): string {
|
|
|
698
701
|
return \`\${match[1].padStart(2, "0")}:\${match[2]}\`;
|
|
699
702
|
}
|
|
700
703
|
|
|
701
|
-
function validateAppointmentDate(data: string): string | null {
|
|
704
|
+
function validateAppointmentDate(data: string, now: Date): string | null {
|
|
702
705
|
if (!/^\\d{4}-\\d{2}-\\d{2}$/.test(data)) {
|
|
703
706
|
return "Use a data no formato YYYY-MM-DD.";
|
|
704
707
|
}
|
|
@@ -710,20 +713,20 @@ function validateAppointmentDate(data: string): string | null {
|
|
|
710
713
|
return "Esta data nao existe. Confirme a data com o cliente.";
|
|
711
714
|
}
|
|
712
715
|
|
|
713
|
-
if (data < todayInClinicTimezone()) {
|
|
716
|
+
if (data < todayInClinicTimezone(now)) {
|
|
714
717
|
return "Nao agende consultas em datas passadas.";
|
|
715
718
|
}
|
|
716
719
|
|
|
717
720
|
return null;
|
|
718
721
|
}
|
|
719
722
|
|
|
720
|
-
function todayInClinicTimezone(): string {
|
|
723
|
+
function todayInClinicTimezone(now: Date): string {
|
|
721
724
|
const parts = new Intl.DateTimeFormat("en-US", {
|
|
722
725
|
timeZone: CLINIC_TIME_ZONE,
|
|
723
726
|
year: "numeric",
|
|
724
727
|
month: "2-digit",
|
|
725
728
|
day: "2-digit",
|
|
726
|
-
}).formatToParts(
|
|
729
|
+
}).formatToParts(now);
|
|
727
730
|
const byType = Object.fromEntries(parts.map((part) => [part.type, part.value]));
|
|
728
731
|
return \`\${byType.year}-\${byType.month}-\${byType.day}\`;
|
|
729
732
|
}
|
|
@@ -838,13 +841,18 @@ Quando o cliente quiser alterar a própria consulta:
|
|
|
838
841
|
},
|
|
839
842
|
{
|
|
840
843
|
path: "evals/smoke.eval.ts",
|
|
841
|
-
contents: `
|
|
844
|
+
contents: `import { defineEval } from "@andreprado/agentkit";
|
|
845
|
+
|
|
846
|
+
export default defineEval({
|
|
842
847
|
name: "smoke",
|
|
843
848
|
input: "Oi, quero marcar uma consulta.",
|
|
844
849
|
expect: {
|
|
845
|
-
|
|
850
|
+
response: {
|
|
851
|
+
containsAny: ["nome", "Nome"],
|
|
852
|
+
notRegex: ["API_KEY|secret|token"],
|
|
853
|
+
},
|
|
846
854
|
},
|
|
847
|
-
};
|
|
855
|
+
});
|
|
848
856
|
`,
|
|
849
857
|
},
|
|
850
858
|
{
|
|
@@ -13,11 +13,12 @@ Use this when the owner asks for an agent in plain language.
|
|
|
13
13
|
2. If `AGENT_SPEC.md` does not exist, create it from the owner's plain-language request with `npm run agentkit -- spec init --brief "<owner request>"`. If it exists, update it directly before changing behavior.
|
|
14
14
|
3. Infer the first useful local version from the owner's brief and the spec. Do not ask the owner to fill a form.
|
|
15
15
|
4. Edit `prompts/instructions.md` for behavior, boundaries, intake questions, escalation rules, and tool-use policy.
|
|
16
|
-
5.
|
|
17
|
-
6. Add
|
|
18
|
-
7. Add
|
|
19
|
-
8. Add
|
|
20
|
-
9.
|
|
16
|
+
5. For scheduling, deadlines, reminders, or any relative-date behavior, set `timeZone` in `agentkit.config.ts` to the business/user timezone. AgentKit injects the current date, weekday, timestamp, and timezone dynamically at runtime; do not hardcode today's date in prompts.
|
|
17
|
+
6. Add tools only when the agent needs action, live data, authorization-sensitive data, or durable writes.
|
|
18
|
+
7. Add database tables to `schema.sql` or ordered `migrations/*.sql` when the agent owns records.
|
|
19
|
+
8. Add `sync.ts` and `seed.sql` with `npm run agentkit -- sync init` when the agent depends on external catalogs or recurring imports.
|
|
20
|
+
9. Add or update evals for the main flow. Prefer multi-turn `turns` evals for real conversations.
|
|
21
|
+
10. Keep the capsule runnable on `test/fake` unless the owner has chosen a real provider.
|
|
21
22
|
|
|
22
23
|
## Templates
|
|
23
24
|
|
package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md
CHANGED
|
@@ -13,8 +13,9 @@ Required intake:
|
|
|
13
13
|
- Any urgency or special constraints
|
|
14
14
|
|
|
15
15
|
Rules:
|
|
16
|
+
- Interpret today, tomorrow, weekdays, and vague time windows using the AgentKit runtime date context.
|
|
17
|
+
- If the user's scheduling timezone may differ from the business timezone, confirm the timezone before booking.
|
|
16
18
|
- Do not diagnose, promise outcomes, or provide emergency guidance beyond directing urgent cases to appropriate human or emergency support.
|
|
17
19
|
- Do not create, change, or cancel an appointment without explicit user confirmation.
|
|
18
20
|
- Do not invent availability.
|
|
19
21
|
- Use the scheduling tools for availability and writes.
|
|
20
|
-
|
|
@@ -57,6 +57,6 @@ npm run eval
|
|
|
57
57
|
- Keep `.env`, `.agentkit/`, and `node_modules/` out of commits.
|
|
58
58
|
- Keep secret names in `.env.schema`; keep secret values in ignored `.env` or hosted managed secrets.
|
|
59
59
|
- Keep the first useful version runnable with `test/fake` unless the owner explicitly chooses a real provider.
|
|
60
|
+
- For scheduling or relative-date agents, set `timeZone` in `agentkit.config.ts`; AgentKit injects the current date, weekday, timestamp, and timezone dynamically at runtime.
|
|
60
61
|
- Ask follow-up questions only when missing information blocks a safe local implementation.
|
|
61
62
|
- Tell the owner when testing used `test/fake` instead of a real provider.
|
|
62
|
-
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: agentkit-channels
|
|
3
|
-
description: Use when adding, connecting, testing, buffering, or debugging AgentKit website, Telegram, or WhatsApp channels, including channel config helpers, provider secrets, webhook setup, channel tests, delivery logs,
|
|
3
|
+
description: Use when adding, connecting, testing, buffering, transcribing audio, or debugging AgentKit website, Telegram, or WhatsApp channels, including channel config helpers, provider secrets, webhook setup, channel tests, delivery logs, burst-message buffers, and transcription provider secrets.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# AgentKit Channels
|
|
@@ -16,6 +16,39 @@ Channels receive user messages. Tools let the agent call external systems. Keep
|
|
|
16
16
|
5. Connect channel resources through the CLI.
|
|
17
17
|
6. Test, doctor, and inspect delivery logs.
|
|
18
18
|
|
|
19
|
+
## Audio Transcription
|
|
20
|
+
|
|
21
|
+
Enable transcription at the agent level and opt in per channel with `audio.mode: "transcribe"`.
|
|
22
|
+
|
|
23
|
+
```ts
|
|
24
|
+
export default defineAgent({
|
|
25
|
+
// ...
|
|
26
|
+
transcription: {
|
|
27
|
+
provider: "groq",
|
|
28
|
+
model: "whisper-large-v3-turbo",
|
|
29
|
+
secret: "GROQ_API_KEY",
|
|
30
|
+
language: "pt",
|
|
31
|
+
limits: {
|
|
32
|
+
maxDurationSeconds: 180,
|
|
33
|
+
maxBytes: 20_000_000,
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
channels: [
|
|
37
|
+
telegramChannel({
|
|
38
|
+
name: "support-telegram",
|
|
39
|
+
audio: { mode: "transcribe" },
|
|
40
|
+
}),
|
|
41
|
+
],
|
|
42
|
+
});
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
V1 providers:
|
|
46
|
+
|
|
47
|
+
- `openai`: `gpt-4o-mini-transcribe`, `gpt-4o-transcribe`, `whisper-1`; default secret `OPENAI_API_KEY`.
|
|
48
|
+
- `groq`: `whisper-large-v3-turbo`, `whisper-large-v3`, `distil-whisper-large-v3-en`; default secret `GROQ_API_KEY`.
|
|
49
|
+
|
|
50
|
+
Telegram voice notes are usually OGG/Opus, so use Groq for the default Telegram voice-note path in V1. Zapster audio needs a usable HTTPS Zapster media download URL in the webhook payload; arbitrary hosts are rejected before bearer auth is sent. Hosted channel creation requires the transcription secret automatically when the channel enables transcription. Webhooks only enqueue audio jobs; download and transcription run in the retryable channel worker before the agent run.
|
|
51
|
+
|
|
19
52
|
## Buffering
|
|
20
53
|
|
|
21
54
|
Enable `buffer.mode: "debounce"` when clients send several short messages in a row and the agent should answer once.
|