@andreprado/agentkit 0.1.0-alpha.14 → 0.1.0-alpha.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/README.md +2 -0
  2. package/docs/guides/add-channel.md +63 -0
  3. package/docs/guides/channel-security.md +32 -0
  4. package/docs/guides/connect-telegram.md +58 -0
  5. package/docs/guides/connect-whatsapp-zapster.md +65 -0
  6. package/docs/guides/run-evals.md +73 -25
  7. package/docs/llms-full.txt +24 -9
  8. package/docs/llms.txt +2 -0
  9. package/package.json +1 -1
  10. package/src/cli/cloud-client.ts +30 -10
  11. package/src/cli/commands/channels.ts +2 -0
  12. package/src/cli/deploy-readiness.ts +32 -11
  13. package/src/cli/index.ts +20 -6
  14. package/src/cloud/client.ts +4 -3
  15. package/src/cloud/contracts.ts +1 -1
  16. package/src/create-project.ts +1 -1
  17. package/src/index.ts +110 -1
  18. package/src/providers/pi.ts +14 -1
  19. package/src/providers/test.ts +36 -0
  20. package/src/runtime/channel-test-harness.ts +2 -0
  21. package/src/runtime/channels/telegram.ts +326 -10
  22. package/src/runtime/channels/whatsapp-zapster.ts +319 -0
  23. package/src/runtime/channels.ts +47 -1
  24. package/src/runtime/chat.ts +59 -42
  25. package/src/runtime/config.ts +96 -4
  26. package/src/runtime/core/manifest.ts +35 -3
  27. package/src/runtime/deploy-readiness.ts +3 -3
  28. package/src/runtime/dev-server.ts +243 -17
  29. package/src/runtime/env.ts +8 -3
  30. package/src/runtime/evals.ts +404 -69
  31. package/src/runtime/inspect.ts +46 -0
  32. package/src/runtime/prompt-context.ts +141 -0
  33. package/src/runtime/runtime-contract.ts +17 -7
  34. package/src/runtime/targets/cloudflare/build.ts +25 -3
  35. package/src/runtime/targets/container/server.ts +1 -1
  36. package/src/runtime/targets/vps/deploy.ts +25 -8
  37. package/src/runtime/tool-runner.ts +7 -0
  38. package/src/runtime/tools.ts +8 -2
  39. package/src/runtime/transcription.ts +483 -0
  40. package/src/templates/blank.ts +8 -3
  41. package/src/templates/dentista.ts +18 -10
  42. package/src/templates/skills/agentkit-build-agent/SKILL.md +6 -5
  43. package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
  44. package/src/templates/skills/agentkit-capsule/SKILL.md +1 -1
  45. package/src/templates/skills/agentkit-channels/SKILL.md +34 -1
  46. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +13 -0
  47. package/src/templates/skills/agentkit-channels/references/telegram.md +32 -0
  48. package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +29 -0
  49. package/src/templates/skills/agentkit-evals/SKILL.md +53 -13
  50. package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
  51. package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
  52. package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
  53. package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
  54. package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
  55. package/src/templates/skills/agentkit-tools/SKILL.md +2 -1
  56. package/src/templates/support.ts +8 -3
@@ -1,12 +1,12 @@
1
- import { mkdir, readdir, stat, writeFile } from "node:fs/promises";
1
+ import { mkdir, readdir, writeFile } from "node:fs/promises";
2
2
  import type { Dirent } from "node:fs";
3
- import { dirname, join, relative, resolve } from "node:path";
3
+ import { dirname, isAbsolute, join, relative, resolve } from "node:path";
4
4
  import { pathToFileURL } from "node:url";
5
5
 
6
6
  import type { AgentRunResult } from "./chat";
7
7
  import { runAgentMessageFromCwd } from "./chat";
8
8
  import { findAgentCapsuleRoot, loadAgentCapsule } from "./config";
9
- import { AgentKitError } from "./errors";
9
+ import { AgentKitError, isAgentKitError } from "./errors";
10
10
  import { openCapsuleStore, type StoredToolCall } from "../storage/sqlite";
11
11
  import { getConversationTraceFromCwd } from "./traces";
12
12
 
@@ -25,38 +25,76 @@ export type EvalResult = {
25
25
  output: string;
26
26
  };
27
27
 
28
- type EvalCase = {
28
+ export type EvalCase = {
29
29
  name?: string;
30
30
  input?: string;
31
+ now?: string | Date;
31
32
  turns?: EvalTurn[];
32
33
  expect?: EvalExpect;
33
34
  };
34
35
 
35
- type EvalTurn =
36
+ export type EvalTurn =
36
37
  | string
37
38
  | {
38
39
  input: string;
39
40
  expect?: EvalExpect;
40
41
  };
41
42
 
42
- type EvalExpect = {
43
+ export type EvalExpect = EvalResponseExpectation & {
44
+ response?: EvalResponseExpectation;
45
+ tools?: EvalToolsExpectation;
46
+ tool_call?: ToolExpectation | ToolExpectation[];
47
+ toolCall?: ToolExpectation | ToolExpectation[];
48
+ tool_calls?: ToolExpectation | ToolExpectation[];
49
+ toolCalls?: ToolExpectation | ToolExpectation[];
50
+ tool_call_count?: number;
51
+ toolCallCount?: number;
52
+ tool_call_order?: string[];
53
+ toolCallOrder?: string[];
54
+ persisted_tool_call?: ToolExpectation | ToolExpectation[];
55
+ persistedToolCall?: ToolExpectation | ToolExpectation[];
56
+ persisted_tool_calls?: ToolExpectation | ToolExpectation[];
57
+ persistedToolCalls?: ToolExpectation | ToolExpectation[];
58
+ };
59
+
60
+ export type EvalResponseExpectation = {
43
61
  contains?: string | string[];
62
+ contains_all?: string | string[];
63
+ containsAll?: string | string[];
64
+ contains_any?: string | string[];
65
+ containsAny?: string | string[];
66
+ case_insensitive_contains?: string | string[];
67
+ caseInsensitiveContains?: string | string[];
44
68
  not_contains?: string | string[];
45
69
  notContains?: string | string[];
46
70
  regex?: string | string[];
47
71
  matches_regex?: string | string[];
48
72
  matchesRegex?: string | string[];
49
- tool_call?: ToolExpectation | ToolExpectation[];
50
- toolCall?: ToolExpectation | ToolExpectation[];
51
- tool_calls?: ToolExpectation | ToolExpectation[];
52
- toolCalls?: ToolExpectation | ToolExpectation[];
73
+ not_regex?: string | string[];
74
+ notRegex?: string | string[];
75
+ max_length?: number;
76
+ maxLength?: number;
77
+ };
78
+
79
+ export type EvalToolsExpectation =
80
+ | ToolExpectation
81
+ | ToolExpectation[]
82
+ | EvalToolsContainerExpectation;
83
+
84
+ export type EvalToolsContainerExpectation = {
85
+ persisted?: ToolExpectation | ToolExpectation[];
53
86
  persisted_tool_call?: ToolExpectation | ToolExpectation[];
54
87
  persistedToolCall?: ToolExpectation | ToolExpectation[];
55
88
  persisted_tool_calls?: ToolExpectation | ToolExpectation[];
56
89
  persistedToolCalls?: ToolExpectation | ToolExpectation[];
90
+ called?: string | string[];
91
+ called_once?: string | string[];
92
+ calledOnce?: string | string[];
93
+ count?: number;
94
+ order?: string[];
57
95
  };
58
96
 
59
- type ToolExpectation =
97
+ export type ToolExpectation =
60
98
  | string
61
99
  | {
62
100
  name?: string;
@@ -83,6 +121,10 @@ export type EvalFromConversationResult = {
83
121
  turns: number;
84
122
  };
85
123
 
124
+ export function defineEval<const T extends EvalCase>(evalCase: T): T {
125
+ return evalCase;
126
+ }
127
+
86
128
  export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
87
129
  const root = await findAgentCapsuleRoot(cwd);
88
130
  const evalFiles = await findEvalFiles(join(root, "evals"));
@@ -94,15 +136,29 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
94
136
  const results: EvalResult[] = [];
95
137
 
96
138
  for (const file of evalFiles) {
97
- const evalCase = await loadEvalCase(file);
98
- const run = await runEvalCase(root, relative(root, file), evalCase);
139
+ const evalFile = relative(root, file);
140
+ let name = evalFile;
141
+ let failures: string[];
142
+ let output: string;
143
+
144
+ try {
145
+ const evalCase = await loadEvalCase(file);
146
+ const run = await runEvalCase(root, evalFile, evalCase);
147
+
148
+ name = evalCase.name ?? evalFile;
149
+ failures = run.failures;
150
+ output = run.output;
151
+ } catch (error) {
152
+ failures = [formatEvalFailure(error)];
153
+ output = "";
154
+ }
99
155
 
100
156
  results.push({
101
- name: evalCase.name ?? relative(root, file),
102
- file: relative(root, file),
103
- passed: run.failures.length === 0,
104
- failures: run.failures,
105
- output: run.output,
157
+ name,
158
+ file: evalFile,
159
+ passed: failures.length === 0,
160
+ failures,
161
+ output,
106
162
  });
107
163
  }
108
164
 
@@ -136,24 +192,29 @@ export async function writeEvalFromConversation(
136
192
  );
137
193
  }
138
194
 
139
- const file = resolve(root, input.out ?? join("evals", `replay-${slugify(trace.title ?? trace.id)}.eval.ts`));
195
+ const file = resolveEvalOutputPath(root, input.out, trace.title ?? trace.id);
196
+ const source = `import { defineEval } from "@andreprado/agentkit";
140
197
 
141
- if (!input.force && await pathExists(file)) {
142
- throw new AgentKitError(
143
- "validation_error",
144
- `${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
145
- );
146
- }
147
-
148
- await mkdir(dirname(file), { recursive: true });
149
- await writeFile(
150
- file,
151
- `export default {
198
+ export default defineEval({
152
199
  name: ${JSON.stringify(input.name ?? `replay ${trace.title ?? trace.id}`)},
153
200
  turns: ${formatEvalTurns(turns)},
154
- };
155
- `,
156
- );
201
+ });
202
+ `;
203
+
204
+ await mkdir(dirname(file), { recursive: true });
205
+
206
+ try {
207
+ await writeFile(file, source, { flag: input.force ? "w" : "wx" });
208
+ } catch (error) {
209
+ if (isNodeError(error) && error.code === "EEXIST") {
210
+ throw new AgentKitError(
211
+ "validation_error",
212
+ `${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
213
+ );
214
+ }
215
+
216
+ throw error;
217
+ }
157
218
 
158
219
  return {
159
220
  root,
@@ -180,29 +241,61 @@ async function runEvalCase(
180
241
  const conversationId = `eval_${crypto.randomUUID()}`;
181
242
  const failures: string[] = [];
182
243
  let output = "";
244
+ const now = resolveEvalNow(file, evalCase.now);
183
245
 
184
246
  for (const [index, turn] of turns.entries()) {
185
- const run = await runAgentMessageFromCwd(root, {
186
- message: turn.input,
187
- conversationId,
188
- runtime: {
189
- environment: "eval",
190
- invocation: "eval",
191
- },
192
- });
193
- const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
194
- const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
247
+ try {
248
+ const run = await runAgentMessageFromCwd(root, {
249
+ message: turn.input,
250
+ conversationId,
251
+ now,
252
+ runtime: {
253
+ environment: "eval",
254
+ invocation: "eval",
255
+ },
256
+ });
257
+ const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
258
+ const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
259
+
260
+ for (const failure of turnFailures) {
261
+ failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
262
+ }
195
263
 
196
- for (const failure of turnFailures) {
197
- failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
264
+ output = run.message.content;
265
+ } catch (error) {
266
+ failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
267
+ break;
198
268
  }
199
-
200
- output = run.message.content;
201
269
  }
202
270
 
203
271
  return { failures, output };
204
272
  }
205
273
 
274
+ function resolveEvalNow(file: string, value: EvalCase["now"]): Date | undefined {
275
+ if (value === undefined) {
276
+ return undefined;
277
+ }
278
+
279
+ if (typeof value === "string" && !hasExplicitIsoOffset(value)) {
280
+ throw new AgentKitError(
281
+ "validation_error",
282
+ `${file} now must be an ISO timestamp with an explicit timezone offset, such as "2026-02-04T02:30:00.000Z" or "2026-02-04T02:30:00-05:00".`,
283
+ );
284
+ }
285
+
286
+ const date = value instanceof Date ? new Date(value.getTime()) : new Date(value);
287
+
288
+ if (Number.isNaN(date.getTime())) {
289
+ throw new AgentKitError("validation_error", `${file} now must be a valid ISO timestamp or Date when provided.`);
290
+ }
291
+
292
+ return date;
293
+ }
294
+
295
+ function hasExplicitIsoOffset(value: string): boolean {
296
+ return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.test(value.trim());
297
+ }
298
+
206
299
  function normalizeEvalTurns(evalCase: EvalCase): Array<{ input: string; expect?: EvalExpect }> {
207
300
  if (evalCase.turns !== undefined) {
208
301
  if (!Array.isArray(evalCase.turns)) {
@@ -303,12 +396,89 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
303
396
  const failures: string[] = [];
304
397
  const content = run.message.content;
305
398
 
306
- for (const expected of list(expect.contains)) {
399
+ for (const responseExpectation of responseExpectations(expect)) {
400
+ failures.push(...evaluateResponseExpectation(responseExpectation, content));
401
+ }
402
+
403
+ const toolAssertions = collectToolAssertions(expect);
404
+
405
+ for (const toolExpectation of toolAssertions.persisted) {
406
+ const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
407
+
408
+ if (!match) {
409
+ failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
410
+ }
411
+ }
412
+
413
+ for (const expectedName of toolAssertions.called) {
414
+ if (!persistedToolCalls.some((toolCall) => toolCall.name === expectedName)) {
415
+ failures.push(`expected persisted tool call named ${JSON.stringify(expectedName)}`);
416
+ }
417
+ }
418
+
419
+ for (const expectedName of toolAssertions.calledOnce) {
420
+ const count = persistedToolCalls.filter((toolCall) => toolCall.name === expectedName).length;
421
+
422
+ if (count !== 1) {
423
+ failures.push(`expected persisted tool call ${JSON.stringify(expectedName)} exactly once, got ${count}`);
424
+ }
425
+ }
426
+
427
+ for (const expectedCount of toolAssertions.counts) {
428
+ if (!Number.isInteger(expectedCount) || expectedCount < 0) {
429
+ failures.push(`expected tool call count to be a non-negative integer, got ${JSON.stringify(expectedCount)}`);
430
+ continue;
431
+ }
432
+
433
+ if (persistedToolCalls.length !== expectedCount) {
434
+ failures.push(`expected ${expectedCount} persisted tool call(s), got ${persistedToolCalls.length}`);
435
+ }
436
+ }
437
+
438
+ for (const expectedOrder of toolAssertions.orders) {
439
+ if (!namesContainSubsequence(persistedToolNames(persistedToolCalls), expectedOrder)) {
440
+ failures.push(
441
+ `expected persisted tool call order ${JSON.stringify(expectedOrder)}, got ${JSON.stringify(
442
+ persistedToolNames(persistedToolCalls),
443
+ )}`,
444
+ );
445
+ }
446
+ }
447
+
448
+ return failures;
449
+ }
450
+
451
+ function responseExpectations(expect: EvalExpect): EvalResponseExpectation[] {
452
+ return [expect, expect.response].filter((value): value is EvalResponseExpectation => value !== undefined);
453
+ }
454
+
455
+ function evaluateResponseExpectation(expect: EvalResponseExpectation, content: string): string[] {
456
+ const failures: string[] = [];
457
+
458
+ for (const expected of [
459
+ ...list(expect.contains),
460
+ ...list(expect.contains_all),
461
+ ...list(expect.containsAll),
462
+ ]) {
307
463
  if (!content.includes(expected)) {
308
464
  failures.push(`expected output to contain ${JSON.stringify(expected)}`);
309
465
  }
310
466
  }
311
467
 
468
+ const containsAny = [...list(expect.contains_any), ...list(expect.containsAny)];
469
+
470
+ if (containsAny.length > 0 && !containsAny.some((expected) => content.includes(expected))) {
471
+ failures.push(`expected output to contain any of ${JSON.stringify(containsAny)}`);
472
+ }
473
+
474
+ const lowerContent = content.toLowerCase();
475
+
476
+ for (const expected of [...list(expect.case_insensitive_contains), ...list(expect.caseInsensitiveContains)]) {
477
+ if (!lowerContent.includes(expected.toLowerCase())) {
478
+ failures.push(`expected output to contain ${JSON.stringify(expected)} case-insensitively`);
479
+ }
480
+ }
481
+
312
482
  for (const forbidden of [...list(expect.not_contains), ...list(expect.notContains)]) {
313
483
  if (content.includes(forbidden)) {
314
484
  failures.push(`expected output not to contain ${JSON.stringify(forbidden)}`);
@@ -321,28 +491,140 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
321
491
  }
322
492
  }
323
493
 
324
- const toolExpectations = [
325
- ...toolList(expect.tool_call),
326
- ...toolList(expect.toolCall),
327
- ...toolList(expect.tool_calls),
328
- ...toolList(expect.toolCalls),
329
- ...toolList(expect.persisted_tool_call),
330
- ...toolList(expect.persistedToolCall),
331
- ...toolList(expect.persisted_tool_calls),
332
- ...toolList(expect.persistedToolCalls),
333
- ];
494
+ for (const pattern of [...list(expect.not_regex), ...list(expect.notRegex)]) {
495
+ if (new RegExp(pattern).test(content)) {
496
+ failures.push(`expected output not to match /${pattern}/`);
497
+ }
498
+ }
334
499
 
335
- for (const toolExpectation of toolExpectations) {
336
- const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
500
+ for (const expectedMaxLength of [expect.max_length, expect.maxLength]) {
501
+ if (expectedMaxLength === undefined) {
502
+ continue;
503
+ }
337
504
 
338
- if (!match) {
339
- failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
505
+ if (!Number.isInteger(expectedMaxLength) || expectedMaxLength < 0) {
506
+ failures.push(`expected response maxLength to be a non-negative integer, got ${JSON.stringify(expectedMaxLength)}`);
507
+ continue;
508
+ }
509
+
510
+ if (content.length > expectedMaxLength) {
511
+ failures.push(`expected output length to be <= ${expectedMaxLength}, got ${content.length}`);
340
512
  }
341
513
  }
342
514
 
343
515
  return failures;
344
516
  }
345
517
 
518
+ type ToolAssertions = {
519
+ persisted: ToolExpectation[];
520
+ called: string[];
521
+ calledOnce: string[];
522
+ counts: number[];
523
+ orders: string[][];
524
+ };
525
+
526
+ function collectToolAssertions(expect: EvalExpect): ToolAssertions {
527
+ const assertions: ToolAssertions = {
528
+ persisted: [
529
+ ...toolList(expect.tool_call),
530
+ ...toolList(expect.toolCall),
531
+ ...toolList(expect.tool_calls),
532
+ ...toolList(expect.toolCalls),
533
+ ...toolList(expect.persisted_tool_call),
534
+ ...toolList(expect.persistedToolCall),
535
+ ...toolList(expect.persisted_tool_calls),
536
+ ...toolList(expect.persistedToolCalls),
537
+ ],
538
+ called: [],
539
+ calledOnce: [],
540
+ counts: [],
541
+ orders: [],
542
+ };
543
+
544
+ if (expect.tool_call_count !== undefined) {
545
+ assertions.counts.push(expect.tool_call_count);
546
+ }
547
+
548
+ if (expect.toolCallCount !== undefined) {
549
+ assertions.counts.push(expect.toolCallCount);
550
+ }
551
+
552
+ if (expect.tool_call_order !== undefined) {
553
+ assertions.orders.push(expect.tool_call_order);
554
+ }
555
+
556
+ if (expect.toolCallOrder !== undefined) {
557
+ assertions.orders.push(expect.toolCallOrder);
558
+ }
559
+
560
+ appendNestedToolAssertions(assertions, expect.tools);
561
+
562
+ return assertions;
563
+ }
564
+
565
+ function appendNestedToolAssertions(assertions: ToolAssertions, tools: EvalToolsExpectation | undefined): void {
566
+ if (tools === undefined) {
567
+ return;
568
+ }
569
+
570
+ if (!isToolsContainer(tools)) {
571
+ assertions.persisted.push(...toolList(tools));
572
+ return;
573
+ }
574
+
575
+ assertions.persisted.push(
576
+ ...toolList(tools.persisted),
577
+ ...toolList(tools.persisted_tool_call),
578
+ ...toolList(tools.persistedToolCall),
579
+ ...toolList(tools.persisted_tool_calls),
580
+ ...toolList(tools.persistedToolCalls),
581
+ );
582
+ assertions.called.push(...list(tools.called));
583
+ assertions.calledOnce.push(...list(tools.called_once), ...list(tools.calledOnce));
584
+
585
+ if (tools.count !== undefined) {
586
+ assertions.counts.push(tools.count);
587
+ }
588
+
589
+ if (tools.order !== undefined) {
590
+ assertions.orders.push(tools.order);
591
+ }
592
+ }
593
+
594
+ function isToolsContainer(value: EvalToolsExpectation): value is EvalToolsContainerExpectation {
595
+ if (!isRecord(value)) {
596
+ return false;
597
+ }
598
+
599
+ if (
600
+ "name" in value ||
601
+ "input" in value ||
602
+ "output" in value ||
603
+ "rendered" in value ||
604
+ "status" in value ||
605
+ "visibility" in value
606
+ ) {
607
+ return false;
608
+ }
609
+
610
+ if (Object.keys(value).length === 0) {
611
+ return true;
612
+ }
613
+
614
+ return (
615
+ "persisted" in value ||
616
+ "persisted_tool_call" in value ||
617
+ "persistedToolCall" in value ||
618
+ "persisted_tool_calls" in value ||
619
+ "persistedToolCalls" in value ||
620
+ "called" in value ||
621
+ "called_once" in value ||
622
+ "calledOnce" in value ||
623
+ "count" in value ||
624
+ "order" in value
625
+ );
626
+ }
627
+
346
628
  function list(value: string | string[] | undefined): string[] {
347
629
  if (value === undefined) {
348
630
  return [];
@@ -351,6 +633,30 @@ function list(value: string | string[] | undefined): string[] {
351
633
  return Array.isArray(value) ? value : [value];
352
634
  }
353
635
 
636
+ function persistedToolNames(toolCalls: ToolCallSnapshot[]): string[] {
637
+ return toolCalls.map((toolCall) => String(toolCall.name ?? ""));
638
+ }
639
+
640
+ function namesContainSubsequence(actual: string[], expected: string[]): boolean {
641
+ if (expected.length === 0) {
642
+ return true;
643
+ }
644
+
645
+ let actualIndex = 0;
646
+
647
+ for (const expectedName of expected) {
648
+ actualIndex = actual.findIndex((actualName, index) => index >= actualIndex && actualName === expectedName);
649
+
650
+ if (actualIndex === -1) {
651
+ return false;
652
+ }
653
+
654
+ actualIndex += 1;
655
+ }
656
+
657
+ return true;
658
+ }
659
+
354
660
  function toolList(value: ToolExpectation | ToolExpectation[] | undefined): ToolExpectation[] {
355
661
  if (value === undefined) {
356
662
  return [];
@@ -428,7 +734,9 @@ function replayTurnsFromMessages(
428
734
  ...(nextAssistant
429
735
  ? {
430
736
  expect: {
431
- contains: nextAssistant.content,
737
+ response: {
738
+ contains: nextAssistant.content,
739
+ },
432
740
  },
433
741
  }
434
742
  : {}),
@@ -455,13 +763,40 @@ function slugify(value: string): string {
455
763
  return slug || "conversation";
456
764
  }
457
765
 
458
- async function pathExists(path: string): Promise<boolean> {
459
- try {
460
- await stat(path);
461
- return true;
462
- } catch {
463
- return false;
766
+ function resolveEvalOutputPath(root: string, out: string | undefined, title: string): string {
767
+ const outputPath = out ?? join("evals", `replay-${slugify(title)}.eval.ts`);
768
+
769
+ if (isAbsolute(outputPath)) {
770
+ throw new AgentKitError("validation_error", "eval --out must be a relative path inside the Agent Capsule.");
771
+ }
772
+
773
+ const file = resolve(root, outputPath);
774
+ const relativePath = relative(root, file);
775
+
776
+ if (relativePath === "" || relativePath.startsWith("..") || isAbsolute(relativePath)) {
777
+ throw new AgentKitError("validation_error", "eval --out must stay inside the Agent Capsule.");
778
+ }
779
+
780
+ if (!/\.(eval|spec)\.[cm]?[tj]s$/.test(file)) {
781
+ throw new AgentKitError(
782
+ "validation_error",
783
+ "eval --out must end with .eval.ts, .eval.js, .spec.ts, or another AgentKit eval/spec extension.",
784
+ );
464
785
  }
786
+
787
+ return file;
788
+ }
789
+
790
+ function formatEvalFailure(error: unknown): string {
791
+ if (isAgentKitError(error)) {
792
+ return `${error.code}: ${error.message}`;
793
+ }
794
+
795
+ if (error instanceof Error) {
796
+ return `runtime_error: ${error.message}`;
797
+ }
798
+
799
+ return `runtime_error: ${String(error)}`;
465
800
  }
466
801
 
467
802
  function jsonContains(actual: unknown, expected: unknown): boolean {
@@ -6,6 +6,8 @@ import { loadAgentCapsule } from "./config";
6
6
  import { loadCapsuleEnv } from "./env";
7
7
  import { resolveKnowledgeConfig } from "./knowledge/config";
8
8
  import { KNOWLEDGE_SEARCH_TOOL_NAME } from "./knowledge/tool";
9
+ import { resolveRuntimeTimeZone } from "./prompt-context";
10
+ import { resolveTranscriptionConfig } from "./transcription";
9
11
 
10
12
  export type SecretState = "set" | "missing";
11
13
 
@@ -13,9 +15,26 @@ export type AgentInspectState = {
13
15
  agent: string;
14
16
  runtime: AgentRuntime;
15
17
  provider: AgentProvider;
18
+ timeZone: {
19
+ configured: string | null;
20
+ effective: string;
21
+ };
16
22
  prompt: string;
17
23
  tools: string[];
18
24
  channels: AgentInspectChannel[];
25
+ transcription: {
26
+ enabled: boolean;
27
+ provider: string | null;
28
+ model: string | null;
29
+ secret: string | null;
30
+ language: string | null;
31
+ prompt: string | null;
32
+ limits: {
33
+ maxDurationSeconds: number | null;
34
+ maxBytes: number | null;
35
+ };
36
+ rawAudioTtlSeconds: number | null;
37
+ };
19
38
  knowledge: {
20
39
  enabled: boolean;
21
40
  sources: string[];
@@ -75,6 +94,7 @@ export type AgentInspectChannel = {
75
94
  provider: AgentChannel["provider"];
76
95
  secrets: string[];
77
96
  buffer?: AgentChannel["buffer"];
97
+ audio?: AgentChannel["audio"];
78
98
  };
79
99
 
80
100
  export async function inspectAgentCapsule(
@@ -91,6 +111,7 @@ export function buildInspectState(
91
111
  ): AgentInspectState {
92
112
  const declaredSecrets = new Set(capsule.config.secrets);
93
113
  const knowledge = resolveKnowledgeConfig(capsule.config);
114
+ const transcription = resolveTranscriptionConfig(capsule.config.transcription);
94
115
 
95
116
  for (const tool of capsule.config.tools ?? []) {
96
117
  for (const secret of tool.secrets ?? []) {
@@ -108,6 +129,10 @@ export function buildInspectState(
108
129
  declaredSecrets.add(knowledge.embedding.secret);
109
130
  }
110
131
 
132
+ if (transcription.secret) {
133
+ declaredSecrets.add(transcription.secret);
134
+ }
135
+
111
136
  const managedSecrets = managedCloudSecrets(capsule);
112
137
  const userSecrets = Array.from(declaredSecrets).sort();
113
138
  const secrets: Record<string, SecretState> = {};
@@ -122,6 +147,13 @@ export function buildInspectState(
122
147
  agent: capsule.config.name,
123
148
  runtime: capsule.config.runtime,
124
149
  provider: capsule.config.provider,
150
+ timeZone: {
151
+ configured: capsule.config.timeZone ?? null,
152
+ effective: resolveRuntimeTimeZone({
153
+ timeZone: capsule.config.timeZone,
154
+ env,
155
+ }),
156
+ },
125
157
  prompt: toCapsulePath(capsule.root, capsule.instructionsPath),
126
158
  tools: (capsule.config.tools ?? []).map((tool) => tool.name),
127
159
  channels: (capsule.config.channels ?? []).map((channel) => ({
@@ -130,7 +162,21 @@ export function buildInspectState(
130
162
  provider: channel.provider,
131
163
  secrets: [...channel.secrets],
132
164
  ...(channel.buffer ? { buffer: channel.buffer } : {}),
165
+ ...(channel.audio ? { audio: channel.audio } : {}),
133
166
  })),
167
+ transcription: {
168
+ enabled: transcription.enabled,
169
+ provider: transcription.enabled ? transcription.provider : null,
170
+ model: transcription.enabled ? transcription.model : null,
171
+ secret: transcription.secret,
172
+ language: transcription.language,
173
+ prompt: transcription.prompt,
174
+ limits: {
175
+ maxDurationSeconds: transcription.limits.maxDurationSeconds ?? null,
176
+ maxBytes: transcription.limits.maxBytes ?? null,
177
+ },
178
+ rawAudioTtlSeconds: transcription.rawAudioTtlSeconds,
179
+ },
134
180
  knowledge: {
135
181
  enabled: knowledge.enabled,
136
182
  sources: knowledge.sources.map((source) => source.value),