@andreprado/agentkit 0.1.0-alpha.9 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/README.md +18 -1
  2. package/docs/guides/add-channel.md +251 -7
  3. package/docs/guides/add-knowledge.md +10 -0
  4. package/docs/guides/add-managed-composio.md +165 -0
  5. package/docs/guides/add-tool.md +10 -3
  6. package/docs/guides/channel-security.md +162 -32
  7. package/docs/guides/connect-discord.md +178 -0
  8. package/docs/guides/connect-slack.md +126 -0
  9. package/docs/guides/connect-telegram.md +61 -1
  10. package/docs/guides/connect-whatsapp-evolution.md +121 -0
  11. package/docs/guides/connect-whatsapp-uazapi.md +139 -0
  12. package/docs/guides/connect-whatsapp-zapster.md +119 -16
  13. package/docs/guides/create-agent.md +31 -4
  14. package/docs/guides/debug-channel.md +159 -0
  15. package/docs/guides/improve-from-production.md +151 -0
  16. package/docs/guides/prepare-deploy.md +32 -14
  17. package/docs/guides/replay-production-traces.md +72 -0
  18. package/docs/guides/run-evals.md +95 -25
  19. package/docs/guides/security-rules.md +9 -5
  20. package/docs/guides/send-feedback.md +135 -0
  21. package/docs/guides/use-jev.md +67 -0
  22. package/docs/guides/use-provider.md +70 -3
  23. package/docs/llms-full.txt +295 -25
  24. package/docs/llms.txt +54 -7
  25. package/package.json +3 -7
  26. package/src/cli/args.ts +23 -2
  27. package/src/cli/cloud-client.ts +121 -9
  28. package/src/cli/commands/channels.ts +856 -36
  29. package/src/cli/commands/feedback.ts +438 -0
  30. package/src/cli/commands/provider.ts +47 -0
  31. package/src/cli/commands/transcribe.ts +171 -0
  32. package/src/cli/deploy-chat-ui.ts +232 -18
  33. package/src/cli/deploy-readiness.ts +227 -14
  34. package/src/cli/help.ts +67 -9
  35. package/src/cli/index.ts +740 -35
  36. package/src/cli/new-command.ts +41 -0
  37. package/src/cloud/client.ts +4 -3
  38. package/src/cloud/contracts.ts +1 -1
  39. package/src/create-project.ts +18 -35
  40. package/src/index.ts +565 -11
  41. package/src/providers/codex-auth.ts +111 -0
  42. package/src/providers/pi.ts +88 -19
  43. package/src/providers/test.ts +36 -0
  44. package/src/providers/types.ts +8 -0
  45. package/src/runtime/channel-test-harness.ts +21 -1
  46. package/src/runtime/channels/discord.ts +904 -0
  47. package/src/runtime/channels/generic-webhook.ts +682 -0
  48. package/src/runtime/channels/net-guard.ts +480 -0
  49. package/src/runtime/channels/provider-fetch.ts +54 -0
  50. package/src/runtime/channels/slack.ts +652 -0
  51. package/src/runtime/channels/telegram.ts +379 -15
  52. package/src/runtime/channels/whatsapp-evolution.ts +1330 -0
  53. package/src/runtime/channels/whatsapp-meta.ts +9 -0
  54. package/src/runtime/channels/whatsapp-uazapi.ts +1192 -0
  55. package/src/runtime/channels/whatsapp-zapster.ts +702 -40
  56. package/src/runtime/channels.ts +83 -3
  57. package/src/runtime/chat.ts +70 -44
  58. package/src/runtime/config.ts +512 -20
  59. package/src/runtime/core/manifest.ts +75 -5
  60. package/src/runtime/core/targets.ts +5 -5
  61. package/src/runtime/deploy-readiness.ts +34 -4
  62. package/src/runtime/dev-server.ts +639 -39
  63. package/src/runtime/env.ts +8 -3
  64. package/src/runtime/evals.ts +445 -74
  65. package/src/runtime/improve.ts +868 -0
  66. package/src/runtime/inspect.ts +173 -4
  67. package/src/runtime/integrations/composio.ts +425 -0
  68. package/src/runtime/knowledge/embeddings.ts +45 -7
  69. package/src/runtime/knowledge/ingest.ts +69 -6
  70. package/src/runtime/knowledge/retrieve.ts +25 -5
  71. package/src/runtime/knowledge/schema.ts +45 -1
  72. package/src/runtime/knowledge/vector.ts +30 -30
  73. package/src/runtime/prompt-context.ts +141 -0
  74. package/src/runtime/runtime-contract.ts +71 -7
  75. package/src/runtime/skills.ts +95 -0
  76. package/src/runtime/targets/cloudflare/build.ts +1010 -208
  77. package/src/runtime/targets/container/server.ts +1 -1
  78. package/src/runtime/targets/vps/deploy.ts +26 -9
  79. package/src/runtime/tool-runner.ts +9 -1
  80. package/src/runtime/tools.ts +26 -2
  81. package/src/runtime/transcription.ts +483 -0
  82. package/src/storage/sqlite.ts +7 -2
  83. package/src/templates/blank.ts +37 -9
  84. package/src/templates/dentista.ts +40 -14
  85. package/src/templates/skills/agentkit-build-agent/SKILL.md +34 -5
  86. package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
  87. package/src/templates/skills/agentkit-capsule/SKILL.md +32 -3
  88. package/src/templates/skills/agentkit-capsule/references/docs-router.md +2 -2
  89. package/src/templates/skills/agentkit-channels/SKILL.md +66 -1
  90. package/src/templates/skills/agentkit-channels/references/channel-buffering.md +8 -1
  91. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +28 -3
  92. package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
  93. package/src/templates/skills/agentkit-channels/references/slack.md +56 -0
  94. package/src/templates/skills/agentkit-channels/references/telegram.md +34 -0
  95. package/src/templates/skills/agentkit-channels/references/whatsapp-evolution.md +57 -0
  96. package/src/templates/skills/agentkit-channels/references/whatsapp-uazapi.md +54 -0
  97. package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +42 -8
  98. package/src/templates/skills/agentkit-database/SKILL.md +11 -0
  99. package/src/templates/skills/agentkit-deploy/SKILL.md +9 -1
  100. package/src/templates/skills/agentkit-evals/SKILL.md +77 -13
  101. package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
  102. package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
  103. package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
  104. package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
  105. package/src/templates/skills/agentkit-improve/SKILL.md +96 -0
  106. package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
  107. package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
  108. package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
  109. package/src/templates/skills/agentkit-integrations/SKILL.md +98 -0
  110. package/src/templates/skills/agentkit-knowledge/SKILL.md +4 -1
  111. package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
  112. package/src/templates/skills/agentkit-provider/SKILL.md +29 -4
  113. package/src/templates/skills/agentkit-security/SKILL.md +5 -2
  114. package/src/templates/skills/agentkit-tools/SKILL.md +8 -1
  115. package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +8 -8
  116. package/src/templates/skills/agentkit-tools/examples/jev-service-fit.tool.md +110 -0
  117. package/src/templates/skills/agentkit-troubleshooting/SKILL.md +25 -1
  118. package/src/templates/support.ts +42 -12
  119. package/docs/guides/agentkit-skills-architecture.md +0 -471
  120. package/docs/guides/channels-implementation-map.md +0 -243
  121. package/docs/guides/channels-production-handoff.md +0 -101
  122. package/docs/portable-deploy-release-checklist.md +0 -41
@@ -1,12 +1,13 @@
1
- import { mkdir, readdir, stat, writeFile } from "node:fs/promises";
1
+ import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises";
2
2
  import type { Dirent } from "node:fs";
3
- import { dirname, join, relative, resolve } from "node:path";
3
+ import { tmpdir } from "node:os";
4
+ import { dirname, isAbsolute, join, relative, resolve } from "node:path";
4
5
  import { pathToFileURL } from "node:url";
5
6
 
6
7
  import type { AgentRunResult } from "./chat";
7
8
  import { runAgentMessageFromCwd } from "./chat";
8
9
  import { findAgentCapsuleRoot, loadAgentCapsule } from "./config";
9
- import { AgentKitError } from "./errors";
10
+ import { AgentKitError, isAgentKitError } from "./errors";
10
11
  import { openCapsuleStore, type StoredToolCall } from "../storage/sqlite";
11
12
  import { getConversationTraceFromCwd } from "./traces";
12
13
 
@@ -25,38 +26,76 @@ export type EvalResult = {
25
26
  output: string;
26
27
  };
27
28
 
28
- type EvalCase = {
29
+ export type EvalCase = {
29
30
  name?: string;
30
31
  input?: string;
32
+ now?: string | Date;
31
33
  turns?: EvalTurn[];
32
34
  expect?: EvalExpect;
33
35
  };
34
36
 
35
- type EvalTurn =
37
+ export type EvalTurn =
36
38
  | string
37
39
  | {
38
40
  input: string;
39
41
  expect?: EvalExpect;
40
42
  };
41
43
 
42
- type EvalExpect = {
44
+ export type EvalExpect = EvalResponseExpectation & {
45
+ response?: EvalResponseExpectation;
46
+ tools?: EvalToolsExpectation;
47
+ tool_call?: ToolExpectation | ToolExpectation[];
48
+ toolCall?: ToolExpectation | ToolExpectation[];
49
+ tool_calls?: ToolExpectation | ToolExpectation[];
50
+ toolCalls?: ToolExpectation | ToolExpectation[];
51
+ tool_call_count?: number;
52
+ toolCallCount?: number;
53
+ tool_call_order?: string[];
54
+ toolCallOrder?: string[];
55
+ persisted_tool_call?: ToolExpectation | ToolExpectation[];
56
+ persistedToolCall?: ToolExpectation | ToolExpectation[];
57
+ persisted_tool_calls?: ToolExpectation | ToolExpectation[];
58
+ persistedToolCalls?: ToolExpectation | ToolExpectation[];
59
+ };
60
+
61
+ export type EvalResponseExpectation = {
43
62
  contains?: string | string[];
63
+ contains_all?: string | string[];
64
+ containsAll?: string | string[];
65
+ contains_any?: string | string[];
66
+ containsAny?: string | string[];
67
+ case_insensitive_contains?: string | string[];
68
+ caseInsensitiveContains?: string | string[];
44
69
  not_contains?: string | string[];
45
70
  notContains?: string | string[];
46
71
  regex?: string | string[];
47
72
  matches_regex?: string | string[];
48
73
  matchesRegex?: string | string[];
49
- tool_call?: ToolExpectation | ToolExpectation[];
50
- toolCall?: ToolExpectation | ToolExpectation[];
51
- tool_calls?: ToolExpectation | ToolExpectation[];
52
- toolCalls?: ToolExpectation | ToolExpectation[];
74
+ not_regex?: string | string[];
75
+ notRegex?: string | string[];
76
+ max_length?: number;
77
+ maxLength?: number;
78
+ };
79
+
80
+ export type EvalToolsExpectation =
81
+ | ToolExpectation
82
+ | ToolExpectation[]
83
+ | EvalToolsContainerExpectation;
84
+
85
+ export type EvalToolsContainerExpectation = {
86
+ persisted?: ToolExpectation | ToolExpectation[];
53
87
  persisted_tool_call?: ToolExpectation | ToolExpectation[];
54
88
  persistedToolCall?: ToolExpectation | ToolExpectation[];
55
89
  persisted_tool_calls?: ToolExpectation | ToolExpectation[];
56
90
  persistedToolCalls?: ToolExpectation | ToolExpectation[];
91
+ called?: string | string[];
92
+ called_once?: string | string[];
93
+ calledOnce?: string | string[];
94
+ count?: number;
95
+ order?: string[];
57
96
  };
58
97
 
59
- type ToolExpectation =
98
+ export type ToolExpectation =
60
99
  | string
61
100
  | {
62
101
  name?: string;
@@ -83,6 +122,10 @@ export type EvalFromConversationResult = {
83
122
  turns: number;
84
123
  };
85
124
 
125
+ export function defineEval<const T extends EvalCase>(evalCase: T): T {
126
+ return evalCase;
127
+ }
128
+
86
129
  export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
87
130
  const root = await findAgentCapsuleRoot(cwd);
88
131
  const evalFiles = await findEvalFiles(join(root, "evals"));
@@ -94,15 +137,29 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
94
137
  const results: EvalResult[] = [];
95
138
 
96
139
  for (const file of evalFiles) {
97
- const evalCase = await loadEvalCase(file);
98
- const run = await runEvalCase(root, relative(root, file), evalCase);
140
+ const evalFile = relative(root, file);
141
+ let name = evalFile;
142
+ let failures: string[];
143
+ let output: string;
144
+
145
+ try {
146
+ const evalCase = await loadEvalCase(file);
147
+ const run = await runEvalCase(root, evalFile, evalCase);
148
+
149
+ name = evalCase.name ?? evalFile;
150
+ failures = run.failures;
151
+ output = run.output;
152
+ } catch (error) {
153
+ failures = [formatEvalFailure(error)];
154
+ output = "";
155
+ }
99
156
 
100
157
  results.push({
101
- name: evalCase.name ?? relative(root, file),
102
- file: relative(root, file),
103
- passed: run.failures.length === 0,
104
- failures: run.failures,
105
- output: run.output,
158
+ name,
159
+ file: evalFile,
160
+ passed: failures.length === 0,
161
+ failures,
162
+ output,
106
163
  });
107
164
  }
108
165
 
@@ -128,6 +185,7 @@ export async function writeEvalFromConversation(
128
185
  const root = await findAgentCapsuleRoot(cwd);
129
186
  const trace = await getConversationTraceFromCwd(root, input.conversationId);
130
187
  const turns = replayTurnsFromMessages(trace.messages);
188
+ const safeTitle = redactEvalText(trace.title ?? trace.id);
131
189
 
132
190
  if (turns.length === 0) {
133
191
  throw new AgentKitError(
@@ -136,24 +194,29 @@ export async function writeEvalFromConversation(
136
194
  );
137
195
  }
138
196
 
139
- const file = resolve(root, input.out ?? join("evals", `replay-${slugify(trace.title ?? trace.id)}.eval.ts`));
197
+ const file = resolveEvalOutputPath(root, input.out, safeTitle);
198
+ const source = `import { defineEval } from "@andreprado/agentkit";
140
199
 
141
- if (!input.force && await pathExists(file)) {
142
- throw new AgentKitError(
143
- "validation_error",
144
- `${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
145
- );
146
- }
200
+ export default defineEval({
201
+ name: ${JSON.stringify(input.name ? redactEvalText(input.name) : `replay ${safeTitle}`)},
202
+ turns: ${formatEvalTurns(turns)},
203
+ });
204
+ `;
147
205
 
148
206
  await mkdir(dirname(file), { recursive: true });
149
- await writeFile(
150
- file,
151
- `export default {
152
- name: ${JSON.stringify(input.name ?? `replay ${trace.title ?? trace.id}`)},
153
- turns: ${formatEvalTurns(turns)},
154
- };
155
- `,
156
- );
207
+
208
+ try {
209
+ await writeFile(file, source, { flag: input.force ? "w" : "wx" });
210
+ } catch (error) {
211
+ if (isNodeError(error) && error.code === "EEXIST") {
212
+ throw new AgentKitError(
213
+ "validation_error",
214
+ `${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
215
+ );
216
+ }
217
+
218
+ throw error;
219
+ }
157
220
 
158
221
  return {
159
222
  root,
@@ -180,29 +243,75 @@ async function runEvalCase(
180
243
  const conversationId = `eval_${crypto.randomUUID()}`;
181
244
  const failures: string[] = [];
182
245
  let output = "";
246
+ const now = resolveEvalNow(file, evalCase.now);
247
+ const evalStore = await createEvalStore();
183
248
 
184
- for (const [index, turn] of turns.entries()) {
185
- const run = await runAgentMessageFromCwd(root, {
186
- message: turn.input,
187
- conversationId,
188
- runtime: {
189
- environment: "eval",
190
- invocation: "eval",
191
- },
192
- });
193
- const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
194
- const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
195
-
196
- for (const failure of turnFailures) {
197
- failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
249
+ try {
250
+ for (const [index, turn] of turns.entries()) {
251
+ try {
252
+ const run = await runAgentMessageFromCwd(root, {
253
+ message: turn.input,
254
+ conversationId,
255
+ now,
256
+ localStoragePath: evalStore.databasePath,
257
+ runtime: {
258
+ environment: "eval",
259
+ invocation: "eval",
260
+ },
261
+ });
262
+ const persistedToolCalls = await loadPersistedToolCalls(root, run.runId, evalStore.databasePath);
263
+ const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
264
+
265
+ for (const failure of turnFailures) {
266
+ failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
267
+ }
268
+
269
+ output = run.message.content;
270
+ } catch (error) {
271
+ failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
272
+ break;
273
+ }
198
274
  }
199
-
200
- output = run.message.content;
275
+ } finally {
276
+ await rm(evalStore.directory, { recursive: true, force: true });
201
277
  }
202
278
 
203
279
  return { failures, output };
204
280
  }
205
281
 
282
+ async function createEvalStore(): Promise<{ directory: string; databasePath: string }> {
283
+ const directory = await mkdtemp(join(tmpdir(), "agentkit-eval-"));
284
+ return {
285
+ directory,
286
+ databasePath: join(directory, "agentkit.db"),
287
+ };
288
+ }
289
+
290
+ function resolveEvalNow(file: string, value: EvalCase["now"]): Date | undefined {
291
+ if (value === undefined) {
292
+ return undefined;
293
+ }
294
+
295
+ if (typeof value === "string" && !hasExplicitIsoOffset(value)) {
296
+ throw new AgentKitError(
297
+ "validation_error",
298
+ `${file} now must be an ISO timestamp with an explicit timezone offset, such as "2026-02-04T02:30:00.000Z" or "2026-02-04T02:30:00-05:00".`,
299
+ );
300
+ }
301
+
302
+ const date = value instanceof Date ? new Date(value.getTime()) : new Date(value);
303
+
304
+ if (Number.isNaN(date.getTime())) {
305
+ throw new AgentKitError("validation_error", `${file} now must be a valid ISO timestamp or Date when provided.`);
306
+ }
307
+
308
+ return date;
309
+ }
310
+
311
+ function hasExplicitIsoOffset(value: string): boolean {
312
+ return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.test(value.trim());
313
+ }
314
+
206
315
  function normalizeEvalTurns(evalCase: EvalCase): Array<{ input: string; expect?: EvalExpect }> {
207
316
  if (evalCase.turns !== undefined) {
208
317
  if (!Array.isArray(evalCase.turns)) {
@@ -288,9 +397,9 @@ async function loadEvalCase(file: string): Promise<EvalCase> {
288
397
  return module.default;
289
398
  }
290
399
 
291
- async function loadPersistedToolCalls(root: string, runId: string): Promise<ToolCallSnapshot[]> {
400
+ async function loadPersistedToolCalls(root: string, runId: string, localStoragePath?: string): Promise<ToolCallSnapshot[]> {
292
401
  const capsule = await loadAgentCapsule(root);
293
- const store = await openCapsuleStore(capsule);
402
+ const store = await openCapsuleStore(localStoragePath ? { ...capsule, storagePath: localStoragePath } : capsule);
294
403
 
295
404
  try {
296
405
  return store.listToolCallsForRun(runId).map(toolCallSnapshotFromStored);
@@ -303,12 +412,89 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
303
412
  const failures: string[] = [];
304
413
  const content = run.message.content;
305
414
 
306
- for (const expected of list(expect.contains)) {
415
+ for (const responseExpectation of responseExpectations(expect)) {
416
+ failures.push(...evaluateResponseExpectation(responseExpectation, content));
417
+ }
418
+
419
+ const toolAssertions = collectToolAssertions(expect);
420
+
421
+ for (const toolExpectation of toolAssertions.persisted) {
422
+ const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
423
+
424
+ if (!match) {
425
+ failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
426
+ }
427
+ }
428
+
429
+ for (const expectedName of toolAssertions.called) {
430
+ if (!persistedToolCalls.some((toolCall) => toolCall.name === expectedName)) {
431
+ failures.push(`expected persisted tool call named ${JSON.stringify(expectedName)}`);
432
+ }
433
+ }
434
+
435
+ for (const expectedName of toolAssertions.calledOnce) {
436
+ const count = persistedToolCalls.filter((toolCall) => toolCall.name === expectedName).length;
437
+
438
+ if (count !== 1) {
439
+ failures.push(`expected persisted tool call ${JSON.stringify(expectedName)} exactly once, got ${count}`);
440
+ }
441
+ }
442
+
443
+ for (const expectedCount of toolAssertions.counts) {
444
+ if (!Number.isInteger(expectedCount) || expectedCount < 0) {
445
+ failures.push(`expected tool call count to be a non-negative integer, got ${JSON.stringify(expectedCount)}`);
446
+ continue;
447
+ }
448
+
449
+ if (persistedToolCalls.length !== expectedCount) {
450
+ failures.push(`expected ${expectedCount} persisted tool call(s), got ${persistedToolCalls.length}`);
451
+ }
452
+ }
453
+
454
+ for (const expectedOrder of toolAssertions.orders) {
455
+ if (!namesContainSubsequence(persistedToolNames(persistedToolCalls), expectedOrder)) {
456
+ failures.push(
457
+ `expected persisted tool call order ${JSON.stringify(expectedOrder)}, got ${JSON.stringify(
458
+ persistedToolNames(persistedToolCalls),
459
+ )}`,
460
+ );
461
+ }
462
+ }
463
+
464
+ return failures;
465
+ }
466
+
467
+ function responseExpectations(expect: EvalExpect): EvalResponseExpectation[] {
468
+ return [expect, expect.response].filter((value): value is EvalResponseExpectation => value !== undefined);
469
+ }
470
+
471
+ function evaluateResponseExpectation(expect: EvalResponseExpectation, content: string): string[] {
472
+ const failures: string[] = [];
473
+
474
+ for (const expected of [
475
+ ...list(expect.contains),
476
+ ...list(expect.contains_all),
477
+ ...list(expect.containsAll),
478
+ ]) {
307
479
  if (!content.includes(expected)) {
308
480
  failures.push(`expected output to contain ${JSON.stringify(expected)}`);
309
481
  }
310
482
  }
311
483
 
484
+ const containsAny = [...list(expect.contains_any), ...list(expect.containsAny)];
485
+
486
+ if (containsAny.length > 0 && !containsAny.some((expected) => content.includes(expected))) {
487
+ failures.push(`expected output to contain any of ${JSON.stringify(containsAny)}`);
488
+ }
489
+
490
+ const lowerContent = content.toLowerCase();
491
+
492
+ for (const expected of [...list(expect.case_insensitive_contains), ...list(expect.caseInsensitiveContains)]) {
493
+ if (!lowerContent.includes(expected.toLowerCase())) {
494
+ failures.push(`expected output to contain ${JSON.stringify(expected)} case-insensitively`);
495
+ }
496
+ }
497
+
312
498
  for (const forbidden of [...list(expect.not_contains), ...list(expect.notContains)]) {
313
499
  if (content.includes(forbidden)) {
314
500
  failures.push(`expected output not to contain ${JSON.stringify(forbidden)}`);
@@ -321,28 +507,140 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
321
507
  }
322
508
  }
323
509
 
324
- const toolExpectations = [
325
- ...toolList(expect.tool_call),
326
- ...toolList(expect.toolCall),
327
- ...toolList(expect.tool_calls),
328
- ...toolList(expect.toolCalls),
329
- ...toolList(expect.persisted_tool_call),
330
- ...toolList(expect.persistedToolCall),
331
- ...toolList(expect.persisted_tool_calls),
332
- ...toolList(expect.persistedToolCalls),
333
- ];
510
+ for (const pattern of [...list(expect.not_regex), ...list(expect.notRegex)]) {
511
+ if (new RegExp(pattern).test(content)) {
512
+ failures.push(`expected output not to match /${pattern}/`);
513
+ }
514
+ }
334
515
 
335
- for (const toolExpectation of toolExpectations) {
336
- const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
516
+ for (const expectedMaxLength of [expect.max_length, expect.maxLength]) {
517
+ if (expectedMaxLength === undefined) {
518
+ continue;
519
+ }
337
520
 
338
- if (!match) {
339
- failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
521
+ if (!Number.isInteger(expectedMaxLength) || expectedMaxLength < 0) {
522
+ failures.push(`expected response maxLength to be a non-negative integer, got ${JSON.stringify(expectedMaxLength)}`);
523
+ continue;
524
+ }
525
+
526
+ if (content.length > expectedMaxLength) {
527
+ failures.push(`expected output length to be <= ${expectedMaxLength}, got ${content.length}`);
340
528
  }
341
529
  }
342
530
 
343
531
  return failures;
344
532
  }
345
533
 
534
+ type ToolAssertions = {
535
+ persisted: ToolExpectation[];
536
+ called: string[];
537
+ calledOnce: string[];
538
+ counts: number[];
539
+ orders: string[][];
540
+ };
541
+
542
+ function collectToolAssertions(expect: EvalExpect): ToolAssertions {
543
+ const assertions: ToolAssertions = {
544
+ persisted: [
545
+ ...toolList(expect.tool_call),
546
+ ...toolList(expect.toolCall),
547
+ ...toolList(expect.tool_calls),
548
+ ...toolList(expect.toolCalls),
549
+ ...toolList(expect.persisted_tool_call),
550
+ ...toolList(expect.persistedToolCall),
551
+ ...toolList(expect.persisted_tool_calls),
552
+ ...toolList(expect.persistedToolCalls),
553
+ ],
554
+ called: [],
555
+ calledOnce: [],
556
+ counts: [],
557
+ orders: [],
558
+ };
559
+
560
+ if (expect.tool_call_count !== undefined) {
561
+ assertions.counts.push(expect.tool_call_count);
562
+ }
563
+
564
+ if (expect.toolCallCount !== undefined) {
565
+ assertions.counts.push(expect.toolCallCount);
566
+ }
567
+
568
+ if (expect.tool_call_order !== undefined) {
569
+ assertions.orders.push(expect.tool_call_order);
570
+ }
571
+
572
+ if (expect.toolCallOrder !== undefined) {
573
+ assertions.orders.push(expect.toolCallOrder);
574
+ }
575
+
576
+ appendNestedToolAssertions(assertions, expect.tools);
577
+
578
+ return assertions;
579
+ }
580
+
581
+ function appendNestedToolAssertions(assertions: ToolAssertions, tools: EvalToolsExpectation | undefined): void {
582
+ if (tools === undefined) {
583
+ return;
584
+ }
585
+
586
+ if (!isToolsContainer(tools)) {
587
+ assertions.persisted.push(...toolList(tools));
588
+ return;
589
+ }
590
+
591
+ assertions.persisted.push(
592
+ ...toolList(tools.persisted),
593
+ ...toolList(tools.persisted_tool_call),
594
+ ...toolList(tools.persistedToolCall),
595
+ ...toolList(tools.persisted_tool_calls),
596
+ ...toolList(tools.persistedToolCalls),
597
+ );
598
+ assertions.called.push(...list(tools.called));
599
+ assertions.calledOnce.push(...list(tools.called_once), ...list(tools.calledOnce));
600
+
601
+ if (tools.count !== undefined) {
602
+ assertions.counts.push(tools.count);
603
+ }
604
+
605
+ if (tools.order !== undefined) {
606
+ assertions.orders.push(tools.order);
607
+ }
608
+ }
609
+
610
+ function isToolsContainer(value: EvalToolsExpectation): value is EvalToolsContainerExpectation {
611
+ if (!isRecord(value)) {
612
+ return false;
613
+ }
614
+
615
+ if (
616
+ "name" in value ||
617
+ "input" in value ||
618
+ "output" in value ||
619
+ "rendered" in value ||
620
+ "status" in value ||
621
+ "visibility" in value
622
+ ) {
623
+ return false;
624
+ }
625
+
626
+ if (Object.keys(value).length === 0) {
627
+ return true;
628
+ }
629
+
630
+ return (
631
+ "persisted" in value ||
632
+ "persisted_tool_call" in value ||
633
+ "persistedToolCall" in value ||
634
+ "persisted_tool_calls" in value ||
635
+ "persistedToolCalls" in value ||
636
+ "called" in value ||
637
+ "called_once" in value ||
638
+ "calledOnce" in value ||
639
+ "count" in value ||
640
+ "order" in value
641
+ );
642
+ }
643
+
346
644
  function list(value: string | string[] | undefined): string[] {
347
645
  if (value === undefined) {
348
646
  return [];
@@ -351,6 +649,30 @@ function list(value: string | string[] | undefined): string[] {
351
649
  return Array.isArray(value) ? value : [value];
352
650
  }
353
651
 
652
+ function persistedToolNames(toolCalls: ToolCallSnapshot[]): string[] {
653
+ return toolCalls.map((toolCall) => String(toolCall.name ?? ""));
654
+ }
655
+
656
+ function namesContainSubsequence(actual: string[], expected: string[]): boolean {
657
+ if (expected.length === 0) {
658
+ return true;
659
+ }
660
+
661
+ let actualIndex = 0;
662
+
663
+ for (const expectedName of expected) {
664
+ actualIndex = actual.findIndex((actualName, index) => index >= actualIndex && actualName === expectedName);
665
+
666
+ if (actualIndex === -1) {
667
+ return false;
668
+ }
669
+
670
+ actualIndex += 1;
671
+ }
672
+
673
+ return true;
674
+ }
675
+
354
676
  function toolList(value: ToolExpectation | ToolExpectation[] | undefined): ToolExpectation[] {
355
677
  if (value === undefined) {
356
678
  return [];
@@ -423,12 +745,22 @@ function replayTurnsFromMessages(
423
745
  }
424
746
 
425
747
  const nextAssistant = messages.slice(index + 1).find((candidate) => candidate.role === "assistant");
748
+ const input = redactEvalText(message.content);
749
+ const contains = nextAssistant ? redactEvalText(nextAssistant.content) : undefined;
750
+ const redactionPresent = input.includes("[redacted_") || Boolean(contains?.includes("[redacted_"));
751
+
426
752
  turns.push({
427
- input: message.content,
753
+ input,
428
754
  ...(nextAssistant
429
755
  ? {
430
756
  expect: {
431
- contains: nextAssistant.content,
757
+ response: redactionPresent
758
+ ? {
759
+ notRegex: ["API_KEY|Bearer\\s+|(?:sk|agk|ak|dpat)[_-]|[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,}"],
760
+ }
761
+ : {
762
+ contains: contains ?? "",
763
+ },
432
764
  },
433
765
  }
434
766
  : {}),
@@ -438,6 +770,18 @@ function replayTurnsFromMessages(
438
770
  return turns;
439
771
  }
440
772
 
773
+ function redactEvalText(value: string): string {
774
+ return value
775
+ .replace(/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi, "[redacted_email]")
776
+ .replace(
777
+ /\b[A-Z0-9_]*(?:API[_-]?KEY|TOKEN|SECRET|PASSWORD|PRIVATE[_-]?KEY|CLIENT[_-]?SECRET)\s*=\s*[^\s,;]+/gi,
778
+ (match) => `${match.slice(0, match.indexOf("=")).trim()}=[redacted_secret]`,
779
+ )
780
+ .replace(/\b(?:sk|agk|ak|dpat)[_-][A-Za-z0-9_-]{8,}\b/g, "[redacted_token]")
781
+ .replace(/\bBearer\s+[A-Za-z0-9._~+/-]+=*/gi, "Bearer [redacted_token]")
782
+ .replace(/\+?\d[\d\s().-]{7,}\d/g, "[redacted_phone]");
783
+ }
784
+
441
785
  function formatEvalTurns(turns: Array<{ input: string; expect?: EvalExpect }>): string {
442
786
  return JSON.stringify(turns, null, 4)
443
787
  .split("\n")
@@ -455,13 +799,40 @@ function slugify(value: string): string {
455
799
  return slug || "conversation";
456
800
  }
457
801
 
458
- async function pathExists(path: string): Promise<boolean> {
459
- try {
460
- await stat(path);
461
- return true;
462
- } catch {
463
- return false;
802
+ function resolveEvalOutputPath(root: string, out: string | undefined, title: string): string {
803
+ const outputPath = out ?? join("evals", `replay-${slugify(title)}.eval.ts`);
804
+
805
+ if (isAbsolute(outputPath)) {
806
+ throw new AgentKitError("validation_error", "eval --out must be a relative path inside the Agent Capsule.");
807
+ }
808
+
809
+ const file = resolve(root, outputPath);
810
+ const relativePath = relative(root, file);
811
+
812
+ if (relativePath === "" || relativePath.startsWith("..") || isAbsolute(relativePath)) {
813
+ throw new AgentKitError("validation_error", "eval --out must stay inside the Agent Capsule.");
464
814
  }
815
+
816
+ if (!/\.(eval|spec)\.[cm]?[tj]s$/.test(file)) {
817
+ throw new AgentKitError(
818
+ "validation_error",
819
+ "eval --out must end with .eval.ts, .eval.js, .spec.ts, or another AgentKit eval/spec extension.",
820
+ );
821
+ }
822
+
823
+ return file;
824
+ }
825
+
826
+ function formatEvalFailure(error: unknown): string {
827
+ if (isAgentKitError(error)) {
828
+ return `${error.code}: ${error.message}`;
829
+ }
830
+
831
+ if (error instanceof Error) {
832
+ return `runtime_error: ${error.message}`;
833
+ }
834
+
835
+ return `runtime_error: ${String(error)}`;
465
836
  }
466
837
 
467
838
  function jsonContains(actual: unknown, expected: unknown): boolean {