@andreprado/agentkit 0.1.0-alpha.2 → 0.1.0-alpha.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/README.md +67 -6
  2. package/docs/guides/add-channel.md +118 -7
  3. package/docs/guides/add-knowledge.md +144 -0
  4. package/docs/guides/add-managed-composio.md +163 -0
  5. package/docs/guides/add-tool.md +1 -1
  6. package/docs/guides/channel-security.md +97 -32
  7. package/docs/guides/connect-discord.md +178 -0
  8. package/docs/guides/connect-slack.md +126 -0
  9. package/docs/guides/connect-telegram.md +78 -1
  10. package/docs/guides/connect-whatsapp-zapster.md +112 -8
  11. package/docs/guides/create-agent.md +45 -4
  12. package/docs/guides/debug-channel.md +147 -0
  13. package/docs/guides/improve-from-production.md +151 -0
  14. package/docs/guides/prepare-deploy.md +47 -17
  15. package/docs/guides/replay-production-traces.md +72 -0
  16. package/docs/guides/run-evals.md +147 -20
  17. package/docs/guides/security-rules.md +7 -6
  18. package/docs/guides/send-feedback.md +135 -0
  19. package/docs/guides/use-provider.md +27 -3
  20. package/docs/llms-full.txt +303 -55
  21. package/docs/llms.txt +57 -7
  22. package/package.json +2 -5
  23. package/src/cli/args.ts +57 -0
  24. package/src/cli/cloud-client.ts +377 -0
  25. package/src/cli/commands/channels.ts +1315 -0
  26. package/src/cli/commands/feedback.ts +438 -0
  27. package/src/cli/commands/knowledge.ts +136 -0
  28. package/src/cli/commands/transcribe.ts +171 -0
  29. package/src/cli/constants.ts +4 -0
  30. package/src/cli/deploy-chat-ui.ts +535 -0
  31. package/src/cli/deploy-readiness.ts +481 -0
  32. package/src/cli/flags.ts +162 -0
  33. package/src/cli/help.ts +236 -0
  34. package/src/cli/index.ts +1167 -1005
  35. package/src/cli/process.ts +31 -0
  36. package/src/cloud/artifact.ts +139 -0
  37. package/src/cloud/client.ts +80 -0
  38. package/src/cloud/contracts.ts +63 -0
  39. package/src/cloud/index.ts +3 -0
  40. package/src/create-project.ts +21 -6
  41. package/src/index.ts +479 -7
  42. package/src/providers/pi.ts +70 -16
  43. package/src/providers/test.ts +88 -1
  44. package/src/providers/types.ts +7 -0
  45. package/src/runtime/channel-buffer.ts +30 -0
  46. package/src/runtime/channel-test-harness.ts +8 -1
  47. package/src/runtime/channels/discord.ts +896 -0
  48. package/src/runtime/channels/slack.ts +646 -0
  49. package/src/runtime/channels/telegram.ts +466 -23
  50. package/src/runtime/channels/whatsapp-meta.ts +9 -0
  51. package/src/runtime/channels/whatsapp-zapster.ts +677 -40
  52. package/src/runtime/channels.ts +86 -3
  53. package/src/runtime/chat.ts +130 -38
  54. package/src/runtime/config.ts +483 -18
  55. package/src/runtime/core/manifest.ts +103 -5
  56. package/src/runtime/core/targets.ts +5 -5
  57. package/src/runtime/database.ts +93 -2
  58. package/src/runtime/db-commands.ts +9 -0
  59. package/src/runtime/deploy-readiness.ts +46 -4
  60. package/src/runtime/deploy.ts +1 -1
  61. package/src/runtime/dev-server.ts +759 -41
  62. package/src/runtime/env.ts +8 -3
  63. package/src/runtime/evals.ts +589 -43
  64. package/src/runtime/improve.ts +868 -0
  65. package/src/runtime/inspect.ts +194 -4
  66. package/src/runtime/integrations/composio.ts +423 -0
  67. package/src/runtime/knowledge/chunk.ts +333 -0
  68. package/src/runtime/knowledge/config.ts +135 -0
  69. package/src/runtime/knowledge/embeddings.ts +133 -0
  70. package/src/runtime/knowledge/ingest.ts +521 -0
  71. package/src/runtime/knowledge/prompt-policy.ts +30 -0
  72. package/src/runtime/knowledge/retrieve.ts +303 -0
  73. package/src/runtime/knowledge/schema.ts +100 -0
  74. package/src/runtime/knowledge/tool.ts +64 -0
  75. package/src/runtime/knowledge/vector.ts +258 -0
  76. package/src/runtime/prompt-context.ts +141 -0
  77. package/src/runtime/runtime-contract.ts +86 -8
  78. package/src/runtime/skills.ts +95 -0
  79. package/src/runtime/spec.ts +152 -0
  80. package/src/runtime/sync.ts +144 -0
  81. package/src/runtime/targets/cloudflare/build.ts +1430 -185
  82. package/src/runtime/targets/container/server.ts +1 -1
  83. package/src/runtime/targets/vps/deploy.ts +26 -9
  84. package/src/runtime/tool-runner.ts +9 -1
  85. package/src/runtime/tools.ts +128 -2
  86. package/src/runtime/traces.ts +41 -0
  87. package/src/runtime/transcription.ts +483 -0
  88. package/src/storage/sqlite.ts +149 -3
  89. package/src/templates/blank.ts +76 -17
  90. package/src/templates/dentista.ts +1011 -0
  91. package/src/templates/index.ts +2 -0
  92. package/src/templates/skills/agentkit-build-agent/SKILL.md +52 -0
  93. package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +21 -0
  94. package/src/templates/skills/agentkit-build-agent/templates/sales-qualifier.instructions.md +17 -0
  95. package/src/templates/skills/agentkit-build-agent/templates/support-agent.instructions.md +16 -0
  96. package/src/templates/skills/agentkit-capsule/SKILL.md +70 -0
  97. package/src/templates/skills/agentkit-capsule/references/docs-router.md +15 -0
  98. package/src/templates/skills/agentkit-channels/SKILL.md +104 -0
  99. package/src/templates/skills/agentkit-channels/references/channel-buffering.md +65 -0
  100. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +66 -0
  101. package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
  102. package/src/templates/skills/agentkit-channels/references/slack.md +56 -0
  103. package/src/templates/skills/agentkit-channels/references/telegram.md +72 -0
  104. package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +77 -0
  105. package/src/templates/skills/agentkit-database/SKILL.md +45 -0
  106. package/src/templates/skills/agentkit-database/templates/appointments.schema.sql +15 -0
  107. package/src/templates/skills/agentkit-database/templates/leads.schema.sql +17 -0
  108. package/src/templates/skills/agentkit-deploy/SKILL.md +50 -0
  109. package/src/templates/skills/agentkit-evals/SKILL.md +109 -0
  110. package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +29 -0
  111. package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +18 -0
  112. package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +18 -0
  113. package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +27 -0
  114. package/src/templates/skills/agentkit-improve/SKILL.md +86 -0
  115. package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
  116. package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
  117. package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
  118. package/src/templates/skills/agentkit-integrations/SKILL.md +76 -0
  119. package/src/templates/skills/agentkit-knowledge/SKILL.md +43 -0
  120. package/src/templates/skills/agentkit-knowledge/templates/faq.md +14 -0
  121. package/src/templates/skills/agentkit-knowledge/templates/policies.md +14 -0
  122. package/src/templates/skills/agentkit-knowledge/templates/prices.csv +3 -0
  123. package/src/templates/skills/agentkit-prompts/SKILL.md +47 -0
  124. package/src/templates/skills/agentkit-prompts/templates/knowledge-grounded-faq.instructions.md +11 -0
  125. package/src/templates/skills/agentkit-provider/SKILL.md +60 -0
  126. package/src/templates/skills/agentkit-security/SKILL.md +56 -0
  127. package/src/templates/skills/agentkit-tools/SKILL.md +37 -0
  128. package/src/templates/skills/agentkit-tools/examples/database-write.tool.md +35 -0
  129. package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +37 -0
  130. package/src/templates/skills/agentkit-tools/examples/lookup-order.tool.md +46 -0
  131. package/src/templates/skills/agentkit-troubleshooting/SKILL.md +76 -0
  132. package/src/templates/support.ts +77 -18
  133. package/docs/guides/channels-production-handoff.md +0 -99
  134. package/docs/portable-deploy-release-checklist.md +0 -41
  135. package/src/runtime/targets/cloudflare/deploy.ts +0 -5475
@@ -1,13 +1,14 @@
1
- import { readdir } from "node:fs/promises";
1
+ import { mkdir, readdir, writeFile } from "node:fs/promises";
2
2
  import type { Dirent } from "node:fs";
3
- import { join, relative } from "node:path";
3
+ import { dirname, isAbsolute, join, relative, resolve } from "node:path";
4
4
  import { pathToFileURL } from "node:url";
5
5
 
6
6
  import type { AgentRunResult } from "./chat";
7
7
  import { runAgentMessageFromCwd } from "./chat";
8
8
  import { findAgentCapsuleRoot, loadAgentCapsule } from "./config";
9
- import { AgentKitError } from "./errors";
9
+ import { AgentKitError, isAgentKitError } from "./errors";
10
10
  import { openCapsuleStore, type StoredToolCall } from "../storage/sqlite";
11
+ import { getConversationTraceFromCwd } from "./traces";
11
12
 
12
13
  export type EvalRunSummary = {
13
14
  root: string;
@@ -24,30 +25,76 @@ export type EvalResult = {
24
25
  output: string;
25
26
  };
26
27
 
27
- type EvalCase = {
28
+ export type EvalCase = {
28
29
  name?: string;
29
30
  input?: string;
31
+ now?: string | Date;
32
+ turns?: EvalTurn[];
30
33
  expect?: EvalExpect;
31
34
  };
32
35
 
33
- type EvalExpect = {
36
+ export type EvalTurn =
37
+ | string
38
+ | {
39
+ input: string;
40
+ expect?: EvalExpect;
41
+ };
42
+
43
+ export type EvalExpect = EvalResponseExpectation & {
44
+ response?: EvalResponseExpectation;
45
+ tools?: EvalToolsExpectation;
46
+ tool_call?: ToolExpectation | ToolExpectation[];
47
+ toolCall?: ToolExpectation | ToolExpectation[];
48
+ tool_calls?: ToolExpectation | ToolExpectation[];
49
+ toolCalls?: ToolExpectation | ToolExpectation[];
50
+ tool_call_count?: number;
51
+ toolCallCount?: number;
52
+ tool_call_order?: string[];
53
+ toolCallOrder?: string[];
54
+ persisted_tool_call?: ToolExpectation | ToolExpectation[];
55
+ persistedToolCall?: ToolExpectation | ToolExpectation[];
56
+ persisted_tool_calls?: ToolExpectation | ToolExpectation[];
57
+ persistedToolCalls?: ToolExpectation | ToolExpectation[];
58
+ };
59
+
60
+ export type EvalResponseExpectation = {
34
61
  contains?: string | string[];
62
+ contains_all?: string | string[];
63
+ containsAll?: string | string[];
64
+ contains_any?: string | string[];
65
+ containsAny?: string | string[];
66
+ case_insensitive_contains?: string | string[];
67
+ caseInsensitiveContains?: string | string[];
35
68
  not_contains?: string | string[];
36
69
  notContains?: string | string[];
37
70
  regex?: string | string[];
38
71
  matches_regex?: string | string[];
39
72
  matchesRegex?: string | string[];
40
- tool_call?: ToolExpectation | ToolExpectation[];
41
- toolCall?: ToolExpectation | ToolExpectation[];
42
- tool_calls?: ToolExpectation | ToolExpectation[];
43
- toolCalls?: ToolExpectation | ToolExpectation[];
73
+ not_regex?: string | string[];
74
+ notRegex?: string | string[];
75
+ max_length?: number;
76
+ maxLength?: number;
77
+ };
78
+
79
+ export type EvalToolsExpectation =
80
+ | ToolExpectation
81
+ | ToolExpectation[]
82
+ | EvalToolsContainerExpectation;
83
+
84
+ export type EvalToolsContainerExpectation = {
85
+ persisted?: ToolExpectation | ToolExpectation[];
44
86
  persisted_tool_call?: ToolExpectation | ToolExpectation[];
45
87
  persistedToolCall?: ToolExpectation | ToolExpectation[];
46
88
  persisted_tool_calls?: ToolExpectation | ToolExpectation[];
47
89
  persistedToolCalls?: ToolExpectation | ToolExpectation[];
90
+ called?: string | string[];
91
+ called_once?: string | string[];
92
+ calledOnce?: string | string[];
93
+ count?: number;
94
+ order?: string[];
48
95
  };
49
96
 
50
- type ToolExpectation =
97
+ export type ToolExpectation =
51
98
  | string
52
99
  | {
53
100
  name?: string;
@@ -55,16 +102,29 @@ type ToolExpectation =
55
102
  output?: unknown;
56
103
  status?: "running" | "completed" | "failed";
57
104
  visibility?: "user" | "internal";
105
+ rendered?: unknown;
58
106
  };
59
107
 
60
108
  type ToolCallSnapshot = {
61
109
  name?: unknown;
62
110
  input?: unknown;
63
111
  output?: unknown;
112
+ rendered?: unknown;
64
113
  status?: unknown;
65
114
  visibility?: unknown;
66
115
  };
67
116
 
117
+ export type EvalFromConversationResult = {
118
+ root: string;
119
+ file: string;
120
+ conversationId: string;
121
+ turns: number;
122
+ };
123
+
124
+ export function defineEval<const T extends EvalCase>(evalCase: T): T {
125
+ return evalCase;
126
+ }
127
+
68
128
  export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
69
129
  const root = await findAgentCapsuleRoot(cwd);
70
130
  const evalFiles = await findEvalFiles(join(root, "evals"));
@@ -76,29 +136,29 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
76
136
  const results: EvalResult[] = [];
77
137
 
78
138
  for (const file of evalFiles) {
79
- const evalCase = await loadEvalCase(file);
80
- const input = evalCase.input;
81
-
82
- if (typeof input !== "string" || input.trim().length === 0) {
83
- throw new AgentKitError("validation_error", `${relative(root, file)} must export an eval with a non-empty input string.`);
139
+ const evalFile = relative(root, file);
140
+ let name = evalFile;
141
+ let failures: string[];
142
+ let output: string;
143
+
144
+ try {
145
+ const evalCase = await loadEvalCase(file);
146
+ const run = await runEvalCase(root, evalFile, evalCase);
147
+
148
+ name = evalCase.name ?? evalFile;
149
+ failures = run.failures;
150
+ output = run.output;
151
+ } catch (error) {
152
+ failures = [formatEvalFailure(error)];
153
+ output = "";
84
154
  }
85
155
 
86
- const run = await runAgentMessageFromCwd(root, {
87
- message: input,
88
- runtime: {
89
- environment: "eval",
90
- invocation: "eval",
91
- },
92
- });
93
- const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
94
- const failures = evaluateExpectations(evalCase.expect ?? {}, run, persistedToolCalls);
95
-
96
156
  results.push({
97
- name: evalCase.name ?? relative(root, file),
98
- file: relative(root, file),
157
+ name,
158
+ file: evalFile,
99
159
  passed: failures.length === 0,
100
160
  failures,
101
- output: run.message.content,
161
+ output,
102
162
  });
103
163
  }
104
164
 
@@ -112,6 +172,171 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
112
172
  };
113
173
  }
114
174
 
175
+ export async function writeEvalFromConversation(
176
+ cwd: string,
177
+ input: {
178
+ conversationId: string;
179
+ out?: string;
180
+ name?: string;
181
+ force?: boolean;
182
+ },
183
+ ): Promise<EvalFromConversationResult> {
184
+ const root = await findAgentCapsuleRoot(cwd);
185
+ const trace = await getConversationTraceFromCwd(root, input.conversationId);
186
+ const turns = replayTurnsFromMessages(trace.messages);
187
+ const safeTitle = redactEvalText(trace.title ?? trace.id);
188
+
189
+ if (turns.length === 0) {
190
+ throw new AgentKitError(
191
+ "validation_error",
192
+ `Conversation "${input.conversationId}" does not contain user/assistant turns that can become an eval.`,
193
+ );
194
+ }
195
+
196
+ const file = resolveEvalOutputPath(root, input.out, safeTitle);
197
+ const source = `import { defineEval } from "@andreprado/agentkit";
198
+
199
+ export default defineEval({
200
+ name: ${JSON.stringify(input.name ? redactEvalText(input.name) : `replay ${safeTitle}`)},
201
+ turns: ${formatEvalTurns(turns)},
202
+ });
203
+ `;
204
+
205
+ await mkdir(dirname(file), { recursive: true });
206
+
207
+ try {
208
+ await writeFile(file, source, { flag: input.force ? "w" : "wx" });
209
+ } catch (error) {
210
+ if (isNodeError(error) && error.code === "EEXIST") {
211
+ throw new AgentKitError(
212
+ "validation_error",
213
+ `${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
214
+ );
215
+ }
216
+
217
+ throw error;
218
+ }
219
+
220
+ return {
221
+ root,
222
+ file,
223
+ conversationId: trace.id,
224
+ turns: turns.length,
225
+ };
226
+ }
227
+
228
+ async function runEvalCase(
229
+ root: string,
230
+ file: string,
231
+ evalCase: EvalCase,
232
+ ): Promise<{ failures: string[]; output: string }> {
233
+ const turns = normalizeEvalTurns(evalCase);
234
+
235
+ if (turns.length === 0) {
236
+ throw new AgentKitError(
237
+ "validation_error",
238
+ `${file} must export either a non-empty input string or a non-empty turns array.`,
239
+ );
240
+ }
241
+
242
+ const conversationId = `eval_${crypto.randomUUID()}`;
243
+ const failures: string[] = [];
244
+ let output = "";
245
+ const now = resolveEvalNow(file, evalCase.now);
246
+
247
+ for (const [index, turn] of turns.entries()) {
248
+ try {
249
+ const run = await runAgentMessageFromCwd(root, {
250
+ message: turn.input,
251
+ conversationId,
252
+ now,
253
+ runtime: {
254
+ environment: "eval",
255
+ invocation: "eval",
256
+ },
257
+ });
258
+ const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
259
+ const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
260
+
261
+ for (const failure of turnFailures) {
262
+ failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
263
+ }
264
+
265
+ output = run.message.content;
266
+ } catch (error) {
267
+ failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
268
+ break;
269
+ }
270
+ }
271
+
272
+ return { failures, output };
273
+ }
274
+
275
+ function resolveEvalNow(file: string, value: EvalCase["now"]): Date | undefined {
276
+ if (value === undefined) {
277
+ return undefined;
278
+ }
279
+
280
+ if (typeof value === "string" && !hasExplicitIsoOffset(value)) {
281
+ throw new AgentKitError(
282
+ "validation_error",
283
+ `${file} now must be an ISO timestamp with an explicit timezone offset, such as "2026-02-04T02:30:00.000Z" or "2026-02-04T02:30:00-05:00".`,
284
+ );
285
+ }
286
+
287
+ const date = value instanceof Date ? new Date(value.getTime()) : new Date(value);
288
+
289
+ if (Number.isNaN(date.getTime())) {
290
+ throw new AgentKitError("validation_error", `${file} now must be a valid ISO timestamp or Date when provided.`);
291
+ }
292
+
293
+ return date;
294
+ }
295
+
296
+ function hasExplicitIsoOffset(value: string): boolean {
297
+ return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.test(value.trim());
298
+ }
299
+
300
+ function normalizeEvalTurns(evalCase: EvalCase): Array<{ input: string; expect?: EvalExpect }> {
301
+ if (evalCase.turns !== undefined) {
302
+ if (!Array.isArray(evalCase.turns)) {
303
+ throw new AgentKitError("validation_error", "eval turns must be an array when provided.");
304
+ }
305
+
306
+ return evalCase.turns.map((turn, index) => normalizeEvalTurn(turn, index));
307
+ }
308
+
309
+ if (typeof evalCase.input === "string" && evalCase.input.trim().length > 0) {
310
+ return [
311
+ {
312
+ input: evalCase.input,
313
+ expect: evalCase.expect,
314
+ },
315
+ ];
316
+ }
317
+
318
+ return [];
319
+ }
320
+
321
+ function normalizeEvalTurn(turn: EvalTurn, index: number): { input: string; expect?: EvalExpect } {
322
+ if (typeof turn === "string") {
323
+ if (turn.trim().length === 0) {
324
+ throw new AgentKitError("validation_error", `eval turns[${index}] must be non-empty.`);
325
+ }
326
+
327
+ return { input: turn };
328
+ }
329
+
330
+ if (!isRecord(turn) || typeof turn.input !== "string" || turn.input.trim().length === 0) {
331
+ throw new AgentKitError("validation_error", `eval turns[${index}].input must be a non-empty string.`);
332
+ }
333
+
334
+ return {
335
+ input: turn.input,
336
+ ...(turn.expect ? { expect: turn.expect } : {}),
337
+ };
338
+ }
339
+
115
340
  async function findEvalFiles(directory: string): Promise<string[]> {
116
341
  let entries: Dirent[];
117
342
 
@@ -172,12 +397,89 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
172
397
  const failures: string[] = [];
173
398
  const content = run.message.content;
174
399
 
175
- for (const expected of list(expect.contains)) {
400
+ for (const responseExpectation of responseExpectations(expect)) {
401
+ failures.push(...evaluateResponseExpectation(responseExpectation, content));
402
+ }
403
+
404
+ const toolAssertions = collectToolAssertions(expect);
405
+
406
+ for (const toolExpectation of toolAssertions.persisted) {
407
+ const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
408
+
409
+ if (!match) {
410
+ failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
411
+ }
412
+ }
413
+
414
+ for (const expectedName of toolAssertions.called) {
415
+ if (!persistedToolCalls.some((toolCall) => toolCall.name === expectedName)) {
416
+ failures.push(`expected persisted tool call named ${JSON.stringify(expectedName)}`);
417
+ }
418
+ }
419
+
420
+ for (const expectedName of toolAssertions.calledOnce) {
421
+ const count = persistedToolCalls.filter((toolCall) => toolCall.name === expectedName).length;
422
+
423
+ if (count !== 1) {
424
+ failures.push(`expected persisted tool call ${JSON.stringify(expectedName)} exactly once, got ${count}`);
425
+ }
426
+ }
427
+
428
+ for (const expectedCount of toolAssertions.counts) {
429
+ if (!Number.isInteger(expectedCount) || expectedCount < 0) {
430
+ failures.push(`expected tool call count to be a non-negative integer, got ${JSON.stringify(expectedCount)}`);
431
+ continue;
432
+ }
433
+
434
+ if (persistedToolCalls.length !== expectedCount) {
435
+ failures.push(`expected ${expectedCount} persisted tool call(s), got ${persistedToolCalls.length}`);
436
+ }
437
+ }
438
+
439
+ for (const expectedOrder of toolAssertions.orders) {
440
+ if (!namesContainSubsequence(persistedToolNames(persistedToolCalls), expectedOrder)) {
441
+ failures.push(
442
+ `expected persisted tool call order ${JSON.stringify(expectedOrder)}, got ${JSON.stringify(
443
+ persistedToolNames(persistedToolCalls),
444
+ )}`,
445
+ );
446
+ }
447
+ }
448
+
449
+ return failures;
450
+ }
451
+
452
+ function responseExpectations(expect: EvalExpect): EvalResponseExpectation[] {
453
+ return [expect, expect.response].filter((value): value is EvalResponseExpectation => value !== undefined);
454
+ }
455
+
456
+ function evaluateResponseExpectation(expect: EvalResponseExpectation, content: string): string[] {
457
+ const failures: string[] = [];
458
+
459
+ for (const expected of [
460
+ ...list(expect.contains),
461
+ ...list(expect.contains_all),
462
+ ...list(expect.containsAll),
463
+ ]) {
176
464
  if (!content.includes(expected)) {
177
465
  failures.push(`expected output to contain ${JSON.stringify(expected)}`);
178
466
  }
179
467
  }
180
468
 
469
+ const containsAny = [...list(expect.contains_any), ...list(expect.containsAny)];
470
+
471
+ if (containsAny.length > 0 && !containsAny.some((expected) => content.includes(expected))) {
472
+ failures.push(`expected output to contain any of ${JSON.stringify(containsAny)}`);
473
+ }
474
+
475
+ const lowerContent = content.toLowerCase();
476
+
477
+ for (const expected of [...list(expect.case_insensitive_contains), ...list(expect.caseInsensitiveContains)]) {
478
+ if (!lowerContent.includes(expected.toLowerCase())) {
479
+ failures.push(`expected output to contain ${JSON.stringify(expected)} case-insensitively`);
480
+ }
481
+ }
482
+
181
483
  for (const forbidden of [...list(expect.not_contains), ...list(expect.notContains)]) {
182
484
  if (content.includes(forbidden)) {
183
485
  failures.push(`expected output not to contain ${JSON.stringify(forbidden)}`);
@@ -190,28 +492,140 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
190
492
  }
191
493
  }
192
494
 
193
- const toolExpectations = [
194
- ...toolList(expect.tool_call),
195
- ...toolList(expect.toolCall),
196
- ...toolList(expect.tool_calls),
197
- ...toolList(expect.toolCalls),
198
- ...toolList(expect.persisted_tool_call),
199
- ...toolList(expect.persistedToolCall),
200
- ...toolList(expect.persisted_tool_calls),
201
- ...toolList(expect.persistedToolCalls),
202
- ];
495
+ for (const pattern of [...list(expect.not_regex), ...list(expect.notRegex)]) {
496
+ if (new RegExp(pattern).test(content)) {
497
+ failures.push(`expected output not to match /${pattern}/`);
498
+ }
499
+ }
203
500
 
204
- for (const toolExpectation of toolExpectations) {
205
- const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
501
+ for (const expectedMaxLength of [expect.max_length, expect.maxLength]) {
502
+ if (expectedMaxLength === undefined) {
503
+ continue;
504
+ }
206
505
 
207
- if (!match) {
208
- failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
506
+ if (!Number.isInteger(expectedMaxLength) || expectedMaxLength < 0) {
507
+ failures.push(`expected response maxLength to be a non-negative integer, got ${JSON.stringify(expectedMaxLength)}`);
508
+ continue;
509
+ }
510
+
511
+ if (content.length > expectedMaxLength) {
512
+ failures.push(`expected output length to be <= ${expectedMaxLength}, got ${content.length}`);
209
513
  }
210
514
  }
211
515
 
212
516
  return failures;
213
517
  }
214
518
 
519
+ type ToolAssertions = {
520
+ persisted: ToolExpectation[];
521
+ called: string[];
522
+ calledOnce: string[];
523
+ counts: number[];
524
+ orders: string[][];
525
+ };
526
+
527
+ function collectToolAssertions(expect: EvalExpect): ToolAssertions {
528
+ const assertions: ToolAssertions = {
529
+ persisted: [
530
+ ...toolList(expect.tool_call),
531
+ ...toolList(expect.toolCall),
532
+ ...toolList(expect.tool_calls),
533
+ ...toolList(expect.toolCalls),
534
+ ...toolList(expect.persisted_tool_call),
535
+ ...toolList(expect.persistedToolCall),
536
+ ...toolList(expect.persisted_tool_calls),
537
+ ...toolList(expect.persistedToolCalls),
538
+ ],
539
+ called: [],
540
+ calledOnce: [],
541
+ counts: [],
542
+ orders: [],
543
+ };
544
+
545
+ if (expect.tool_call_count !== undefined) {
546
+ assertions.counts.push(expect.tool_call_count);
547
+ }
548
+
549
+ if (expect.toolCallCount !== undefined) {
550
+ assertions.counts.push(expect.toolCallCount);
551
+ }
552
+
553
+ if (expect.tool_call_order !== undefined) {
554
+ assertions.orders.push(expect.tool_call_order);
555
+ }
556
+
557
+ if (expect.toolCallOrder !== undefined) {
558
+ assertions.orders.push(expect.toolCallOrder);
559
+ }
560
+
561
+ appendNestedToolAssertions(assertions, expect.tools);
562
+
563
+ return assertions;
564
+ }
565
+
566
+ function appendNestedToolAssertions(assertions: ToolAssertions, tools: EvalToolsExpectation | undefined): void {
567
+ if (tools === undefined) {
568
+ return;
569
+ }
570
+
571
+ if (!isToolsContainer(tools)) {
572
+ assertions.persisted.push(...toolList(tools));
573
+ return;
574
+ }
575
+
576
+ assertions.persisted.push(
577
+ ...toolList(tools.persisted),
578
+ ...toolList(tools.persisted_tool_call),
579
+ ...toolList(tools.persistedToolCall),
580
+ ...toolList(tools.persisted_tool_calls),
581
+ ...toolList(tools.persistedToolCalls),
582
+ );
583
+ assertions.called.push(...list(tools.called));
584
+ assertions.calledOnce.push(...list(tools.called_once), ...list(tools.calledOnce));
585
+
586
+ if (tools.count !== undefined) {
587
+ assertions.counts.push(tools.count);
588
+ }
589
+
590
+ if (tools.order !== undefined) {
591
+ assertions.orders.push(tools.order);
592
+ }
593
+ }
594
+
595
+ function isToolsContainer(value: EvalToolsExpectation): value is EvalToolsContainerExpectation {
596
+ if (!isRecord(value)) {
597
+ return false;
598
+ }
599
+
600
+ if (
601
+ "name" in value ||
602
+ "input" in value ||
603
+ "output" in value ||
604
+ "rendered" in value ||
605
+ "status" in value ||
606
+ "visibility" in value
607
+ ) {
608
+ return false;
609
+ }
610
+
611
+ if (Object.keys(value).length === 0) {
612
+ return true;
613
+ }
614
+
615
+ return (
616
+ "persisted" in value ||
617
+ "persisted_tool_call" in value ||
618
+ "persistedToolCall" in value ||
619
+ "persisted_tool_calls" in value ||
620
+ "persistedToolCalls" in value ||
621
+ "called" in value ||
622
+ "called_once" in value ||
623
+ "calledOnce" in value ||
624
+ "count" in value ||
625
+ "order" in value
626
+ );
627
+ }
628
+
215
629
  function list(value: string | string[] | undefined): string[] {
216
630
  if (value === undefined) {
217
631
  return [];
@@ -220,6 +634,30 @@ function list(value: string | string[] | undefined): string[] {
220
634
  return Array.isArray(value) ? value : [value];
221
635
  }
222
636
 
637
+ function persistedToolNames(toolCalls: ToolCallSnapshot[]): string[] {
638
+ return toolCalls.map((toolCall) => String(toolCall.name ?? ""));
639
+ }
640
+
641
+ function namesContainSubsequence(actual: string[], expected: string[]): boolean {
642
+ if (expected.length === 0) {
643
+ return true;
644
+ }
645
+
646
+ let actualIndex = 0;
647
+
648
+ for (const expectedName of expected) {
649
+ actualIndex = actual.findIndex((actualName, index) => index >= actualIndex && actualName === expectedName);
650
+
651
+ if (actualIndex === -1) {
652
+ return false;
653
+ }
654
+
655
+ actualIndex += 1;
656
+ }
657
+
658
+ return true;
659
+ }
660
+
223
661
  function toolList(value: ToolExpectation | ToolExpectation[] | undefined): ToolExpectation[] {
224
662
  if (value === undefined) {
225
663
  return [];
@@ -253,6 +691,10 @@ function matchesToolExpectation(toolCall: ToolCallSnapshot, expectation: ToolExp
253
691
  return false;
254
692
  }
255
693
 
694
+ if ("rendered" in expectation && !jsonContains(toolCall.rendered, expectation.rendered)) {
695
+ return false;
696
+ }
697
+
256
698
  return true;
257
699
  }
258
700
 
@@ -269,11 +711,115 @@ function toolCallSnapshotFromStored(toolCall: StoredToolCall): ToolCallSnapshot
269
711
  name: toolCall.toolName,
270
712
  input: toolCall.input,
271
713
  output: toolCall.output,
714
+ rendered: toolCall.rendered,
272
715
  status: toolCall.status,
273
716
  visibility: toolCall.visibility,
274
717
  };
275
718
  }
276
719
 
720
+ function replayTurnsFromMessages(
721
+ messages: Array<{ role: string; content: string }>,
722
+ ): Array<{ input: string; expect?: EvalExpect }> {
723
+ const turns: Array<{ input: string; expect?: EvalExpect }> = [];
724
+
725
+ for (let index = 0; index < messages.length; index += 1) {
726
+ const message = messages[index];
727
+
728
+ if (message.role !== "user") {
729
+ continue;
730
+ }
731
+
732
+ const nextAssistant = messages.slice(index + 1).find((candidate) => candidate.role === "assistant");
733
+ const input = redactEvalText(message.content);
734
+ const contains = nextAssistant ? redactEvalText(nextAssistant.content) : undefined;
735
+ const redactionPresent = input.includes("[redacted_") || Boolean(contains?.includes("[redacted_"));
736
+
737
+ turns.push({
738
+ input,
739
+ ...(nextAssistant
740
+ ? {
741
+ expect: {
742
+ response: redactionPresent
743
+ ? {
744
+ notRegex: ["API_KEY|Bearer\\s+|(?:sk|agk|ak|dpat)[_-]|[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,}"],
745
+ }
746
+ : {
747
+ contains: contains ?? "",
748
+ },
749
+ },
750
+ }
751
+ : {}),
752
+ });
753
+ }
754
+
755
+ return turns;
756
+ }
757
+
758
+ function redactEvalText(value: string): string {
759
+ return value
760
+ .replace(/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi, "[redacted_email]")
761
+ .replace(
762
+ /\b[A-Z0-9_]*(?:API[_-]?KEY|TOKEN|SECRET|PASSWORD|PRIVATE[_-]?KEY|CLIENT[_-]?SECRET)\s*=\s*[^\s,;]+/gi,
763
+ (match) => `${match.slice(0, match.indexOf("=")).trim()}=[redacted_secret]`,
764
+ )
765
+ .replace(/\b(?:sk|agk|ak|dpat)[_-][A-Za-z0-9_-]{8,}\b/g, "[redacted_token]")
766
+ .replace(/\bBearer\s+[A-Za-z0-9._~+/-]+=*/gi, "Bearer [redacted_token]")
767
+ .replace(/\+?\d[\d\s().-]{7,}\d/g, "[redacted_phone]");
768
+ }
769
+
770
+ function formatEvalTurns(turns: Array<{ input: string; expect?: EvalExpect }>): string {
771
+ return JSON.stringify(turns, null, 4)
772
+ .split("\n")
773
+ .map((line, index) => (index === 0 ? line : ` ${line}`))
774
+ .join("\n");
775
+ }
776
+
777
+ function slugify(value: string): string {
778
+ const slug = value
779
+ .toLowerCase()
780
+ .replace(/[^a-z0-9]+/g, "-")
781
+ .replace(/^-+|-+$/g, "")
782
+ .slice(0, 48);
783
+
784
+ return slug || "conversation";
785
+ }
786
+
787
+ function resolveEvalOutputPath(root: string, out: string | undefined, title: string): string {
788
+ const outputPath = out ?? join("evals", `replay-${slugify(title)}.eval.ts`);
789
+
790
+ if (isAbsolute(outputPath)) {
791
+ throw new AgentKitError("validation_error", "eval --out must be a relative path inside the Agent Capsule.");
792
+ }
793
+
794
+ const file = resolve(root, outputPath);
795
+ const relativePath = relative(root, file);
796
+
797
+ if (relativePath === "" || relativePath.startsWith("..") || isAbsolute(relativePath)) {
798
+ throw new AgentKitError("validation_error", "eval --out must stay inside the Agent Capsule.");
799
+ }
800
+
801
+ if (!/\.(eval|spec)\.[cm]?[tj]s$/.test(file)) {
802
+ throw new AgentKitError(
803
+ "validation_error",
804
+ "eval --out must end with .eval.ts, .eval.js, .spec.ts, or another AgentKit eval/spec extension.",
805
+ );
806
+ }
807
+
808
+ return file;
809
+ }
810
+
811
+ function formatEvalFailure(error: unknown): string {
812
+ if (isAgentKitError(error)) {
813
+ return `${error.code}: ${error.message}`;
814
+ }
815
+
816
+ if (error instanceof Error) {
817
+ return `runtime_error: ${error.message}`;
818
+ }
819
+
820
+ return `runtime_error: ${String(error)}`;
821
+ }
822
+
277
823
  function jsonContains(actual: unknown, expected: unknown): boolean {
278
824
  if (Array.isArray(expected)) {
279
825
  return jsonEqual(actual, expected);