@andreprado/agentkit 0.1.0-alpha.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/README.md +69 -0
  2. package/bin/agentkit.mjs +23 -0
  3. package/docs/guides/add-channel.md +114 -0
  4. package/docs/guides/add-knowledge.md +134 -0
  5. package/docs/guides/add-tool.md +342 -0
  6. package/docs/guides/agentkit-skills-architecture.md +471 -0
  7. package/docs/guides/channel-security.md +81 -0
  8. package/docs/guides/channels-implementation-map.md +243 -0
  9. package/docs/guides/channels-production-handoff.md +102 -0
  10. package/docs/guides/connect-telegram.md +110 -0
  11. package/docs/guides/connect-whatsapp-zapster.md +119 -0
  12. package/docs/guides/create-agent.md +220 -0
  13. package/docs/guides/prepare-deploy.md +209 -0
  14. package/docs/guides/run-evals.md +179 -0
  15. package/docs/guides/security-rules.md +156 -0
  16. package/docs/guides/use-provider.md +140 -0
  17. package/docs/llms-full.txt +876 -0
  18. package/docs/llms.txt +83 -0
  19. package/docs/portable-deploy-release-checklist.md +41 -0
  20. package/package.json +47 -0
  21. package/src/cli/args.ts +36 -0
  22. package/src/cli/cloud-client.ts +265 -0
  23. package/src/cli/commands/channels.ts +810 -0
  24. package/src/cli/commands/knowledge.ts +136 -0
  25. package/src/cli/constants.ts +4 -0
  26. package/src/cli/deploy-chat-ui.ts +392 -0
  27. package/src/cli/deploy-readiness.ts +348 -0
  28. package/src/cli/flags.ts +162 -0
  29. package/src/cli/help.ts +184 -0
  30. package/src/cli/index.ts +1276 -0
  31. package/src/cli/process.ts +31 -0
  32. package/src/cloud/artifact.ts +139 -0
  33. package/src/cloud/client.ts +79 -0
  34. package/src/cloud/contracts.ts +63 -0
  35. package/src/cloud/index.ts +3 -0
  36. package/src/create-project.ts +177 -0
  37. package/src/index.ts +408 -0
  38. package/src/providers/index.ts +25 -0
  39. package/src/providers/pi.ts +286 -0
  40. package/src/providers/test.ts +133 -0
  41. package/src/providers/types.ts +34 -0
  42. package/src/runtime/build.ts +43 -0
  43. package/src/runtime/channel-buffer.ts +30 -0
  44. package/src/runtime/channel-test-harness.ts +112 -0
  45. package/src/runtime/channels/telegram.ts +360 -0
  46. package/src/runtime/channels/website.ts +132 -0
  47. package/src/runtime/channels/whatsapp-meta.ts +71 -0
  48. package/src/runtime/channels/whatsapp-zapster.ts +278 -0
  49. package/src/runtime/channels.ts +138 -0
  50. package/src/runtime/chat.ts +218 -0
  51. package/src/runtime/config.ts +684 -0
  52. package/src/runtime/conversations.ts +38 -0
  53. package/src/runtime/core/deploy-state.ts +54 -0
  54. package/src/runtime/core/manifest.ts +213 -0
  55. package/src/runtime/core/targets.ts +133 -0
  56. package/src/runtime/database.ts +256 -0
  57. package/src/runtime/db-commands.ts +167 -0
  58. package/src/runtime/deploy-readiness.ts +105 -0
  59. package/src/runtime/deploy.ts +1 -0
  60. package/src/runtime/dev-server.ts +1247 -0
  61. package/src/runtime/docs.ts +36 -0
  62. package/src/runtime/env.ts +152 -0
  63. package/src/runtime/errors.ts +13 -0
  64. package/src/runtime/evals.ts +509 -0
  65. package/src/runtime/inspect.ts +203 -0
  66. package/src/runtime/knowledge/chunk.ts +333 -0
  67. package/src/runtime/knowledge/config.ts +135 -0
  68. package/src/runtime/knowledge/embeddings.ts +133 -0
  69. package/src/runtime/knowledge/ingest.ts +521 -0
  70. package/src/runtime/knowledge/prompt-policy.ts +30 -0
  71. package/src/runtime/knowledge/retrieve.ts +283 -0
  72. package/src/runtime/knowledge/schema.ts +56 -0
  73. package/src/runtime/knowledge/tool.ts +64 -0
  74. package/src/runtime/knowledge/vector.ts +258 -0
  75. package/src/runtime/runtime-contract.ts +93 -0
  76. package/src/runtime/spec.ts +152 -0
  77. package/src/runtime/sync.ts +144 -0
  78. package/src/runtime/targets/cloudflare/build.ts +2517 -0
  79. package/src/runtime/targets/container/build.ts +146 -0
  80. package/src/runtime/targets/container/server.ts +33 -0
  81. package/src/runtime/targets/vps/deploy.ts +206 -0
  82. package/src/runtime/tool-runner.ts +65 -0
  83. package/src/runtime/tools.ts +470 -0
  84. package/src/runtime/traces.ts +41 -0
  85. package/src/storage/sqlite.ts +1118 -0
  86. package/src/templates/blank.ts +394 -0
  87. package/src/templates/dentista.ts +1003 -0
  88. package/src/templates/index.ts +33 -0
  89. package/src/templates/skills/agentkit-build-agent/SKILL.md +51 -0
  90. package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +20 -0
  91. package/src/templates/skills/agentkit-build-agent/templates/sales-qualifier.instructions.md +17 -0
  92. package/src/templates/skills/agentkit-build-agent/templates/support-agent.instructions.md +16 -0
  93. package/src/templates/skills/agentkit-capsule/SKILL.md +62 -0
  94. package/src/templates/skills/agentkit-capsule/references/docs-router.md +15 -0
  95. package/src/templates/skills/agentkit-channels/SKILL.md +62 -0
  96. package/src/templates/skills/agentkit-channels/references/channel-buffering.md +58 -0
  97. package/src/templates/skills/agentkit-channels/references/channel-debugging.md +41 -0
  98. package/src/templates/skills/agentkit-channels/references/telegram.md +38 -0
  99. package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +44 -0
  100. package/src/templates/skills/agentkit-database/SKILL.md +45 -0
  101. package/src/templates/skills/agentkit-database/templates/appointments.schema.sql +15 -0
  102. package/src/templates/skills/agentkit-database/templates/leads.schema.sql +17 -0
  103. package/src/templates/skills/agentkit-deploy/SKILL.md +44 -0
  104. package/src/templates/skills/agentkit-evals/SKILL.md +60 -0
  105. package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +22 -0
  106. package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +14 -0
  107. package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +14 -0
  108. package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +18 -0
  109. package/src/templates/skills/agentkit-knowledge/SKILL.md +40 -0
  110. package/src/templates/skills/agentkit-knowledge/templates/faq.md +14 -0
  111. package/src/templates/skills/agentkit-knowledge/templates/policies.md +14 -0
  112. package/src/templates/skills/agentkit-knowledge/templates/prices.csv +3 -0
  113. package/src/templates/skills/agentkit-prompts/SKILL.md +45 -0
  114. package/src/templates/skills/agentkit-prompts/templates/knowledge-grounded-faq.instructions.md +11 -0
  115. package/src/templates/skills/agentkit-provider/SKILL.md +57 -0
  116. package/src/templates/skills/agentkit-security/SKILL.md +55 -0
  117. package/src/templates/skills/agentkit-tools/SKILL.md +36 -0
  118. package/src/templates/skills/agentkit-tools/examples/database-write.tool.md +35 -0
  119. package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +37 -0
  120. package/src/templates/skills/agentkit-tools/examples/lookup-order.tool.md +46 -0
  121. package/src/templates/skills/agentkit-troubleshooting/SKILL.md +52 -0
  122. package/src/templates/support.ts +401 -0
@@ -0,0 +1,36 @@
1
+ import { existsSync } from "node:fs";
2
+ import { dirname, join, resolve } from "node:path";
3
+ import { fileURLToPath } from "node:url";
4
+
5
+ import { AgentKitError } from "./errors";
6
+
7
+ export type AgentKitDocsInfo = {
8
+ root: string;
9
+ llms: string;
10
+ llmsFull: string;
11
+ };
12
+
13
+ export function findAgentKitDocs(): AgentKitDocsInfo {
14
+ const candidates = [
15
+ resolve(dirname(fileURLToPath(import.meta.url)), "../../docs"),
16
+ resolve(dirname(fileURLToPath(import.meta.url)), "../../../../docs"),
17
+ ];
18
+
19
+ for (const root of candidates) {
20
+ const llms = join(root, "llms.txt");
21
+ const llmsFull = join(root, "llms-full.txt");
22
+
23
+ if (existsSync(llms) && existsSync(llmsFull)) {
24
+ return {
25
+ root,
26
+ llms,
27
+ llmsFull,
28
+ };
29
+ }
30
+ }
31
+
32
+ throw new AgentKitError(
33
+ "docs_not_found",
34
+ "Could not find packaged AgentKit docs. Reinstall @andreprado/agentkit or read the repository docs directly.",
35
+ );
36
+ }
@@ -0,0 +1,152 @@
1
+ import { readFile, writeFile } from "node:fs/promises";
2
+ import { join } from "node:path";
3
+
4
+ import { AgentKitError } from "./errors";
5
+
6
+ export async function loadCapsuleEnv(
7
+ root: string,
8
+ baseEnv: Record<string, string | undefined> = process.env,
9
+ ): Promise<Record<string, string | undefined>> {
10
+ const parsed = await readDotEnv(join(root, ".env"));
11
+ return {
12
+ ...parsed,
13
+ ...baseEnv,
14
+ };
15
+ }
16
+
17
+ async function readDotEnv(path: string): Promise<Record<string, string>> {
18
+ let source: string;
19
+
20
+ try {
21
+ source = await readFile(path, "utf8");
22
+ } catch {
23
+ return {};
24
+ }
25
+
26
+ const env: Record<string, string> = {};
27
+
28
+ for (const rawLine of source.split(/\r?\n/)) {
29
+ const line = rawLine.trim();
30
+
31
+ if (!line || line.startsWith("#")) {
32
+ continue;
33
+ }
34
+
35
+ const equalsIndex = line.indexOf("=");
36
+
37
+ if (equalsIndex <= 0) {
38
+ continue;
39
+ }
40
+
41
+ const name = line.slice(0, equalsIndex).trim();
42
+ const rawValue = line.slice(equalsIndex + 1).trim();
43
+
44
+ if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(name)) {
45
+ continue;
46
+ }
47
+
48
+ env[name] = unquote(rawValue);
49
+ }
50
+
51
+ return env;
52
+ }
53
+
54
+ export async function listCapsuleEnv(root: string): Promise<string[]> {
55
+ return Object.keys(await readDotEnv(join(root, ".env"))).sort();
56
+ }
57
+
58
+ export async function setCapsuleEnv(root: string, name: string, value: string): Promise<void> {
59
+ validateEnvName(name);
60
+
61
+ const path = join(root, ".env");
62
+ const source = await readDotEnvSource(path);
63
+ const line = `${name}=${quoteDotEnvValue(value)}`;
64
+ const lines = source.length > 0 ? source.split(/\r?\n/) : [];
65
+ let updated = false;
66
+
67
+ const nextLines = lines.map((existingLine) => {
68
+ if (isEnvAssignmentFor(existingLine, name)) {
69
+ updated = true;
70
+ return line;
71
+ }
72
+
73
+ return existingLine;
74
+ });
75
+
76
+ if (!updated) {
77
+ if (nextLines.length > 0 && nextLines[nextLines.length - 1] !== "") {
78
+ nextLines.push(line);
79
+ } else if (nextLines.length > 0) {
80
+ nextLines.splice(nextLines.length - 1, 0, line);
81
+ } else {
82
+ nextLines.push(line);
83
+ }
84
+ }
85
+
86
+ await writeFile(path, `${nextLines.join("\n").replace(/\n*$/, "")}\n`);
87
+ }
88
+
89
+ export async function unsetCapsuleEnv(root: string, name: string): Promise<boolean> {
90
+ validateEnvName(name);
91
+
92
+ const path = join(root, ".env");
93
+ const source = await readDotEnvSource(path);
94
+
95
+ if (!source) {
96
+ return false;
97
+ }
98
+
99
+ const lines = source.split(/\r?\n/);
100
+ const nextLines = lines.filter((line) => !isEnvAssignmentFor(line, name));
101
+
102
+ if (nextLines.length === lines.length) {
103
+ return false;
104
+ }
105
+
106
+ await writeFile(path, `${nextLines.join("\n").replace(/\n*$/, "")}\n`);
107
+ return true;
108
+ }
109
+
110
+ async function readDotEnvSource(path: string): Promise<string> {
111
+ try {
112
+ return await readFile(path, "utf8");
113
+ } catch {
114
+ return "";
115
+ }
116
+ }
117
+
118
+ function validateEnvName(name: string): void {
119
+ if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(name)) {
120
+ throw new AgentKitError(
121
+ "validation_error",
122
+ "Environment variable names must start with a letter or underscore and contain only letters, numbers, and underscores.",
123
+ );
124
+ }
125
+ }
126
+
127
+ function isEnvAssignmentFor(line: string, name: string): boolean {
128
+ const match = line.match(/^\s*([A-Za-z_][A-Za-z0-9_]*)\s*=/);
129
+ return match?.[1] === name;
130
+ }
131
+
132
+ function unquote(value: string): string {
133
+ if (value.length >= 2 && value.startsWith('"') && value.endsWith('"')) {
134
+ return value
135
+ .slice(1, -1)
136
+ .replace(/\\n/g, "\n")
137
+ .replace(/\\r/g, "\r")
138
+ .replace(/\\"/g, '"')
139
+ .replace(/\\\\/g, "\\");
140
+ }
141
+
142
+ if (value.length >= 2 && value.startsWith("'") && value.endsWith("'")) {
143
+ return value.slice(1, -1);
144
+ }
145
+
146
+ const commentIndex = value.search(/\s#/);
147
+ return (commentIndex === -1 ? value : value.slice(0, commentIndex)).trim();
148
+ }
149
+
150
+ function quoteDotEnvValue(value: string): string {
151
+ return `"${value.replace(/\\/g, "\\\\").replace(/\r/g, "\\r").replace(/\n/g, "\\n").replace(/"/g, '\\"')}"`;
152
+ }
@@ -0,0 +1,13 @@
1
+ export class AgentKitError extends Error {
2
+ readonly code: string;
3
+
4
+ constructor(code: string, message: string, options?: ErrorOptions) {
5
+ super(message, options);
6
+ this.name = "AgentKitError";
7
+ this.code = code;
8
+ }
9
+ }
10
+
11
+ export function isAgentKitError(error: unknown): error is AgentKitError {
12
+ return error instanceof AgentKitError;
13
+ }
@@ -0,0 +1,509 @@
1
+ import { mkdir, readdir, stat, writeFile } from "node:fs/promises";
2
+ import type { Dirent } from "node:fs";
3
+ import { dirname, join, relative, resolve } from "node:path";
4
+ import { pathToFileURL } from "node:url";
5
+
6
+ import type { AgentRunResult } from "./chat";
7
+ import { runAgentMessageFromCwd } from "./chat";
8
+ import { findAgentCapsuleRoot, loadAgentCapsule } from "./config";
9
+ import { AgentKitError } from "./errors";
10
+ import { openCapsuleStore, type StoredToolCall } from "../storage/sqlite";
11
+ import { getConversationTraceFromCwd } from "./traces";
12
+
13
+ export type EvalRunSummary = {
14
+ root: string;
15
+ results: EvalResult[];
16
+ passed: number;
17
+ failed: number;
18
+ };
19
+
20
+ export type EvalResult = {
21
+ name: string;
22
+ file: string;
23
+ passed: boolean;
24
+ failures: string[];
25
+ output: string;
26
+ };
27
+
28
+ type EvalCase = {
29
+ name?: string;
30
+ input?: string;
31
+ turns?: EvalTurn[];
32
+ expect?: EvalExpect;
33
+ };
34
+
35
+ type EvalTurn =
36
+ | string
37
+ | {
38
+ input: string;
39
+ expect?: EvalExpect;
40
+ };
41
+
42
+ type EvalExpect = {
43
+ contains?: string | string[];
44
+ not_contains?: string | string[];
45
+ notContains?: string | string[];
46
+ regex?: string | string[];
47
+ matches_regex?: string | string[];
48
+ matchesRegex?: string | string[];
49
+ tool_call?: ToolExpectation | ToolExpectation[];
50
+ toolCall?: ToolExpectation | ToolExpectation[];
51
+ tool_calls?: ToolExpectation | ToolExpectation[];
52
+ toolCalls?: ToolExpectation | ToolExpectation[];
53
+ persisted_tool_call?: ToolExpectation | ToolExpectation[];
54
+ persistedToolCall?: ToolExpectation | ToolExpectation[];
55
+ persisted_tool_calls?: ToolExpectation | ToolExpectation[];
56
+ persistedToolCalls?: ToolExpectation | ToolExpectation[];
57
+ };
58
+
59
+ type ToolExpectation =
60
+ | string
61
+ | {
62
+ name?: string;
63
+ input?: unknown;
64
+ output?: unknown;
65
+ status?: "running" | "completed" | "failed";
66
+ visibility?: "user" | "internal";
67
+ rendered?: unknown;
68
+ };
69
+
70
+ type ToolCallSnapshot = {
71
+ name?: unknown;
72
+ input?: unknown;
73
+ output?: unknown;
74
+ rendered?: unknown;
75
+ status?: unknown;
76
+ visibility?: unknown;
77
+ };
78
+
79
+ export type EvalFromConversationResult = {
80
+ root: string;
81
+ file: string;
82
+ conversationId: string;
83
+ turns: number;
84
+ };
85
+
86
+ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
87
+ const root = await findAgentCapsuleRoot(cwd);
88
+ const evalFiles = await findEvalFiles(join(root, "evals"));
89
+
90
+ if (evalFiles.length === 0) {
91
+ throw new AgentKitError("validation_error", "No eval files found in evals/. Add evals/<name>.eval.ts first.");
92
+ }
93
+
94
+ const results: EvalResult[] = [];
95
+
96
+ for (const file of evalFiles) {
97
+ const evalCase = await loadEvalCase(file);
98
+ const run = await runEvalCase(root, relative(root, file), evalCase);
99
+
100
+ results.push({
101
+ name: evalCase.name ?? relative(root, file),
102
+ file: relative(root, file),
103
+ passed: run.failures.length === 0,
104
+ failures: run.failures,
105
+ output: run.output,
106
+ });
107
+ }
108
+
109
+ const passed = results.filter((result) => result.passed).length;
110
+
111
+ return {
112
+ root,
113
+ results,
114
+ passed,
115
+ failed: results.length - passed,
116
+ };
117
+ }
118
+
119
+ export async function writeEvalFromConversation(
120
+ cwd: string,
121
+ input: {
122
+ conversationId: string;
123
+ out?: string;
124
+ name?: string;
125
+ force?: boolean;
126
+ },
127
+ ): Promise<EvalFromConversationResult> {
128
+ const root = await findAgentCapsuleRoot(cwd);
129
+ const trace = await getConversationTraceFromCwd(root, input.conversationId);
130
+ const turns = replayTurnsFromMessages(trace.messages);
131
+
132
+ if (turns.length === 0) {
133
+ throw new AgentKitError(
134
+ "validation_error",
135
+ `Conversation "${input.conversationId}" does not contain user/assistant turns that can become an eval.`,
136
+ );
137
+ }
138
+
139
+ const file = resolve(root, input.out ?? join("evals", `replay-${slugify(trace.title ?? trace.id)}.eval.ts`));
140
+
141
+ if (!input.force && await pathExists(file)) {
142
+ throw new AgentKitError(
143
+ "validation_error",
144
+ `${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
145
+ );
146
+ }
147
+
148
+ await mkdir(dirname(file), { recursive: true });
149
+ await writeFile(
150
+ file,
151
+ `export default {
152
+ name: ${JSON.stringify(input.name ?? `replay ${trace.title ?? trace.id}`)},
153
+ turns: ${formatEvalTurns(turns)},
154
+ };
155
+ `,
156
+ );
157
+
158
+ return {
159
+ root,
160
+ file,
161
+ conversationId: trace.id,
162
+ turns: turns.length,
163
+ };
164
+ }
165
+
166
+ async function runEvalCase(
167
+ root: string,
168
+ file: string,
169
+ evalCase: EvalCase,
170
+ ): Promise<{ failures: string[]; output: string }> {
171
+ const turns = normalizeEvalTurns(evalCase);
172
+
173
+ if (turns.length === 0) {
174
+ throw new AgentKitError(
175
+ "validation_error",
176
+ `${file} must export either a non-empty input string or a non-empty turns array.`,
177
+ );
178
+ }
179
+
180
+ const conversationId = `eval_${crypto.randomUUID()}`;
181
+ const failures: string[] = [];
182
+ let output = "";
183
+
184
+ for (const [index, turn] of turns.entries()) {
185
+ const run = await runAgentMessageFromCwd(root, {
186
+ message: turn.input,
187
+ conversationId,
188
+ runtime: {
189
+ environment: "eval",
190
+ invocation: "eval",
191
+ },
192
+ });
193
+ const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
194
+ const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
195
+
196
+ for (const failure of turnFailures) {
197
+ failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
198
+ }
199
+
200
+ output = run.message.content;
201
+ }
202
+
203
+ return { failures, output };
204
+ }
205
+
206
+ function normalizeEvalTurns(evalCase: EvalCase): Array<{ input: string; expect?: EvalExpect }> {
207
+ if (evalCase.turns !== undefined) {
208
+ if (!Array.isArray(evalCase.turns)) {
209
+ throw new AgentKitError("validation_error", "eval turns must be an array when provided.");
210
+ }
211
+
212
+ return evalCase.turns.map((turn, index) => normalizeEvalTurn(turn, index));
213
+ }
214
+
215
+ if (typeof evalCase.input === "string" && evalCase.input.trim().length > 0) {
216
+ return [
217
+ {
218
+ input: evalCase.input,
219
+ expect: evalCase.expect,
220
+ },
221
+ ];
222
+ }
223
+
224
+ return [];
225
+ }
226
+
227
+ function normalizeEvalTurn(turn: EvalTurn, index: number): { input: string; expect?: EvalExpect } {
228
+ if (typeof turn === "string") {
229
+ if (turn.trim().length === 0) {
230
+ throw new AgentKitError("validation_error", `eval turns[${index}] must be non-empty.`);
231
+ }
232
+
233
+ return { input: turn };
234
+ }
235
+
236
+ if (!isRecord(turn) || typeof turn.input !== "string" || turn.input.trim().length === 0) {
237
+ throw new AgentKitError("validation_error", `eval turns[${index}].input must be a non-empty string.`);
238
+ }
239
+
240
+ return {
241
+ input: turn.input,
242
+ ...(turn.expect ? { expect: turn.expect } : {}),
243
+ };
244
+ }
245
+
246
+ async function findEvalFiles(directory: string): Promise<string[]> {
247
+ let entries: Dirent[];
248
+
249
+ try {
250
+ entries = await readdir(directory, { withFileTypes: true });
251
+ } catch (error) {
252
+ if (isNodeError(error) && error.code === "ENOENT") {
253
+ return [];
254
+ }
255
+
256
+ throw error;
257
+ }
258
+
259
+ const files = await Promise.all(
260
+ entries.map(async (entry) => {
261
+ const path = join(directory, entry.name);
262
+
263
+ if (entry.isDirectory()) {
264
+ return findEvalFiles(path);
265
+ }
266
+
267
+ if (entry.isFile() && /\.(eval|spec)\.[cm]?[tj]s$/.test(entry.name)) {
268
+ return [path];
269
+ }
270
+
271
+ return [];
272
+ }),
273
+ );
274
+
275
+ return files.flat().sort();
276
+ }
277
+
278
+ async function loadEvalCase(file: string): Promise<EvalCase> {
279
+ const url = pathToFileURL(file);
280
+ url.searchParams.set("t", String(Date.now()));
281
+
282
+ const module = (await import(url.href)) as { default?: unknown };
283
+
284
+ if (!isRecord(module.default)) {
285
+ throw new AgentKitError("validation_error", `${file} must default export an eval object.`);
286
+ }
287
+
288
+ return module.default;
289
+ }
290
+
291
+ async function loadPersistedToolCalls(root: string, runId: string): Promise<ToolCallSnapshot[]> {
292
+ const capsule = await loadAgentCapsule(root);
293
+ const store = await openCapsuleStore(capsule);
294
+
295
+ try {
296
+ return store.listToolCallsForRun(runId).map(toolCallSnapshotFromStored);
297
+ } finally {
298
+ store.close();
299
+ }
300
+ }
301
+
302
+ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persistedToolCalls: ToolCallSnapshot[]): string[] {
303
+ const failures: string[] = [];
304
+ const content = run.message.content;
305
+
306
+ for (const expected of list(expect.contains)) {
307
+ if (!content.includes(expected)) {
308
+ failures.push(`expected output to contain ${JSON.stringify(expected)}`);
309
+ }
310
+ }
311
+
312
+ for (const forbidden of [...list(expect.not_contains), ...list(expect.notContains)]) {
313
+ if (content.includes(forbidden)) {
314
+ failures.push(`expected output not to contain ${JSON.stringify(forbidden)}`);
315
+ }
316
+ }
317
+
318
+ for (const pattern of [...list(expect.regex), ...list(expect.matches_regex), ...list(expect.matchesRegex)]) {
319
+ if (!new RegExp(pattern).test(content)) {
320
+ failures.push(`expected output to match /${pattern}/`);
321
+ }
322
+ }
323
+
324
+ const toolExpectations = [
325
+ ...toolList(expect.tool_call),
326
+ ...toolList(expect.toolCall),
327
+ ...toolList(expect.tool_calls),
328
+ ...toolList(expect.toolCalls),
329
+ ...toolList(expect.persisted_tool_call),
330
+ ...toolList(expect.persistedToolCall),
331
+ ...toolList(expect.persisted_tool_calls),
332
+ ...toolList(expect.persistedToolCalls),
333
+ ];
334
+
335
+ for (const toolExpectation of toolExpectations) {
336
+ const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
337
+
338
+ if (!match) {
339
+ failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
340
+ }
341
+ }
342
+
343
+ return failures;
344
+ }
345
+
346
+ function list(value: string | string[] | undefined): string[] {
347
+ if (value === undefined) {
348
+ return [];
349
+ }
350
+
351
+ return Array.isArray(value) ? value : [value];
352
+ }
353
+
354
+ function toolList(value: ToolExpectation | ToolExpectation[] | undefined): ToolExpectation[] {
355
+ if (value === undefined) {
356
+ return [];
357
+ }
358
+
359
+ return Array.isArray(value) ? value : [value];
360
+ }
361
+
362
+ function matchesToolExpectation(toolCall: ToolCallSnapshot, expectation: ToolExpectation): boolean {
363
+ if (typeof expectation === "string") {
364
+ return toolCall.name === expectation;
365
+ }
366
+
367
+ if (expectation.name !== undefined && toolCall.name !== expectation.name) {
368
+ return false;
369
+ }
370
+
371
+ if ("status" in expectation && toolCall.status !== expectation.status) {
372
+ return false;
373
+ }
374
+
375
+ if ("visibility" in expectation && toolCall.visibility !== expectation.visibility) {
376
+ return false;
377
+ }
378
+
379
+ if ("input" in expectation && !jsonContains(toolCall.input, expectation.input)) {
380
+ return false;
381
+ }
382
+
383
+ if ("output" in expectation && !jsonContains(toolCall.output, expectation.output)) {
384
+ return false;
385
+ }
386
+
387
+ if ("rendered" in expectation && !jsonContains(toolCall.rendered, expectation.rendered)) {
388
+ return false;
389
+ }
390
+
391
+ return true;
392
+ }
393
+
394
+ function formatToolExpectation(expectation: ToolExpectation): string {
395
+ if (typeof expectation === "string") {
396
+ return JSON.stringify({ name: expectation });
397
+ }
398
+
399
+ return JSON.stringify(expectation);
400
+ }
401
+
402
+ function toolCallSnapshotFromStored(toolCall: StoredToolCall): ToolCallSnapshot {
403
+ return {
404
+ name: toolCall.toolName,
405
+ input: toolCall.input,
406
+ output: toolCall.output,
407
+ rendered: toolCall.rendered,
408
+ status: toolCall.status,
409
+ visibility: toolCall.visibility,
410
+ };
411
+ }
412
+
413
+ function replayTurnsFromMessages(
414
+ messages: Array<{ role: string; content: string }>,
415
+ ): Array<{ input: string; expect?: EvalExpect }> {
416
+ const turns: Array<{ input: string; expect?: EvalExpect }> = [];
417
+
418
+ for (let index = 0; index < messages.length; index += 1) {
419
+ const message = messages[index];
420
+
421
+ if (message.role !== "user") {
422
+ continue;
423
+ }
424
+
425
+ const nextAssistant = messages.slice(index + 1).find((candidate) => candidate.role === "assistant");
426
+ turns.push({
427
+ input: message.content,
428
+ ...(nextAssistant
429
+ ? {
430
+ expect: {
431
+ contains: nextAssistant.content,
432
+ },
433
+ }
434
+ : {}),
435
+ });
436
+ }
437
+
438
+ return turns;
439
+ }
440
+
441
+ function formatEvalTurns(turns: Array<{ input: string; expect?: EvalExpect }>): string {
442
+ return JSON.stringify(turns, null, 4)
443
+ .split("\n")
444
+ .map((line, index) => (index === 0 ? line : ` ${line}`))
445
+ .join("\n");
446
+ }
447
+
448
+ function slugify(value: string): string {
449
+ const slug = value
450
+ .toLowerCase()
451
+ .replace(/[^a-z0-9]+/g, "-")
452
+ .replace(/^-+|-+$/g, "")
453
+ .slice(0, 48);
454
+
455
+ return slug || "conversation";
456
+ }
457
+
458
+ async function pathExists(path: string): Promise<boolean> {
459
+ try {
460
+ await stat(path);
461
+ return true;
462
+ } catch {
463
+ return false;
464
+ }
465
+ }
466
+
467
+ function jsonContains(actual: unknown, expected: unknown): boolean {
468
+ if (Array.isArray(expected)) {
469
+ return jsonEqual(actual, expected);
470
+ }
471
+
472
+ if (isRecord(expected)) {
473
+ if (!isRecord(actual)) {
474
+ return false;
475
+ }
476
+
477
+ return Object.entries(expected).every(([key, value]) => key in actual && jsonContains(actual[key], value));
478
+ }
479
+
480
+ return jsonEqual(actual, expected);
481
+ }
482
+
483
+ function jsonEqual(left: unknown, right: unknown): boolean {
484
+ return JSON.stringify(sortJson(left)) === JSON.stringify(sortJson(right));
485
+ }
486
+
487
+ function sortJson(value: unknown): unknown {
488
+ if (Array.isArray(value)) {
489
+ return value.map(sortJson);
490
+ }
491
+
492
+ if (!isRecord(value)) {
493
+ return value;
494
+ }
495
+
496
+ return Object.fromEntries(
497
+ Object.entries(value)
498
+ .sort(([left], [right]) => left.localeCompare(right))
499
+ .map(([key, item]) => [key, sortJson(item)]),
500
+ );
501
+ }
502
+
503
+ function isRecord(value: unknown): value is Record<string, unknown> {
504
+ return typeof value === "object" && value !== null && !Array.isArray(value);
505
+ }
506
+
507
+ function isNodeError(error: unknown): error is NodeJS.ErrnoException {
508
+ return error instanceof Error && "code" in error;
509
+ }