@andreprado/agentkit 0.1.0-alpha.2 → 0.1.0-alpha.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +67 -6
- package/docs/guides/add-channel.md +118 -7
- package/docs/guides/add-knowledge.md +144 -0
- package/docs/guides/add-managed-composio.md +163 -0
- package/docs/guides/add-tool.md +1 -1
- package/docs/guides/channel-security.md +97 -32
- package/docs/guides/connect-discord.md +178 -0
- package/docs/guides/connect-slack.md +126 -0
- package/docs/guides/connect-telegram.md +78 -1
- package/docs/guides/connect-whatsapp-zapster.md +112 -8
- package/docs/guides/create-agent.md +45 -4
- package/docs/guides/debug-channel.md +147 -0
- package/docs/guides/improve-from-production.md +151 -0
- package/docs/guides/prepare-deploy.md +47 -17
- package/docs/guides/replay-production-traces.md +72 -0
- package/docs/guides/run-evals.md +147 -20
- package/docs/guides/security-rules.md +7 -6
- package/docs/guides/send-feedback.md +135 -0
- package/docs/guides/use-provider.md +27 -3
- package/docs/llms-full.txt +303 -55
- package/docs/llms.txt +57 -7
- package/package.json +2 -5
- package/src/cli/args.ts +57 -0
- package/src/cli/cloud-client.ts +377 -0
- package/src/cli/commands/channels.ts +1315 -0
- package/src/cli/commands/feedback.ts +438 -0
- package/src/cli/commands/knowledge.ts +136 -0
- package/src/cli/commands/transcribe.ts +171 -0
- package/src/cli/constants.ts +4 -0
- package/src/cli/deploy-chat-ui.ts +535 -0
- package/src/cli/deploy-readiness.ts +481 -0
- package/src/cli/flags.ts +162 -0
- package/src/cli/help.ts +236 -0
- package/src/cli/index.ts +1167 -1005
- package/src/cli/process.ts +31 -0
- package/src/cloud/artifact.ts +139 -0
- package/src/cloud/client.ts +80 -0
- package/src/cloud/contracts.ts +63 -0
- package/src/cloud/index.ts +3 -0
- package/src/create-project.ts +21 -6
- package/src/index.ts +479 -7
- package/src/providers/pi.ts +70 -16
- package/src/providers/test.ts +88 -1
- package/src/providers/types.ts +7 -0
- package/src/runtime/channel-buffer.ts +30 -0
- package/src/runtime/channel-test-harness.ts +8 -1
- package/src/runtime/channels/discord.ts +896 -0
- package/src/runtime/channels/slack.ts +646 -0
- package/src/runtime/channels/telegram.ts +466 -23
- package/src/runtime/channels/whatsapp-meta.ts +9 -0
- package/src/runtime/channels/whatsapp-zapster.ts +677 -40
- package/src/runtime/channels.ts +86 -3
- package/src/runtime/chat.ts +130 -38
- package/src/runtime/config.ts +483 -18
- package/src/runtime/core/manifest.ts +103 -5
- package/src/runtime/core/targets.ts +5 -5
- package/src/runtime/database.ts +93 -2
- package/src/runtime/db-commands.ts +9 -0
- package/src/runtime/deploy-readiness.ts +46 -4
- package/src/runtime/deploy.ts +1 -1
- package/src/runtime/dev-server.ts +759 -41
- package/src/runtime/env.ts +8 -3
- package/src/runtime/evals.ts +589 -43
- package/src/runtime/improve.ts +868 -0
- package/src/runtime/inspect.ts +194 -4
- package/src/runtime/integrations/composio.ts +423 -0
- package/src/runtime/knowledge/chunk.ts +333 -0
- package/src/runtime/knowledge/config.ts +135 -0
- package/src/runtime/knowledge/embeddings.ts +133 -0
- package/src/runtime/knowledge/ingest.ts +521 -0
- package/src/runtime/knowledge/prompt-policy.ts +30 -0
- package/src/runtime/knowledge/retrieve.ts +303 -0
- package/src/runtime/knowledge/schema.ts +100 -0
- package/src/runtime/knowledge/tool.ts +64 -0
- package/src/runtime/knowledge/vector.ts +258 -0
- package/src/runtime/prompt-context.ts +141 -0
- package/src/runtime/runtime-contract.ts +86 -8
- package/src/runtime/skills.ts +95 -0
- package/src/runtime/spec.ts +152 -0
- package/src/runtime/sync.ts +144 -0
- package/src/runtime/targets/cloudflare/build.ts +1430 -185
- package/src/runtime/targets/container/server.ts +1 -1
- package/src/runtime/targets/vps/deploy.ts +26 -9
- package/src/runtime/tool-runner.ts +9 -1
- package/src/runtime/tools.ts +128 -2
- package/src/runtime/traces.ts +41 -0
- package/src/runtime/transcription.ts +483 -0
- package/src/storage/sqlite.ts +149 -3
- package/src/templates/blank.ts +76 -17
- package/src/templates/dentista.ts +1011 -0
- package/src/templates/index.ts +2 -0
- package/src/templates/skills/agentkit-build-agent/SKILL.md +52 -0
- package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +21 -0
- package/src/templates/skills/agentkit-build-agent/templates/sales-qualifier.instructions.md +17 -0
- package/src/templates/skills/agentkit-build-agent/templates/support-agent.instructions.md +16 -0
- package/src/templates/skills/agentkit-capsule/SKILL.md +70 -0
- package/src/templates/skills/agentkit-capsule/references/docs-router.md +15 -0
- package/src/templates/skills/agentkit-channels/SKILL.md +104 -0
- package/src/templates/skills/agentkit-channels/references/channel-buffering.md +65 -0
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +66 -0
- package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
- package/src/templates/skills/agentkit-channels/references/slack.md +56 -0
- package/src/templates/skills/agentkit-channels/references/telegram.md +72 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +77 -0
- package/src/templates/skills/agentkit-database/SKILL.md +45 -0
- package/src/templates/skills/agentkit-database/templates/appointments.schema.sql +15 -0
- package/src/templates/skills/agentkit-database/templates/leads.schema.sql +17 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +50 -0
- package/src/templates/skills/agentkit-evals/SKILL.md +109 -0
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +29 -0
- package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +18 -0
- package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +18 -0
- package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +27 -0
- package/src/templates/skills/agentkit-improve/SKILL.md +86 -0
- package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
- package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
- package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
- package/src/templates/skills/agentkit-integrations/SKILL.md +76 -0
- package/src/templates/skills/agentkit-knowledge/SKILL.md +43 -0
- package/src/templates/skills/agentkit-knowledge/templates/faq.md +14 -0
- package/src/templates/skills/agentkit-knowledge/templates/policies.md +14 -0
- package/src/templates/skills/agentkit-knowledge/templates/prices.csv +3 -0
- package/src/templates/skills/agentkit-prompts/SKILL.md +47 -0
- package/src/templates/skills/agentkit-prompts/templates/knowledge-grounded-faq.instructions.md +11 -0
- package/src/templates/skills/agentkit-provider/SKILL.md +60 -0
- package/src/templates/skills/agentkit-security/SKILL.md +56 -0
- package/src/templates/skills/agentkit-tools/SKILL.md +37 -0
- package/src/templates/skills/agentkit-tools/examples/database-write.tool.md +35 -0
- package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +37 -0
- package/src/templates/skills/agentkit-tools/examples/lookup-order.tool.md +46 -0
- package/src/templates/skills/agentkit-troubleshooting/SKILL.md +76 -0
- package/src/templates/support.ts +77 -18
- package/docs/guides/channels-production-handoff.md +0 -99
- package/docs/portable-deploy-release-checklist.md +0 -41
- package/src/runtime/targets/cloudflare/deploy.ts +0 -5475
package/src/runtime/evals.ts
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
|
-
import { readdir } from "node:fs/promises";
|
|
1
|
+
import { mkdir, readdir, writeFile } from "node:fs/promises";
|
|
2
2
|
import type { Dirent } from "node:fs";
|
|
3
|
-
import { join, relative } from "node:path";
|
|
3
|
+
import { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
4
4
|
import { pathToFileURL } from "node:url";
|
|
5
5
|
|
|
6
6
|
import type { AgentRunResult } from "./chat";
|
|
7
7
|
import { runAgentMessageFromCwd } from "./chat";
|
|
8
8
|
import { findAgentCapsuleRoot, loadAgentCapsule } from "./config";
|
|
9
|
-
import { AgentKitError } from "./errors";
|
|
9
|
+
import { AgentKitError, isAgentKitError } from "./errors";
|
|
10
10
|
import { openCapsuleStore, type StoredToolCall } from "../storage/sqlite";
|
|
11
|
+
import { getConversationTraceFromCwd } from "./traces";
|
|
11
12
|
|
|
12
13
|
export type EvalRunSummary = {
|
|
13
14
|
root: string;
|
|
@@ -24,30 +25,76 @@ export type EvalResult = {
|
|
|
24
25
|
output: string;
|
|
25
26
|
};
|
|
26
27
|
|
|
27
|
-
type EvalCase = {
|
|
28
|
+
export type EvalCase = {
|
|
28
29
|
name?: string;
|
|
29
30
|
input?: string;
|
|
31
|
+
now?: string | Date;
|
|
32
|
+
turns?: EvalTurn[];
|
|
30
33
|
expect?: EvalExpect;
|
|
31
34
|
};
|
|
32
35
|
|
|
33
|
-
type
|
|
36
|
+
export type EvalTurn =
|
|
37
|
+
| string
|
|
38
|
+
| {
|
|
39
|
+
input: string;
|
|
40
|
+
expect?: EvalExpect;
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
export type EvalExpect = EvalResponseExpectation & {
|
|
44
|
+
response?: EvalResponseExpectation;
|
|
45
|
+
tools?: EvalToolsExpectation;
|
|
46
|
+
tool_call?: ToolExpectation | ToolExpectation[];
|
|
47
|
+
toolCall?: ToolExpectation | ToolExpectation[];
|
|
48
|
+
tool_calls?: ToolExpectation | ToolExpectation[];
|
|
49
|
+
toolCalls?: ToolExpectation | ToolExpectation[];
|
|
50
|
+
tool_call_count?: number;
|
|
51
|
+
toolCallCount?: number;
|
|
52
|
+
tool_call_order?: string[];
|
|
53
|
+
toolCallOrder?: string[];
|
|
54
|
+
persisted_tool_call?: ToolExpectation | ToolExpectation[];
|
|
55
|
+
persistedToolCall?: ToolExpectation | ToolExpectation[];
|
|
56
|
+
persisted_tool_calls?: ToolExpectation | ToolExpectation[];
|
|
57
|
+
persistedToolCalls?: ToolExpectation | ToolExpectation[];
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
export type EvalResponseExpectation = {
|
|
34
61
|
contains?: string | string[];
|
|
62
|
+
contains_all?: string | string[];
|
|
63
|
+
containsAll?: string | string[];
|
|
64
|
+
contains_any?: string | string[];
|
|
65
|
+
containsAny?: string | string[];
|
|
66
|
+
case_insensitive_contains?: string | string[];
|
|
67
|
+
caseInsensitiveContains?: string | string[];
|
|
35
68
|
not_contains?: string | string[];
|
|
36
69
|
notContains?: string | string[];
|
|
37
70
|
regex?: string | string[];
|
|
38
71
|
matches_regex?: string | string[];
|
|
39
72
|
matchesRegex?: string | string[];
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
73
|
+
not_regex?: string | string[];
|
|
74
|
+
notRegex?: string | string[];
|
|
75
|
+
max_length?: number;
|
|
76
|
+
maxLength?: number;
|
|
77
|
+
};
|
|
78
|
+
|
|
79
|
+
export type EvalToolsExpectation =
|
|
80
|
+
| ToolExpectation
|
|
81
|
+
| ToolExpectation[]
|
|
82
|
+
| EvalToolsContainerExpectation;
|
|
83
|
+
|
|
84
|
+
export type EvalToolsContainerExpectation = {
|
|
85
|
+
persisted?: ToolExpectation | ToolExpectation[];
|
|
44
86
|
persisted_tool_call?: ToolExpectation | ToolExpectation[];
|
|
45
87
|
persistedToolCall?: ToolExpectation | ToolExpectation[];
|
|
46
88
|
persisted_tool_calls?: ToolExpectation | ToolExpectation[];
|
|
47
89
|
persistedToolCalls?: ToolExpectation | ToolExpectation[];
|
|
90
|
+
called?: string | string[];
|
|
91
|
+
called_once?: string | string[];
|
|
92
|
+
calledOnce?: string | string[];
|
|
93
|
+
count?: number;
|
|
94
|
+
order?: string[];
|
|
48
95
|
};
|
|
49
96
|
|
|
50
|
-
type ToolExpectation =
|
|
97
|
+
export type ToolExpectation =
|
|
51
98
|
| string
|
|
52
99
|
| {
|
|
53
100
|
name?: string;
|
|
@@ -55,16 +102,29 @@ type ToolExpectation =
|
|
|
55
102
|
output?: unknown;
|
|
56
103
|
status?: "running" | "completed" | "failed";
|
|
57
104
|
visibility?: "user" | "internal";
|
|
105
|
+
rendered?: unknown;
|
|
58
106
|
};
|
|
59
107
|
|
|
60
108
|
type ToolCallSnapshot = {
|
|
61
109
|
name?: unknown;
|
|
62
110
|
input?: unknown;
|
|
63
111
|
output?: unknown;
|
|
112
|
+
rendered?: unknown;
|
|
64
113
|
status?: unknown;
|
|
65
114
|
visibility?: unknown;
|
|
66
115
|
};
|
|
67
116
|
|
|
117
|
+
export type EvalFromConversationResult = {
|
|
118
|
+
root: string;
|
|
119
|
+
file: string;
|
|
120
|
+
conversationId: string;
|
|
121
|
+
turns: number;
|
|
122
|
+
};
|
|
123
|
+
|
|
124
|
+
export function defineEval<const T extends EvalCase>(evalCase: T): T {
|
|
125
|
+
return evalCase;
|
|
126
|
+
}
|
|
127
|
+
|
|
68
128
|
export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
|
|
69
129
|
const root = await findAgentCapsuleRoot(cwd);
|
|
70
130
|
const evalFiles = await findEvalFiles(join(root, "evals"));
|
|
@@ -76,29 +136,29 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
|
|
|
76
136
|
const results: EvalResult[] = [];
|
|
77
137
|
|
|
78
138
|
for (const file of evalFiles) {
|
|
79
|
-
const
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
139
|
+
const evalFile = relative(root, file);
|
|
140
|
+
let name = evalFile;
|
|
141
|
+
let failures: string[];
|
|
142
|
+
let output: string;
|
|
143
|
+
|
|
144
|
+
try {
|
|
145
|
+
const evalCase = await loadEvalCase(file);
|
|
146
|
+
const run = await runEvalCase(root, evalFile, evalCase);
|
|
147
|
+
|
|
148
|
+
name = evalCase.name ?? evalFile;
|
|
149
|
+
failures = run.failures;
|
|
150
|
+
output = run.output;
|
|
151
|
+
} catch (error) {
|
|
152
|
+
failures = [formatEvalFailure(error)];
|
|
153
|
+
output = "";
|
|
84
154
|
}
|
|
85
155
|
|
|
86
|
-
const run = await runAgentMessageFromCwd(root, {
|
|
87
|
-
message: input,
|
|
88
|
-
runtime: {
|
|
89
|
-
environment: "eval",
|
|
90
|
-
invocation: "eval",
|
|
91
|
-
},
|
|
92
|
-
});
|
|
93
|
-
const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
|
|
94
|
-
const failures = evaluateExpectations(evalCase.expect ?? {}, run, persistedToolCalls);
|
|
95
|
-
|
|
96
156
|
results.push({
|
|
97
|
-
name
|
|
98
|
-
file:
|
|
157
|
+
name,
|
|
158
|
+
file: evalFile,
|
|
99
159
|
passed: failures.length === 0,
|
|
100
160
|
failures,
|
|
101
|
-
output
|
|
161
|
+
output,
|
|
102
162
|
});
|
|
103
163
|
}
|
|
104
164
|
|
|
@@ -112,6 +172,171 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
|
|
|
112
172
|
};
|
|
113
173
|
}
|
|
114
174
|
|
|
175
|
+
export async function writeEvalFromConversation(
|
|
176
|
+
cwd: string,
|
|
177
|
+
input: {
|
|
178
|
+
conversationId: string;
|
|
179
|
+
out?: string;
|
|
180
|
+
name?: string;
|
|
181
|
+
force?: boolean;
|
|
182
|
+
},
|
|
183
|
+
): Promise<EvalFromConversationResult> {
|
|
184
|
+
const root = await findAgentCapsuleRoot(cwd);
|
|
185
|
+
const trace = await getConversationTraceFromCwd(root, input.conversationId);
|
|
186
|
+
const turns = replayTurnsFromMessages(trace.messages);
|
|
187
|
+
const safeTitle = redactEvalText(trace.title ?? trace.id);
|
|
188
|
+
|
|
189
|
+
if (turns.length === 0) {
|
|
190
|
+
throw new AgentKitError(
|
|
191
|
+
"validation_error",
|
|
192
|
+
`Conversation "${input.conversationId}" does not contain user/assistant turns that can become an eval.`,
|
|
193
|
+
);
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
const file = resolveEvalOutputPath(root, input.out, safeTitle);
|
|
197
|
+
const source = `import { defineEval } from "@andreprado/agentkit";
|
|
198
|
+
|
|
199
|
+
export default defineEval({
|
|
200
|
+
name: ${JSON.stringify(input.name ? redactEvalText(input.name) : `replay ${safeTitle}`)},
|
|
201
|
+
turns: ${formatEvalTurns(turns)},
|
|
202
|
+
});
|
|
203
|
+
`;
|
|
204
|
+
|
|
205
|
+
await mkdir(dirname(file), { recursive: true });
|
|
206
|
+
|
|
207
|
+
try {
|
|
208
|
+
await writeFile(file, source, { flag: input.force ? "w" : "wx" });
|
|
209
|
+
} catch (error) {
|
|
210
|
+
if (isNodeError(error) && error.code === "EEXIST") {
|
|
211
|
+
throw new AgentKitError(
|
|
212
|
+
"validation_error",
|
|
213
|
+
`${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
|
|
214
|
+
);
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
throw error;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
return {
|
|
221
|
+
root,
|
|
222
|
+
file,
|
|
223
|
+
conversationId: trace.id,
|
|
224
|
+
turns: turns.length,
|
|
225
|
+
};
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
async function runEvalCase(
|
|
229
|
+
root: string,
|
|
230
|
+
file: string,
|
|
231
|
+
evalCase: EvalCase,
|
|
232
|
+
): Promise<{ failures: string[]; output: string }> {
|
|
233
|
+
const turns = normalizeEvalTurns(evalCase);
|
|
234
|
+
|
|
235
|
+
if (turns.length === 0) {
|
|
236
|
+
throw new AgentKitError(
|
|
237
|
+
"validation_error",
|
|
238
|
+
`${file} must export either a non-empty input string or a non-empty turns array.`,
|
|
239
|
+
);
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
const conversationId = `eval_${crypto.randomUUID()}`;
|
|
243
|
+
const failures: string[] = [];
|
|
244
|
+
let output = "";
|
|
245
|
+
const now = resolveEvalNow(file, evalCase.now);
|
|
246
|
+
|
|
247
|
+
for (const [index, turn] of turns.entries()) {
|
|
248
|
+
try {
|
|
249
|
+
const run = await runAgentMessageFromCwd(root, {
|
|
250
|
+
message: turn.input,
|
|
251
|
+
conversationId,
|
|
252
|
+
now,
|
|
253
|
+
runtime: {
|
|
254
|
+
environment: "eval",
|
|
255
|
+
invocation: "eval",
|
|
256
|
+
},
|
|
257
|
+
});
|
|
258
|
+
const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
|
|
259
|
+
const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
|
|
260
|
+
|
|
261
|
+
for (const failure of turnFailures) {
|
|
262
|
+
failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
output = run.message.content;
|
|
266
|
+
} catch (error) {
|
|
267
|
+
failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
|
|
268
|
+
break;
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
return { failures, output };
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
function resolveEvalNow(file: string, value: EvalCase["now"]): Date | undefined {
|
|
276
|
+
if (value === undefined) {
|
|
277
|
+
return undefined;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
if (typeof value === "string" && !hasExplicitIsoOffset(value)) {
|
|
281
|
+
throw new AgentKitError(
|
|
282
|
+
"validation_error",
|
|
283
|
+
`${file} now must be an ISO timestamp with an explicit timezone offset, such as "2026-02-04T02:30:00.000Z" or "2026-02-04T02:30:00-05:00".`,
|
|
284
|
+
);
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
const date = value instanceof Date ? new Date(value.getTime()) : new Date(value);
|
|
288
|
+
|
|
289
|
+
if (Number.isNaN(date.getTime())) {
|
|
290
|
+
throw new AgentKitError("validation_error", `${file} now must be a valid ISO timestamp or Date when provided.`);
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
return date;
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
function hasExplicitIsoOffset(value: string): boolean {
|
|
297
|
+
return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.test(value.trim());
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
function normalizeEvalTurns(evalCase: EvalCase): Array<{ input: string; expect?: EvalExpect }> {
|
|
301
|
+
if (evalCase.turns !== undefined) {
|
|
302
|
+
if (!Array.isArray(evalCase.turns)) {
|
|
303
|
+
throw new AgentKitError("validation_error", "eval turns must be an array when provided.");
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
return evalCase.turns.map((turn, index) => normalizeEvalTurn(turn, index));
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if (typeof evalCase.input === "string" && evalCase.input.trim().length > 0) {
|
|
310
|
+
return [
|
|
311
|
+
{
|
|
312
|
+
input: evalCase.input,
|
|
313
|
+
expect: evalCase.expect,
|
|
314
|
+
},
|
|
315
|
+
];
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
return [];
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
function normalizeEvalTurn(turn: EvalTurn, index: number): { input: string; expect?: EvalExpect } {
|
|
322
|
+
if (typeof turn === "string") {
|
|
323
|
+
if (turn.trim().length === 0) {
|
|
324
|
+
throw new AgentKitError("validation_error", `eval turns[${index}] must be non-empty.`);
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
return { input: turn };
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
if (!isRecord(turn) || typeof turn.input !== "string" || turn.input.trim().length === 0) {
|
|
331
|
+
throw new AgentKitError("validation_error", `eval turns[${index}].input must be a non-empty string.`);
|
|
332
|
+
}
|
|
333
|
+
|
|
334
|
+
return {
|
|
335
|
+
input: turn.input,
|
|
336
|
+
...(turn.expect ? { expect: turn.expect } : {}),
|
|
337
|
+
};
|
|
338
|
+
}
|
|
339
|
+
|
|
115
340
|
async function findEvalFiles(directory: string): Promise<string[]> {
|
|
116
341
|
let entries: Dirent[];
|
|
117
342
|
|
|
@@ -172,12 +397,89 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
|
|
|
172
397
|
const failures: string[] = [];
|
|
173
398
|
const content = run.message.content;
|
|
174
399
|
|
|
175
|
-
for (const
|
|
400
|
+
for (const responseExpectation of responseExpectations(expect)) {
|
|
401
|
+
failures.push(...evaluateResponseExpectation(responseExpectation, content));
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
const toolAssertions = collectToolAssertions(expect);
|
|
405
|
+
|
|
406
|
+
for (const toolExpectation of toolAssertions.persisted) {
|
|
407
|
+
const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
|
|
408
|
+
|
|
409
|
+
if (!match) {
|
|
410
|
+
failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
for (const expectedName of toolAssertions.called) {
|
|
415
|
+
if (!persistedToolCalls.some((toolCall) => toolCall.name === expectedName)) {
|
|
416
|
+
failures.push(`expected persisted tool call named ${JSON.stringify(expectedName)}`);
|
|
417
|
+
}
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
for (const expectedName of toolAssertions.calledOnce) {
|
|
421
|
+
const count = persistedToolCalls.filter((toolCall) => toolCall.name === expectedName).length;
|
|
422
|
+
|
|
423
|
+
if (count !== 1) {
|
|
424
|
+
failures.push(`expected persisted tool call ${JSON.stringify(expectedName)} exactly once, got ${count}`);
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
for (const expectedCount of toolAssertions.counts) {
|
|
429
|
+
if (!Number.isInteger(expectedCount) || expectedCount < 0) {
|
|
430
|
+
failures.push(`expected tool call count to be a non-negative integer, got ${JSON.stringify(expectedCount)}`);
|
|
431
|
+
continue;
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
if (persistedToolCalls.length !== expectedCount) {
|
|
435
|
+
failures.push(`expected ${expectedCount} persisted tool call(s), got ${persistedToolCalls.length}`);
|
|
436
|
+
}
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
for (const expectedOrder of toolAssertions.orders) {
|
|
440
|
+
if (!namesContainSubsequence(persistedToolNames(persistedToolCalls), expectedOrder)) {
|
|
441
|
+
failures.push(
|
|
442
|
+
`expected persisted tool call order ${JSON.stringify(expectedOrder)}, got ${JSON.stringify(
|
|
443
|
+
persistedToolNames(persistedToolCalls),
|
|
444
|
+
)}`,
|
|
445
|
+
);
|
|
446
|
+
}
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
return failures;
|
|
450
|
+
}
|
|
451
|
+
|
|
452
|
+
function responseExpectations(expect: EvalExpect): EvalResponseExpectation[] {
|
|
453
|
+
return [expect, expect.response].filter((value): value is EvalResponseExpectation => value !== undefined);
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
function evaluateResponseExpectation(expect: EvalResponseExpectation, content: string): string[] {
|
|
457
|
+
const failures: string[] = [];
|
|
458
|
+
|
|
459
|
+
for (const expected of [
|
|
460
|
+
...list(expect.contains),
|
|
461
|
+
...list(expect.contains_all),
|
|
462
|
+
...list(expect.containsAll),
|
|
463
|
+
]) {
|
|
176
464
|
if (!content.includes(expected)) {
|
|
177
465
|
failures.push(`expected output to contain ${JSON.stringify(expected)}`);
|
|
178
466
|
}
|
|
179
467
|
}
|
|
180
468
|
|
|
469
|
+
const containsAny = [...list(expect.contains_any), ...list(expect.containsAny)];
|
|
470
|
+
|
|
471
|
+
if (containsAny.length > 0 && !containsAny.some((expected) => content.includes(expected))) {
|
|
472
|
+
failures.push(`expected output to contain any of ${JSON.stringify(containsAny)}`);
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
const lowerContent = content.toLowerCase();
|
|
476
|
+
|
|
477
|
+
for (const expected of [...list(expect.case_insensitive_contains), ...list(expect.caseInsensitiveContains)]) {
|
|
478
|
+
if (!lowerContent.includes(expected.toLowerCase())) {
|
|
479
|
+
failures.push(`expected output to contain ${JSON.stringify(expected)} case-insensitively`);
|
|
480
|
+
}
|
|
481
|
+
}
|
|
482
|
+
|
|
181
483
|
for (const forbidden of [...list(expect.not_contains), ...list(expect.notContains)]) {
|
|
182
484
|
if (content.includes(forbidden)) {
|
|
183
485
|
failures.push(`expected output not to contain ${JSON.stringify(forbidden)}`);
|
|
@@ -190,28 +492,140 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
|
|
|
190
492
|
}
|
|
191
493
|
}
|
|
192
494
|
|
|
193
|
-
const
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
...toolList(expect.persisted_tool_call),
|
|
199
|
-
...toolList(expect.persistedToolCall),
|
|
200
|
-
...toolList(expect.persisted_tool_calls),
|
|
201
|
-
...toolList(expect.persistedToolCalls),
|
|
202
|
-
];
|
|
495
|
+
for (const pattern of [...list(expect.not_regex), ...list(expect.notRegex)]) {
|
|
496
|
+
if (new RegExp(pattern).test(content)) {
|
|
497
|
+
failures.push(`expected output not to match /${pattern}/`);
|
|
498
|
+
}
|
|
499
|
+
}
|
|
203
500
|
|
|
204
|
-
for (const
|
|
205
|
-
|
|
501
|
+
for (const expectedMaxLength of [expect.max_length, expect.maxLength]) {
|
|
502
|
+
if (expectedMaxLength === undefined) {
|
|
503
|
+
continue;
|
|
504
|
+
}
|
|
206
505
|
|
|
207
|
-
if (!
|
|
208
|
-
failures.push(`expected
|
|
506
|
+
if (!Number.isInteger(expectedMaxLength) || expectedMaxLength < 0) {
|
|
507
|
+
failures.push(`expected response maxLength to be a non-negative integer, got ${JSON.stringify(expectedMaxLength)}`);
|
|
508
|
+
continue;
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
if (content.length > expectedMaxLength) {
|
|
512
|
+
failures.push(`expected output length to be <= ${expectedMaxLength}, got ${content.length}`);
|
|
209
513
|
}
|
|
210
514
|
}
|
|
211
515
|
|
|
212
516
|
return failures;
|
|
213
517
|
}
|
|
214
518
|
|
|
519
|
+
type ToolAssertions = {
|
|
520
|
+
persisted: ToolExpectation[];
|
|
521
|
+
called: string[];
|
|
522
|
+
calledOnce: string[];
|
|
523
|
+
counts: number[];
|
|
524
|
+
orders: string[][];
|
|
525
|
+
};
|
|
526
|
+
|
|
527
|
+
function collectToolAssertions(expect: EvalExpect): ToolAssertions {
|
|
528
|
+
const assertions: ToolAssertions = {
|
|
529
|
+
persisted: [
|
|
530
|
+
...toolList(expect.tool_call),
|
|
531
|
+
...toolList(expect.toolCall),
|
|
532
|
+
...toolList(expect.tool_calls),
|
|
533
|
+
...toolList(expect.toolCalls),
|
|
534
|
+
...toolList(expect.persisted_tool_call),
|
|
535
|
+
...toolList(expect.persistedToolCall),
|
|
536
|
+
...toolList(expect.persisted_tool_calls),
|
|
537
|
+
...toolList(expect.persistedToolCalls),
|
|
538
|
+
],
|
|
539
|
+
called: [],
|
|
540
|
+
calledOnce: [],
|
|
541
|
+
counts: [],
|
|
542
|
+
orders: [],
|
|
543
|
+
};
|
|
544
|
+
|
|
545
|
+
if (expect.tool_call_count !== undefined) {
|
|
546
|
+
assertions.counts.push(expect.tool_call_count);
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
if (expect.toolCallCount !== undefined) {
|
|
550
|
+
assertions.counts.push(expect.toolCallCount);
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
if (expect.tool_call_order !== undefined) {
|
|
554
|
+
assertions.orders.push(expect.tool_call_order);
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
if (expect.toolCallOrder !== undefined) {
|
|
558
|
+
assertions.orders.push(expect.toolCallOrder);
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
appendNestedToolAssertions(assertions, expect.tools);
|
|
562
|
+
|
|
563
|
+
return assertions;
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
function appendNestedToolAssertions(assertions: ToolAssertions, tools: EvalToolsExpectation | undefined): void {
|
|
567
|
+
if (tools === undefined) {
|
|
568
|
+
return;
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
if (!isToolsContainer(tools)) {
|
|
572
|
+
assertions.persisted.push(...toolList(tools));
|
|
573
|
+
return;
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
assertions.persisted.push(
|
|
577
|
+
...toolList(tools.persisted),
|
|
578
|
+
...toolList(tools.persisted_tool_call),
|
|
579
|
+
...toolList(tools.persistedToolCall),
|
|
580
|
+
...toolList(tools.persisted_tool_calls),
|
|
581
|
+
...toolList(tools.persistedToolCalls),
|
|
582
|
+
);
|
|
583
|
+
assertions.called.push(...list(tools.called));
|
|
584
|
+
assertions.calledOnce.push(...list(tools.called_once), ...list(tools.calledOnce));
|
|
585
|
+
|
|
586
|
+
if (tools.count !== undefined) {
|
|
587
|
+
assertions.counts.push(tools.count);
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
if (tools.order !== undefined) {
|
|
591
|
+
assertions.orders.push(tools.order);
|
|
592
|
+
}
|
|
593
|
+
}
|
|
594
|
+
|
|
595
|
+
function isToolsContainer(value: EvalToolsExpectation): value is EvalToolsContainerExpectation {
|
|
596
|
+
if (!isRecord(value)) {
|
|
597
|
+
return false;
|
|
598
|
+
}
|
|
599
|
+
|
|
600
|
+
if (
|
|
601
|
+
"name" in value ||
|
|
602
|
+
"input" in value ||
|
|
603
|
+
"output" in value ||
|
|
604
|
+
"rendered" in value ||
|
|
605
|
+
"status" in value ||
|
|
606
|
+
"visibility" in value
|
|
607
|
+
) {
|
|
608
|
+
return false;
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
if (Object.keys(value).length === 0) {
|
|
612
|
+
return true;
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
return (
|
|
616
|
+
"persisted" in value ||
|
|
617
|
+
"persisted_tool_call" in value ||
|
|
618
|
+
"persistedToolCall" in value ||
|
|
619
|
+
"persisted_tool_calls" in value ||
|
|
620
|
+
"persistedToolCalls" in value ||
|
|
621
|
+
"called" in value ||
|
|
622
|
+
"called_once" in value ||
|
|
623
|
+
"calledOnce" in value ||
|
|
624
|
+
"count" in value ||
|
|
625
|
+
"order" in value
|
|
626
|
+
);
|
|
627
|
+
}
|
|
628
|
+
|
|
215
629
|
function list(value: string | string[] | undefined): string[] {
|
|
216
630
|
if (value === undefined) {
|
|
217
631
|
return [];
|
|
@@ -220,6 +634,30 @@ function list(value: string | string[] | undefined): string[] {
|
|
|
220
634
|
return Array.isArray(value) ? value : [value];
|
|
221
635
|
}
|
|
222
636
|
|
|
637
|
+
function persistedToolNames(toolCalls: ToolCallSnapshot[]): string[] {
|
|
638
|
+
return toolCalls.map((toolCall) => String(toolCall.name ?? ""));
|
|
639
|
+
}
|
|
640
|
+
|
|
641
|
+
function namesContainSubsequence(actual: string[], expected: string[]): boolean {
|
|
642
|
+
if (expected.length === 0) {
|
|
643
|
+
return true;
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
let actualIndex = 0;
|
|
647
|
+
|
|
648
|
+
for (const expectedName of expected) {
|
|
649
|
+
actualIndex = actual.findIndex((actualName, index) => index >= actualIndex && actualName === expectedName);
|
|
650
|
+
|
|
651
|
+
if (actualIndex === -1) {
|
|
652
|
+
return false;
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
actualIndex += 1;
|
|
656
|
+
}
|
|
657
|
+
|
|
658
|
+
return true;
|
|
659
|
+
}
|
|
660
|
+
|
|
223
661
|
function toolList(value: ToolExpectation | ToolExpectation[] | undefined): ToolExpectation[] {
|
|
224
662
|
if (value === undefined) {
|
|
225
663
|
return [];
|
|
@@ -253,6 +691,10 @@ function matchesToolExpectation(toolCall: ToolCallSnapshot, expectation: ToolExp
|
|
|
253
691
|
return false;
|
|
254
692
|
}
|
|
255
693
|
|
|
694
|
+
if ("rendered" in expectation && !jsonContains(toolCall.rendered, expectation.rendered)) {
|
|
695
|
+
return false;
|
|
696
|
+
}
|
|
697
|
+
|
|
256
698
|
return true;
|
|
257
699
|
}
|
|
258
700
|
|
|
@@ -269,11 +711,115 @@ function toolCallSnapshotFromStored(toolCall: StoredToolCall): ToolCallSnapshot
|
|
|
269
711
|
name: toolCall.toolName,
|
|
270
712
|
input: toolCall.input,
|
|
271
713
|
output: toolCall.output,
|
|
714
|
+
rendered: toolCall.rendered,
|
|
272
715
|
status: toolCall.status,
|
|
273
716
|
visibility: toolCall.visibility,
|
|
274
717
|
};
|
|
275
718
|
}
|
|
276
719
|
|
|
720
|
+
function replayTurnsFromMessages(
|
|
721
|
+
messages: Array<{ role: string; content: string }>,
|
|
722
|
+
): Array<{ input: string; expect?: EvalExpect }> {
|
|
723
|
+
const turns: Array<{ input: string; expect?: EvalExpect }> = [];
|
|
724
|
+
|
|
725
|
+
for (let index = 0; index < messages.length; index += 1) {
|
|
726
|
+
const message = messages[index];
|
|
727
|
+
|
|
728
|
+
if (message.role !== "user") {
|
|
729
|
+
continue;
|
|
730
|
+
}
|
|
731
|
+
|
|
732
|
+
const nextAssistant = messages.slice(index + 1).find((candidate) => candidate.role === "assistant");
|
|
733
|
+
const input = redactEvalText(message.content);
|
|
734
|
+
const contains = nextAssistant ? redactEvalText(nextAssistant.content) : undefined;
|
|
735
|
+
const redactionPresent = input.includes("[redacted_") || Boolean(contains?.includes("[redacted_"));
|
|
736
|
+
|
|
737
|
+
turns.push({
|
|
738
|
+
input,
|
|
739
|
+
...(nextAssistant
|
|
740
|
+
? {
|
|
741
|
+
expect: {
|
|
742
|
+
response: redactionPresent
|
|
743
|
+
? {
|
|
744
|
+
notRegex: ["API_KEY|Bearer\\s+|(?:sk|agk|ak|dpat)[_-]|[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,}"],
|
|
745
|
+
}
|
|
746
|
+
: {
|
|
747
|
+
contains: contains ?? "",
|
|
748
|
+
},
|
|
749
|
+
},
|
|
750
|
+
}
|
|
751
|
+
: {}),
|
|
752
|
+
});
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
return turns;
|
|
756
|
+
}
|
|
757
|
+
|
|
758
|
+
function redactEvalText(value: string): string {
|
|
759
|
+
return value
|
|
760
|
+
.replace(/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi, "[redacted_email]")
|
|
761
|
+
.replace(
|
|
762
|
+
/\b[A-Z0-9_]*(?:API[_-]?KEY|TOKEN|SECRET|PASSWORD|PRIVATE[_-]?KEY|CLIENT[_-]?SECRET)\s*=\s*[^\s,;]+/gi,
|
|
763
|
+
(match) => `${match.slice(0, match.indexOf("=")).trim()}=[redacted_secret]`,
|
|
764
|
+
)
|
|
765
|
+
.replace(/\b(?:sk|agk|ak|dpat)[_-][A-Za-z0-9_-]{8,}\b/g, "[redacted_token]")
|
|
766
|
+
.replace(/\bBearer\s+[A-Za-z0-9._~+/-]+=*/gi, "Bearer [redacted_token]")
|
|
767
|
+
.replace(/\+?\d[\d\s().-]{7,}\d/g, "[redacted_phone]");
|
|
768
|
+
}
|
|
769
|
+
|
|
770
|
+
function formatEvalTurns(turns: Array<{ input: string; expect?: EvalExpect }>): string {
|
|
771
|
+
return JSON.stringify(turns, null, 4)
|
|
772
|
+
.split("\n")
|
|
773
|
+
.map((line, index) => (index === 0 ? line : ` ${line}`))
|
|
774
|
+
.join("\n");
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
function slugify(value: string): string {
|
|
778
|
+
const slug = value
|
|
779
|
+
.toLowerCase()
|
|
780
|
+
.replace(/[^a-z0-9]+/g, "-")
|
|
781
|
+
.replace(/^-+|-+$/g, "")
|
|
782
|
+
.slice(0, 48);
|
|
783
|
+
|
|
784
|
+
return slug || "conversation";
|
|
785
|
+
}
|
|
786
|
+
|
|
787
|
+
function resolveEvalOutputPath(root: string, out: string | undefined, title: string): string {
|
|
788
|
+
const outputPath = out ?? join("evals", `replay-${slugify(title)}.eval.ts`);
|
|
789
|
+
|
|
790
|
+
if (isAbsolute(outputPath)) {
|
|
791
|
+
throw new AgentKitError("validation_error", "eval --out must be a relative path inside the Agent Capsule.");
|
|
792
|
+
}
|
|
793
|
+
|
|
794
|
+
const file = resolve(root, outputPath);
|
|
795
|
+
const relativePath = relative(root, file);
|
|
796
|
+
|
|
797
|
+
if (relativePath === "" || relativePath.startsWith("..") || isAbsolute(relativePath)) {
|
|
798
|
+
throw new AgentKitError("validation_error", "eval --out must stay inside the Agent Capsule.");
|
|
799
|
+
}
|
|
800
|
+
|
|
801
|
+
if (!/\.(eval|spec)\.[cm]?[tj]s$/.test(file)) {
|
|
802
|
+
throw new AgentKitError(
|
|
803
|
+
"validation_error",
|
|
804
|
+
"eval --out must end with .eval.ts, .eval.js, .spec.ts, or another AgentKit eval/spec extension.",
|
|
805
|
+
);
|
|
806
|
+
}
|
|
807
|
+
|
|
808
|
+
return file;
|
|
809
|
+
}
|
|
810
|
+
|
|
811
|
+
function formatEvalFailure(error: unknown): string {
|
|
812
|
+
if (isAgentKitError(error)) {
|
|
813
|
+
return `${error.code}: ${error.message}`;
|
|
814
|
+
}
|
|
815
|
+
|
|
816
|
+
if (error instanceof Error) {
|
|
817
|
+
return `runtime_error: ${error.message}`;
|
|
818
|
+
}
|
|
819
|
+
|
|
820
|
+
return `runtime_error: ${String(error)}`;
|
|
821
|
+
}
|
|
822
|
+
|
|
277
823
|
function jsonContains(actual: unknown, expected: unknown): boolean {
|
|
278
824
|
if (Array.isArray(expected)) {
|
|
279
825
|
return jsonEqual(actual, expected);
|