@andreprado/agentkit 0.1.0-alpha.9 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -1
- package/docs/guides/add-channel.md +251 -7
- package/docs/guides/add-knowledge.md +10 -0
- package/docs/guides/add-managed-composio.md +165 -0
- package/docs/guides/add-tool.md +10 -3
- package/docs/guides/channel-security.md +162 -32
- package/docs/guides/connect-discord.md +178 -0
- package/docs/guides/connect-slack.md +126 -0
- package/docs/guides/connect-telegram.md +61 -1
- package/docs/guides/connect-whatsapp-evolution.md +121 -0
- package/docs/guides/connect-whatsapp-uazapi.md +139 -0
- package/docs/guides/connect-whatsapp-zapster.md +119 -16
- package/docs/guides/create-agent.md +31 -4
- package/docs/guides/debug-channel.md +159 -0
- package/docs/guides/improve-from-production.md +151 -0
- package/docs/guides/prepare-deploy.md +32 -14
- package/docs/guides/replay-production-traces.md +72 -0
- package/docs/guides/run-evals.md +95 -25
- package/docs/guides/security-rules.md +9 -5
- package/docs/guides/send-feedback.md +135 -0
- package/docs/guides/use-jev.md +67 -0
- package/docs/guides/use-provider.md +70 -3
- package/docs/llms-full.txt +295 -25
- package/docs/llms.txt +54 -7
- package/package.json +3 -7
- package/src/cli/args.ts +23 -2
- package/src/cli/cloud-client.ts +121 -9
- package/src/cli/commands/channels.ts +856 -36
- package/src/cli/commands/feedback.ts +438 -0
- package/src/cli/commands/provider.ts +47 -0
- package/src/cli/commands/transcribe.ts +171 -0
- package/src/cli/deploy-chat-ui.ts +232 -18
- package/src/cli/deploy-readiness.ts +227 -14
- package/src/cli/help.ts +67 -9
- package/src/cli/index.ts +740 -35
- package/src/cli/new-command.ts +41 -0
- package/src/cloud/client.ts +4 -3
- package/src/cloud/contracts.ts +1 -1
- package/src/create-project.ts +18 -35
- package/src/index.ts +565 -11
- package/src/providers/codex-auth.ts +111 -0
- package/src/providers/pi.ts +88 -19
- package/src/providers/test.ts +36 -0
- package/src/providers/types.ts +8 -0
- package/src/runtime/channel-test-harness.ts +21 -1
- package/src/runtime/channels/discord.ts +904 -0
- package/src/runtime/channels/generic-webhook.ts +682 -0
- package/src/runtime/channels/net-guard.ts +480 -0
- package/src/runtime/channels/provider-fetch.ts +54 -0
- package/src/runtime/channels/slack.ts +652 -0
- package/src/runtime/channels/telegram.ts +379 -15
- package/src/runtime/channels/whatsapp-evolution.ts +1330 -0
- package/src/runtime/channels/whatsapp-meta.ts +9 -0
- package/src/runtime/channels/whatsapp-uazapi.ts +1192 -0
- package/src/runtime/channels/whatsapp-zapster.ts +702 -40
- package/src/runtime/channels.ts +83 -3
- package/src/runtime/chat.ts +70 -44
- package/src/runtime/config.ts +512 -20
- package/src/runtime/core/manifest.ts +75 -5
- package/src/runtime/core/targets.ts +5 -5
- package/src/runtime/deploy-readiness.ts +34 -4
- package/src/runtime/dev-server.ts +639 -39
- package/src/runtime/env.ts +8 -3
- package/src/runtime/evals.ts +445 -74
- package/src/runtime/improve.ts +868 -0
- package/src/runtime/inspect.ts +173 -4
- package/src/runtime/integrations/composio.ts +425 -0
- package/src/runtime/knowledge/embeddings.ts +45 -7
- package/src/runtime/knowledge/ingest.ts +69 -6
- package/src/runtime/knowledge/retrieve.ts +25 -5
- package/src/runtime/knowledge/schema.ts +45 -1
- package/src/runtime/knowledge/vector.ts +30 -30
- package/src/runtime/prompt-context.ts +141 -0
- package/src/runtime/runtime-contract.ts +71 -7
- package/src/runtime/skills.ts +95 -0
- package/src/runtime/targets/cloudflare/build.ts +1010 -208
- package/src/runtime/targets/container/server.ts +1 -1
- package/src/runtime/targets/vps/deploy.ts +26 -9
- package/src/runtime/tool-runner.ts +9 -1
- package/src/runtime/tools.ts +26 -2
- package/src/runtime/transcription.ts +483 -0
- package/src/storage/sqlite.ts +7 -2
- package/src/templates/blank.ts +37 -9
- package/src/templates/dentista.ts +40 -14
- package/src/templates/skills/agentkit-build-agent/SKILL.md +34 -5
- package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
- package/src/templates/skills/agentkit-capsule/SKILL.md +32 -3
- package/src/templates/skills/agentkit-capsule/references/docs-router.md +2 -2
- package/src/templates/skills/agentkit-channels/SKILL.md +66 -1
- package/src/templates/skills/agentkit-channels/references/channel-buffering.md +8 -1
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +28 -3
- package/src/templates/skills/agentkit-channels/references/discord.md +93 -0
- package/src/templates/skills/agentkit-channels/references/slack.md +56 -0
- package/src/templates/skills/agentkit-channels/references/telegram.md +34 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-evolution.md +57 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-uazapi.md +54 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +42 -8
- package/src/templates/skills/agentkit-database/SKILL.md +11 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +9 -1
- package/src/templates/skills/agentkit-evals/SKILL.md +77 -13
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
- package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
- package/src/templates/skills/agentkit-improve/SKILL.md +96 -0
- package/src/templates/skills/agentkit-improve/references/replay-side-effects.md +18 -0
- package/src/templates/skills/agentkit-improve/references/trace-packets.md +22 -0
- package/src/templates/skills/agentkit-improve/templates/regression.eval.md +18 -0
- package/src/templates/skills/agentkit-integrations/SKILL.md +98 -0
- package/src/templates/skills/agentkit-knowledge/SKILL.md +4 -1
- package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
- package/src/templates/skills/agentkit-provider/SKILL.md +29 -4
- package/src/templates/skills/agentkit-security/SKILL.md +5 -2
- package/src/templates/skills/agentkit-tools/SKILL.md +8 -1
- package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +8 -8
- package/src/templates/skills/agentkit-tools/examples/jev-service-fit.tool.md +110 -0
- package/src/templates/skills/agentkit-troubleshooting/SKILL.md +25 -1
- package/src/templates/support.ts +42 -12
- package/docs/guides/agentkit-skills-architecture.md +0 -471
- package/docs/guides/channels-implementation-map.md +0 -243
- package/docs/guides/channels-production-handoff.md +0 -101
- package/docs/portable-deploy-release-checklist.md +0 -41
package/src/runtime/evals.ts
CHANGED
|
@@ -1,12 +1,13 @@
|
|
|
1
|
-
import { mkdir, readdir,
|
|
1
|
+
import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises";
|
|
2
2
|
import type { Dirent } from "node:fs";
|
|
3
|
-
import {
|
|
3
|
+
import { tmpdir } from "node:os";
|
|
4
|
+
import { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
4
5
|
import { pathToFileURL } from "node:url";
|
|
5
6
|
|
|
6
7
|
import type { AgentRunResult } from "./chat";
|
|
7
8
|
import { runAgentMessageFromCwd } from "./chat";
|
|
8
9
|
import { findAgentCapsuleRoot, loadAgentCapsule } from "./config";
|
|
9
|
-
import { AgentKitError } from "./errors";
|
|
10
|
+
import { AgentKitError, isAgentKitError } from "./errors";
|
|
10
11
|
import { openCapsuleStore, type StoredToolCall } from "../storage/sqlite";
|
|
11
12
|
import { getConversationTraceFromCwd } from "./traces";
|
|
12
13
|
|
|
@@ -25,38 +26,76 @@ export type EvalResult = {
|
|
|
25
26
|
output: string;
|
|
26
27
|
};
|
|
27
28
|
|
|
28
|
-
type EvalCase = {
|
|
29
|
+
export type EvalCase = {
|
|
29
30
|
name?: string;
|
|
30
31
|
input?: string;
|
|
32
|
+
now?: string | Date;
|
|
31
33
|
turns?: EvalTurn[];
|
|
32
34
|
expect?: EvalExpect;
|
|
33
35
|
};
|
|
34
36
|
|
|
35
|
-
type EvalTurn =
|
|
37
|
+
export type EvalTurn =
|
|
36
38
|
| string
|
|
37
39
|
| {
|
|
38
40
|
input: string;
|
|
39
41
|
expect?: EvalExpect;
|
|
40
42
|
};
|
|
41
43
|
|
|
42
|
-
type EvalExpect = {
|
|
44
|
+
export type EvalExpect = EvalResponseExpectation & {
|
|
45
|
+
response?: EvalResponseExpectation;
|
|
46
|
+
tools?: EvalToolsExpectation;
|
|
47
|
+
tool_call?: ToolExpectation | ToolExpectation[];
|
|
48
|
+
toolCall?: ToolExpectation | ToolExpectation[];
|
|
49
|
+
tool_calls?: ToolExpectation | ToolExpectation[];
|
|
50
|
+
toolCalls?: ToolExpectation | ToolExpectation[];
|
|
51
|
+
tool_call_count?: number;
|
|
52
|
+
toolCallCount?: number;
|
|
53
|
+
tool_call_order?: string[];
|
|
54
|
+
toolCallOrder?: string[];
|
|
55
|
+
persisted_tool_call?: ToolExpectation | ToolExpectation[];
|
|
56
|
+
persistedToolCall?: ToolExpectation | ToolExpectation[];
|
|
57
|
+
persisted_tool_calls?: ToolExpectation | ToolExpectation[];
|
|
58
|
+
persistedToolCalls?: ToolExpectation | ToolExpectation[];
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
export type EvalResponseExpectation = {
|
|
43
62
|
contains?: string | string[];
|
|
63
|
+
contains_all?: string | string[];
|
|
64
|
+
containsAll?: string | string[];
|
|
65
|
+
contains_any?: string | string[];
|
|
66
|
+
containsAny?: string | string[];
|
|
67
|
+
case_insensitive_contains?: string | string[];
|
|
68
|
+
caseInsensitiveContains?: string | string[];
|
|
44
69
|
not_contains?: string | string[];
|
|
45
70
|
notContains?: string | string[];
|
|
46
71
|
regex?: string | string[];
|
|
47
72
|
matches_regex?: string | string[];
|
|
48
73
|
matchesRegex?: string | string[];
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
74
|
+
not_regex?: string | string[];
|
|
75
|
+
notRegex?: string | string[];
|
|
76
|
+
max_length?: number;
|
|
77
|
+
maxLength?: number;
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
export type EvalToolsExpectation =
|
|
81
|
+
| ToolExpectation
|
|
82
|
+
| ToolExpectation[]
|
|
83
|
+
| EvalToolsContainerExpectation;
|
|
84
|
+
|
|
85
|
+
export type EvalToolsContainerExpectation = {
|
|
86
|
+
persisted?: ToolExpectation | ToolExpectation[];
|
|
53
87
|
persisted_tool_call?: ToolExpectation | ToolExpectation[];
|
|
54
88
|
persistedToolCall?: ToolExpectation | ToolExpectation[];
|
|
55
89
|
persisted_tool_calls?: ToolExpectation | ToolExpectation[];
|
|
56
90
|
persistedToolCalls?: ToolExpectation | ToolExpectation[];
|
|
91
|
+
called?: string | string[];
|
|
92
|
+
called_once?: string | string[];
|
|
93
|
+
calledOnce?: string | string[];
|
|
94
|
+
count?: number;
|
|
95
|
+
order?: string[];
|
|
57
96
|
};
|
|
58
97
|
|
|
59
|
-
type ToolExpectation =
|
|
98
|
+
export type ToolExpectation =
|
|
60
99
|
| string
|
|
61
100
|
| {
|
|
62
101
|
name?: string;
|
|
@@ -83,6 +122,10 @@ export type EvalFromConversationResult = {
|
|
|
83
122
|
turns: number;
|
|
84
123
|
};
|
|
85
124
|
|
|
125
|
+
export function defineEval<const T extends EvalCase>(evalCase: T): T {
|
|
126
|
+
return evalCase;
|
|
127
|
+
}
|
|
128
|
+
|
|
86
129
|
export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
|
|
87
130
|
const root = await findAgentCapsuleRoot(cwd);
|
|
88
131
|
const evalFiles = await findEvalFiles(join(root, "evals"));
|
|
@@ -94,15 +137,29 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
|
|
|
94
137
|
const results: EvalResult[] = [];
|
|
95
138
|
|
|
96
139
|
for (const file of evalFiles) {
|
|
97
|
-
const
|
|
98
|
-
|
|
140
|
+
const evalFile = relative(root, file);
|
|
141
|
+
let name = evalFile;
|
|
142
|
+
let failures: string[];
|
|
143
|
+
let output: string;
|
|
144
|
+
|
|
145
|
+
try {
|
|
146
|
+
const evalCase = await loadEvalCase(file);
|
|
147
|
+
const run = await runEvalCase(root, evalFile, evalCase);
|
|
148
|
+
|
|
149
|
+
name = evalCase.name ?? evalFile;
|
|
150
|
+
failures = run.failures;
|
|
151
|
+
output = run.output;
|
|
152
|
+
} catch (error) {
|
|
153
|
+
failures = [formatEvalFailure(error)];
|
|
154
|
+
output = "";
|
|
155
|
+
}
|
|
99
156
|
|
|
100
157
|
results.push({
|
|
101
|
-
name
|
|
102
|
-
file:
|
|
103
|
-
passed:
|
|
104
|
-
failures
|
|
105
|
-
output
|
|
158
|
+
name,
|
|
159
|
+
file: evalFile,
|
|
160
|
+
passed: failures.length === 0,
|
|
161
|
+
failures,
|
|
162
|
+
output,
|
|
106
163
|
});
|
|
107
164
|
}
|
|
108
165
|
|
|
@@ -128,6 +185,7 @@ export async function writeEvalFromConversation(
|
|
|
128
185
|
const root = await findAgentCapsuleRoot(cwd);
|
|
129
186
|
const trace = await getConversationTraceFromCwd(root, input.conversationId);
|
|
130
187
|
const turns = replayTurnsFromMessages(trace.messages);
|
|
188
|
+
const safeTitle = redactEvalText(trace.title ?? trace.id);
|
|
131
189
|
|
|
132
190
|
if (turns.length === 0) {
|
|
133
191
|
throw new AgentKitError(
|
|
@@ -136,24 +194,29 @@ export async function writeEvalFromConversation(
|
|
|
136
194
|
);
|
|
137
195
|
}
|
|
138
196
|
|
|
139
|
-
const file =
|
|
197
|
+
const file = resolveEvalOutputPath(root, input.out, safeTitle);
|
|
198
|
+
const source = `import { defineEval } from "@andreprado/agentkit";
|
|
140
199
|
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
}
|
|
200
|
+
export default defineEval({
|
|
201
|
+
name: ${JSON.stringify(input.name ? redactEvalText(input.name) : `replay ${safeTitle}`)},
|
|
202
|
+
turns: ${formatEvalTurns(turns)},
|
|
203
|
+
});
|
|
204
|
+
`;
|
|
147
205
|
|
|
148
206
|
await mkdir(dirname(file), { recursive: true });
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
207
|
+
|
|
208
|
+
try {
|
|
209
|
+
await writeFile(file, source, { flag: input.force ? "w" : "wx" });
|
|
210
|
+
} catch (error) {
|
|
211
|
+
if (isNodeError(error) && error.code === "EEXIST") {
|
|
212
|
+
throw new AgentKitError(
|
|
213
|
+
"validation_error",
|
|
214
|
+
`${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
|
|
215
|
+
);
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
throw error;
|
|
219
|
+
}
|
|
157
220
|
|
|
158
221
|
return {
|
|
159
222
|
root,
|
|
@@ -180,29 +243,75 @@ async function runEvalCase(
|
|
|
180
243
|
const conversationId = `eval_${crypto.randomUUID()}`;
|
|
181
244
|
const failures: string[] = [];
|
|
182
245
|
let output = "";
|
|
246
|
+
const now = resolveEvalNow(file, evalCase.now);
|
|
247
|
+
const evalStore = await createEvalStore();
|
|
183
248
|
|
|
184
|
-
|
|
185
|
-
const
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
249
|
+
try {
|
|
250
|
+
for (const [index, turn] of turns.entries()) {
|
|
251
|
+
try {
|
|
252
|
+
const run = await runAgentMessageFromCwd(root, {
|
|
253
|
+
message: turn.input,
|
|
254
|
+
conversationId,
|
|
255
|
+
now,
|
|
256
|
+
localStoragePath: evalStore.databasePath,
|
|
257
|
+
runtime: {
|
|
258
|
+
environment: "eval",
|
|
259
|
+
invocation: "eval",
|
|
260
|
+
},
|
|
261
|
+
});
|
|
262
|
+
const persistedToolCalls = await loadPersistedToolCalls(root, run.runId, evalStore.databasePath);
|
|
263
|
+
const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
|
|
264
|
+
|
|
265
|
+
for (const failure of turnFailures) {
|
|
266
|
+
failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
output = run.message.content;
|
|
270
|
+
} catch (error) {
|
|
271
|
+
failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
|
|
272
|
+
break;
|
|
273
|
+
}
|
|
198
274
|
}
|
|
199
|
-
|
|
200
|
-
|
|
275
|
+
} finally {
|
|
276
|
+
await rm(evalStore.directory, { recursive: true, force: true });
|
|
201
277
|
}
|
|
202
278
|
|
|
203
279
|
return { failures, output };
|
|
204
280
|
}
|
|
205
281
|
|
|
282
|
+
async function createEvalStore(): Promise<{ directory: string; databasePath: string }> {
|
|
283
|
+
const directory = await mkdtemp(join(tmpdir(), "agentkit-eval-"));
|
|
284
|
+
return {
|
|
285
|
+
directory,
|
|
286
|
+
databasePath: join(directory, "agentkit.db"),
|
|
287
|
+
};
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
function resolveEvalNow(file: string, value: EvalCase["now"]): Date | undefined {
|
|
291
|
+
if (value === undefined) {
|
|
292
|
+
return undefined;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
if (typeof value === "string" && !hasExplicitIsoOffset(value)) {
|
|
296
|
+
throw new AgentKitError(
|
|
297
|
+
"validation_error",
|
|
298
|
+
`${file} now must be an ISO timestamp with an explicit timezone offset, such as "2026-02-04T02:30:00.000Z" or "2026-02-04T02:30:00-05:00".`,
|
|
299
|
+
);
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
const date = value instanceof Date ? new Date(value.getTime()) : new Date(value);
|
|
303
|
+
|
|
304
|
+
if (Number.isNaN(date.getTime())) {
|
|
305
|
+
throw new AgentKitError("validation_error", `${file} now must be a valid ISO timestamp or Date when provided.`);
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
return date;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
function hasExplicitIsoOffset(value: string): boolean {
|
|
312
|
+
return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.test(value.trim());
|
|
313
|
+
}
|
|
314
|
+
|
|
206
315
|
function normalizeEvalTurns(evalCase: EvalCase): Array<{ input: string; expect?: EvalExpect }> {
|
|
207
316
|
if (evalCase.turns !== undefined) {
|
|
208
317
|
if (!Array.isArray(evalCase.turns)) {
|
|
@@ -288,9 +397,9 @@ async function loadEvalCase(file: string): Promise<EvalCase> {
|
|
|
288
397
|
return module.default;
|
|
289
398
|
}
|
|
290
399
|
|
|
291
|
-
async function loadPersistedToolCalls(root: string, runId: string): Promise<ToolCallSnapshot[]> {
|
|
400
|
+
async function loadPersistedToolCalls(root: string, runId: string, localStoragePath?: string): Promise<ToolCallSnapshot[]> {
|
|
292
401
|
const capsule = await loadAgentCapsule(root);
|
|
293
|
-
const store = await openCapsuleStore(capsule);
|
|
402
|
+
const store = await openCapsuleStore(localStoragePath ? { ...capsule, storagePath: localStoragePath } : capsule);
|
|
294
403
|
|
|
295
404
|
try {
|
|
296
405
|
return store.listToolCallsForRun(runId).map(toolCallSnapshotFromStored);
|
|
@@ -303,12 +412,89 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
|
|
|
303
412
|
const failures: string[] = [];
|
|
304
413
|
const content = run.message.content;
|
|
305
414
|
|
|
306
|
-
for (const
|
|
415
|
+
for (const responseExpectation of responseExpectations(expect)) {
|
|
416
|
+
failures.push(...evaluateResponseExpectation(responseExpectation, content));
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
const toolAssertions = collectToolAssertions(expect);
|
|
420
|
+
|
|
421
|
+
for (const toolExpectation of toolAssertions.persisted) {
|
|
422
|
+
const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
|
|
423
|
+
|
|
424
|
+
if (!match) {
|
|
425
|
+
failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
|
|
426
|
+
}
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
for (const expectedName of toolAssertions.called) {
|
|
430
|
+
if (!persistedToolCalls.some((toolCall) => toolCall.name === expectedName)) {
|
|
431
|
+
failures.push(`expected persisted tool call named ${JSON.stringify(expectedName)}`);
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
for (const expectedName of toolAssertions.calledOnce) {
|
|
436
|
+
const count = persistedToolCalls.filter((toolCall) => toolCall.name === expectedName).length;
|
|
437
|
+
|
|
438
|
+
if (count !== 1) {
|
|
439
|
+
failures.push(`expected persisted tool call ${JSON.stringify(expectedName)} exactly once, got ${count}`);
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
for (const expectedCount of toolAssertions.counts) {
|
|
444
|
+
if (!Number.isInteger(expectedCount) || expectedCount < 0) {
|
|
445
|
+
failures.push(`expected tool call count to be a non-negative integer, got ${JSON.stringify(expectedCount)}`);
|
|
446
|
+
continue;
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
if (persistedToolCalls.length !== expectedCount) {
|
|
450
|
+
failures.push(`expected ${expectedCount} persisted tool call(s), got ${persistedToolCalls.length}`);
|
|
451
|
+
}
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
for (const expectedOrder of toolAssertions.orders) {
|
|
455
|
+
if (!namesContainSubsequence(persistedToolNames(persistedToolCalls), expectedOrder)) {
|
|
456
|
+
failures.push(
|
|
457
|
+
`expected persisted tool call order ${JSON.stringify(expectedOrder)}, got ${JSON.stringify(
|
|
458
|
+
persistedToolNames(persistedToolCalls),
|
|
459
|
+
)}`,
|
|
460
|
+
);
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
|
|
464
|
+
return failures;
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
function responseExpectations(expect: EvalExpect): EvalResponseExpectation[] {
|
|
468
|
+
return [expect, expect.response].filter((value): value is EvalResponseExpectation => value !== undefined);
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
function evaluateResponseExpectation(expect: EvalResponseExpectation, content: string): string[] {
|
|
472
|
+
const failures: string[] = [];
|
|
473
|
+
|
|
474
|
+
for (const expected of [
|
|
475
|
+
...list(expect.contains),
|
|
476
|
+
...list(expect.contains_all),
|
|
477
|
+
...list(expect.containsAll),
|
|
478
|
+
]) {
|
|
307
479
|
if (!content.includes(expected)) {
|
|
308
480
|
failures.push(`expected output to contain ${JSON.stringify(expected)}`);
|
|
309
481
|
}
|
|
310
482
|
}
|
|
311
483
|
|
|
484
|
+
const containsAny = [...list(expect.contains_any), ...list(expect.containsAny)];
|
|
485
|
+
|
|
486
|
+
if (containsAny.length > 0 && !containsAny.some((expected) => content.includes(expected))) {
|
|
487
|
+
failures.push(`expected output to contain any of ${JSON.stringify(containsAny)}`);
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
const lowerContent = content.toLowerCase();
|
|
491
|
+
|
|
492
|
+
for (const expected of [...list(expect.case_insensitive_contains), ...list(expect.caseInsensitiveContains)]) {
|
|
493
|
+
if (!lowerContent.includes(expected.toLowerCase())) {
|
|
494
|
+
failures.push(`expected output to contain ${JSON.stringify(expected)} case-insensitively`);
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
|
|
312
498
|
for (const forbidden of [...list(expect.not_contains), ...list(expect.notContains)]) {
|
|
313
499
|
if (content.includes(forbidden)) {
|
|
314
500
|
failures.push(`expected output not to contain ${JSON.stringify(forbidden)}`);
|
|
@@ -321,28 +507,140 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
|
|
|
321
507
|
}
|
|
322
508
|
}
|
|
323
509
|
|
|
324
|
-
const
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
...toolList(expect.persisted_tool_call),
|
|
330
|
-
...toolList(expect.persistedToolCall),
|
|
331
|
-
...toolList(expect.persisted_tool_calls),
|
|
332
|
-
...toolList(expect.persistedToolCalls),
|
|
333
|
-
];
|
|
510
|
+
for (const pattern of [...list(expect.not_regex), ...list(expect.notRegex)]) {
|
|
511
|
+
if (new RegExp(pattern).test(content)) {
|
|
512
|
+
failures.push(`expected output not to match /${pattern}/`);
|
|
513
|
+
}
|
|
514
|
+
}
|
|
334
515
|
|
|
335
|
-
for (const
|
|
336
|
-
|
|
516
|
+
for (const expectedMaxLength of [expect.max_length, expect.maxLength]) {
|
|
517
|
+
if (expectedMaxLength === undefined) {
|
|
518
|
+
continue;
|
|
519
|
+
}
|
|
337
520
|
|
|
338
|
-
if (!
|
|
339
|
-
failures.push(`expected
|
|
521
|
+
if (!Number.isInteger(expectedMaxLength) || expectedMaxLength < 0) {
|
|
522
|
+
failures.push(`expected response maxLength to be a non-negative integer, got ${JSON.stringify(expectedMaxLength)}`);
|
|
523
|
+
continue;
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
if (content.length > expectedMaxLength) {
|
|
527
|
+
failures.push(`expected output length to be <= ${expectedMaxLength}, got ${content.length}`);
|
|
340
528
|
}
|
|
341
529
|
}
|
|
342
530
|
|
|
343
531
|
return failures;
|
|
344
532
|
}
|
|
345
533
|
|
|
534
|
+
type ToolAssertions = {
|
|
535
|
+
persisted: ToolExpectation[];
|
|
536
|
+
called: string[];
|
|
537
|
+
calledOnce: string[];
|
|
538
|
+
counts: number[];
|
|
539
|
+
orders: string[][];
|
|
540
|
+
};
|
|
541
|
+
|
|
542
|
+
function collectToolAssertions(expect: EvalExpect): ToolAssertions {
|
|
543
|
+
const assertions: ToolAssertions = {
|
|
544
|
+
persisted: [
|
|
545
|
+
...toolList(expect.tool_call),
|
|
546
|
+
...toolList(expect.toolCall),
|
|
547
|
+
...toolList(expect.tool_calls),
|
|
548
|
+
...toolList(expect.toolCalls),
|
|
549
|
+
...toolList(expect.persisted_tool_call),
|
|
550
|
+
...toolList(expect.persistedToolCall),
|
|
551
|
+
...toolList(expect.persisted_tool_calls),
|
|
552
|
+
...toolList(expect.persistedToolCalls),
|
|
553
|
+
],
|
|
554
|
+
called: [],
|
|
555
|
+
calledOnce: [],
|
|
556
|
+
counts: [],
|
|
557
|
+
orders: [],
|
|
558
|
+
};
|
|
559
|
+
|
|
560
|
+
if (expect.tool_call_count !== undefined) {
|
|
561
|
+
assertions.counts.push(expect.tool_call_count);
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
if (expect.toolCallCount !== undefined) {
|
|
565
|
+
assertions.counts.push(expect.toolCallCount);
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
if (expect.tool_call_order !== undefined) {
|
|
569
|
+
assertions.orders.push(expect.tool_call_order);
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
if (expect.toolCallOrder !== undefined) {
|
|
573
|
+
assertions.orders.push(expect.toolCallOrder);
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
appendNestedToolAssertions(assertions, expect.tools);
|
|
577
|
+
|
|
578
|
+
return assertions;
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
function appendNestedToolAssertions(assertions: ToolAssertions, tools: EvalToolsExpectation | undefined): void {
|
|
582
|
+
if (tools === undefined) {
|
|
583
|
+
return;
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
if (!isToolsContainer(tools)) {
|
|
587
|
+
assertions.persisted.push(...toolList(tools));
|
|
588
|
+
return;
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
assertions.persisted.push(
|
|
592
|
+
...toolList(tools.persisted),
|
|
593
|
+
...toolList(tools.persisted_tool_call),
|
|
594
|
+
...toolList(tools.persistedToolCall),
|
|
595
|
+
...toolList(tools.persisted_tool_calls),
|
|
596
|
+
...toolList(tools.persistedToolCalls),
|
|
597
|
+
);
|
|
598
|
+
assertions.called.push(...list(tools.called));
|
|
599
|
+
assertions.calledOnce.push(...list(tools.called_once), ...list(tools.calledOnce));
|
|
600
|
+
|
|
601
|
+
if (tools.count !== undefined) {
|
|
602
|
+
assertions.counts.push(tools.count);
|
|
603
|
+
}
|
|
604
|
+
|
|
605
|
+
if (tools.order !== undefined) {
|
|
606
|
+
assertions.orders.push(tools.order);
|
|
607
|
+
}
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
function isToolsContainer(value: EvalToolsExpectation): value is EvalToolsContainerExpectation {
|
|
611
|
+
if (!isRecord(value)) {
|
|
612
|
+
return false;
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
if (
|
|
616
|
+
"name" in value ||
|
|
617
|
+
"input" in value ||
|
|
618
|
+
"output" in value ||
|
|
619
|
+
"rendered" in value ||
|
|
620
|
+
"status" in value ||
|
|
621
|
+
"visibility" in value
|
|
622
|
+
) {
|
|
623
|
+
return false;
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
if (Object.keys(value).length === 0) {
|
|
627
|
+
return true;
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
return (
|
|
631
|
+
"persisted" in value ||
|
|
632
|
+
"persisted_tool_call" in value ||
|
|
633
|
+
"persistedToolCall" in value ||
|
|
634
|
+
"persisted_tool_calls" in value ||
|
|
635
|
+
"persistedToolCalls" in value ||
|
|
636
|
+
"called" in value ||
|
|
637
|
+
"called_once" in value ||
|
|
638
|
+
"calledOnce" in value ||
|
|
639
|
+
"count" in value ||
|
|
640
|
+
"order" in value
|
|
641
|
+
);
|
|
642
|
+
}
|
|
643
|
+
|
|
346
644
|
function list(value: string | string[] | undefined): string[] {
|
|
347
645
|
if (value === undefined) {
|
|
348
646
|
return [];
|
|
@@ -351,6 +649,30 @@ function list(value: string | string[] | undefined): string[] {
|
|
|
351
649
|
return Array.isArray(value) ? value : [value];
|
|
352
650
|
}
|
|
353
651
|
|
|
652
|
+
function persistedToolNames(toolCalls: ToolCallSnapshot[]): string[] {
|
|
653
|
+
return toolCalls.map((toolCall) => String(toolCall.name ?? ""));
|
|
654
|
+
}
|
|
655
|
+
|
|
656
|
+
function namesContainSubsequence(actual: string[], expected: string[]): boolean {
|
|
657
|
+
if (expected.length === 0) {
|
|
658
|
+
return true;
|
|
659
|
+
}
|
|
660
|
+
|
|
661
|
+
let actualIndex = 0;
|
|
662
|
+
|
|
663
|
+
for (const expectedName of expected) {
|
|
664
|
+
actualIndex = actual.findIndex((actualName, index) => index >= actualIndex && actualName === expectedName);
|
|
665
|
+
|
|
666
|
+
if (actualIndex === -1) {
|
|
667
|
+
return false;
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
actualIndex += 1;
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
return true;
|
|
674
|
+
}
|
|
675
|
+
|
|
354
676
|
function toolList(value: ToolExpectation | ToolExpectation[] | undefined): ToolExpectation[] {
|
|
355
677
|
if (value === undefined) {
|
|
356
678
|
return [];
|
|
@@ -423,12 +745,22 @@ function replayTurnsFromMessages(
|
|
|
423
745
|
}
|
|
424
746
|
|
|
425
747
|
const nextAssistant = messages.slice(index + 1).find((candidate) => candidate.role === "assistant");
|
|
748
|
+
const input = redactEvalText(message.content);
|
|
749
|
+
const contains = nextAssistant ? redactEvalText(nextAssistant.content) : undefined;
|
|
750
|
+
const redactionPresent = input.includes("[redacted_") || Boolean(contains?.includes("[redacted_"));
|
|
751
|
+
|
|
426
752
|
turns.push({
|
|
427
|
-
input
|
|
753
|
+
input,
|
|
428
754
|
...(nextAssistant
|
|
429
755
|
? {
|
|
430
756
|
expect: {
|
|
431
|
-
|
|
757
|
+
response: redactionPresent
|
|
758
|
+
? {
|
|
759
|
+
notRegex: ["API_KEY|Bearer\\s+|(?:sk|agk|ak|dpat)[_-]|[A-Z0-9._%+-]+@[A-Z0-9.-]+\\.[A-Z]{2,}"],
|
|
760
|
+
}
|
|
761
|
+
: {
|
|
762
|
+
contains: contains ?? "",
|
|
763
|
+
},
|
|
432
764
|
},
|
|
433
765
|
}
|
|
434
766
|
: {}),
|
|
@@ -438,6 +770,18 @@ function replayTurnsFromMessages(
|
|
|
438
770
|
return turns;
|
|
439
771
|
}
|
|
440
772
|
|
|
773
|
+
function redactEvalText(value: string): string {
|
|
774
|
+
return value
|
|
775
|
+
.replace(/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi, "[redacted_email]")
|
|
776
|
+
.replace(
|
|
777
|
+
/\b[A-Z0-9_]*(?:API[_-]?KEY|TOKEN|SECRET|PASSWORD|PRIVATE[_-]?KEY|CLIENT[_-]?SECRET)\s*=\s*[^\s,;]+/gi,
|
|
778
|
+
(match) => `${match.slice(0, match.indexOf("=")).trim()}=[redacted_secret]`,
|
|
779
|
+
)
|
|
780
|
+
.replace(/\b(?:sk|agk|ak|dpat)[_-][A-Za-z0-9_-]{8,}\b/g, "[redacted_token]")
|
|
781
|
+
.replace(/\bBearer\s+[A-Za-z0-9._~+/-]+=*/gi, "Bearer [redacted_token]")
|
|
782
|
+
.replace(/\+?\d[\d\s().-]{7,}\d/g, "[redacted_phone]");
|
|
783
|
+
}
|
|
784
|
+
|
|
441
785
|
function formatEvalTurns(turns: Array<{ input: string; expect?: EvalExpect }>): string {
|
|
442
786
|
return JSON.stringify(turns, null, 4)
|
|
443
787
|
.split("\n")
|
|
@@ -455,13 +799,40 @@ function slugify(value: string): string {
|
|
|
455
799
|
return slug || "conversation";
|
|
456
800
|
}
|
|
457
801
|
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
802
|
+
function resolveEvalOutputPath(root: string, out: string | undefined, title: string): string {
|
|
803
|
+
const outputPath = out ?? join("evals", `replay-${slugify(title)}.eval.ts`);
|
|
804
|
+
|
|
805
|
+
if (isAbsolute(outputPath)) {
|
|
806
|
+
throw new AgentKitError("validation_error", "eval --out must be a relative path inside the Agent Capsule.");
|
|
807
|
+
}
|
|
808
|
+
|
|
809
|
+
const file = resolve(root, outputPath);
|
|
810
|
+
const relativePath = relative(root, file);
|
|
811
|
+
|
|
812
|
+
if (relativePath === "" || relativePath.startsWith("..") || isAbsolute(relativePath)) {
|
|
813
|
+
throw new AgentKitError("validation_error", "eval --out must stay inside the Agent Capsule.");
|
|
464
814
|
}
|
|
815
|
+
|
|
816
|
+
if (!/\.(eval|spec)\.[cm]?[tj]s$/.test(file)) {
|
|
817
|
+
throw new AgentKitError(
|
|
818
|
+
"validation_error",
|
|
819
|
+
"eval --out must end with .eval.ts, .eval.js, .spec.ts, or another AgentKit eval/spec extension.",
|
|
820
|
+
);
|
|
821
|
+
}
|
|
822
|
+
|
|
823
|
+
return file;
|
|
824
|
+
}
|
|
825
|
+
|
|
826
|
+
function formatEvalFailure(error: unknown): string {
|
|
827
|
+
if (isAgentKitError(error)) {
|
|
828
|
+
return `${error.code}: ${error.message}`;
|
|
829
|
+
}
|
|
830
|
+
|
|
831
|
+
if (error instanceof Error) {
|
|
832
|
+
return `runtime_error: ${error.message}`;
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
return `runtime_error: ${String(error)}`;
|
|
465
836
|
}
|
|
466
837
|
|
|
467
838
|
function jsonContains(actual: unknown, expected: unknown): boolean {
|