@andreprado/agentkit 0.1.0-alpha.14 → 0.1.0-alpha.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/docs/guides/run-evals.md +73 -25
- package/docs/llms-full.txt +24 -9
- package/docs/llms.txt +2 -0
- package/package.json +1 -1
- package/src/cli/cloud-client.ts +30 -10
- package/src/cli/deploy-readiness.ts +32 -11
- package/src/cli/index.ts +20 -6
- package/src/cloud/client.ts +4 -3
- package/src/cloud/contracts.ts +1 -1
- package/src/create-project.ts +1 -1
- package/src/index.ts +24 -1
- package/src/providers/pi.ts +14 -1
- package/src/providers/test.ts +36 -0
- package/src/runtime/chat.ts +59 -42
- package/src/runtime/config.ts +11 -0
- package/src/runtime/core/manifest.ts +3 -3
- package/src/runtime/deploy-readiness.ts +3 -3
- package/src/runtime/dev-server.ts +171 -13
- package/src/runtime/env.ts +8 -3
- package/src/runtime/evals.ts +404 -69
- package/src/runtime/inspect.ts +12 -0
- package/src/runtime/prompt-context.ts +141 -0
- package/src/runtime/runtime-contract.ts +17 -7
- package/src/runtime/targets/cloudflare/build.ts +23 -3
- package/src/runtime/targets/container/server.ts +1 -1
- package/src/runtime/targets/vps/deploy.ts +25 -8
- package/src/runtime/tool-runner.ts +7 -0
- package/src/runtime/tools.ts +8 -2
- package/src/templates/blank.ts +8 -3
- package/src/templates/dentista.ts +18 -10
- package/src/templates/skills/agentkit-build-agent/SKILL.md +6 -5
- package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +2 -1
- package/src/templates/skills/agentkit-capsule/SKILL.md +1 -1
- package/src/templates/skills/agentkit-evals/SKILL.md +53 -13
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +13 -6
- package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +8 -4
- package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +16 -7
- package/src/templates/skills/agentkit-prompts/SKILL.md +3 -1
- package/src/templates/skills/agentkit-tools/SKILL.md +2 -1
- package/src/templates/support.ts +8 -3
package/src/runtime/evals.ts
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
import { mkdir, readdir,
|
|
1
|
+
import { mkdir, readdir, writeFile } from "node:fs/promises";
|
|
2
2
|
import type { Dirent } from "node:fs";
|
|
3
|
-
import { dirname, join, relative, resolve } from "node:path";
|
|
3
|
+
import { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
4
4
|
import { pathToFileURL } from "node:url";
|
|
5
5
|
|
|
6
6
|
import type { AgentRunResult } from "./chat";
|
|
7
7
|
import { runAgentMessageFromCwd } from "./chat";
|
|
8
8
|
import { findAgentCapsuleRoot, loadAgentCapsule } from "./config";
|
|
9
|
-
import { AgentKitError } from "./errors";
|
|
9
|
+
import { AgentKitError, isAgentKitError } from "./errors";
|
|
10
10
|
import { openCapsuleStore, type StoredToolCall } from "../storage/sqlite";
|
|
11
11
|
import { getConversationTraceFromCwd } from "./traces";
|
|
12
12
|
|
|
@@ -25,38 +25,76 @@ export type EvalResult = {
|
|
|
25
25
|
output: string;
|
|
26
26
|
};
|
|
27
27
|
|
|
28
|
-
type EvalCase = {
|
|
28
|
+
export type EvalCase = {
|
|
29
29
|
name?: string;
|
|
30
30
|
input?: string;
|
|
31
|
+
now?: string | Date;
|
|
31
32
|
turns?: EvalTurn[];
|
|
32
33
|
expect?: EvalExpect;
|
|
33
34
|
};
|
|
34
35
|
|
|
35
|
-
type EvalTurn =
|
|
36
|
+
export type EvalTurn =
|
|
36
37
|
| string
|
|
37
38
|
| {
|
|
38
39
|
input: string;
|
|
39
40
|
expect?: EvalExpect;
|
|
40
41
|
};
|
|
41
42
|
|
|
42
|
-
type EvalExpect = {
|
|
43
|
+
export type EvalExpect = EvalResponseExpectation & {
|
|
44
|
+
response?: EvalResponseExpectation;
|
|
45
|
+
tools?: EvalToolsExpectation;
|
|
46
|
+
tool_call?: ToolExpectation | ToolExpectation[];
|
|
47
|
+
toolCall?: ToolExpectation | ToolExpectation[];
|
|
48
|
+
tool_calls?: ToolExpectation | ToolExpectation[];
|
|
49
|
+
toolCalls?: ToolExpectation | ToolExpectation[];
|
|
50
|
+
tool_call_count?: number;
|
|
51
|
+
toolCallCount?: number;
|
|
52
|
+
tool_call_order?: string[];
|
|
53
|
+
toolCallOrder?: string[];
|
|
54
|
+
persisted_tool_call?: ToolExpectation | ToolExpectation[];
|
|
55
|
+
persistedToolCall?: ToolExpectation | ToolExpectation[];
|
|
56
|
+
persisted_tool_calls?: ToolExpectation | ToolExpectation[];
|
|
57
|
+
persistedToolCalls?: ToolExpectation | ToolExpectation[];
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
export type EvalResponseExpectation = {
|
|
43
61
|
contains?: string | string[];
|
|
62
|
+
contains_all?: string | string[];
|
|
63
|
+
containsAll?: string | string[];
|
|
64
|
+
contains_any?: string | string[];
|
|
65
|
+
containsAny?: string | string[];
|
|
66
|
+
case_insensitive_contains?: string | string[];
|
|
67
|
+
caseInsensitiveContains?: string | string[];
|
|
44
68
|
not_contains?: string | string[];
|
|
45
69
|
notContains?: string | string[];
|
|
46
70
|
regex?: string | string[];
|
|
47
71
|
matches_regex?: string | string[];
|
|
48
72
|
matchesRegex?: string | string[];
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
73
|
+
not_regex?: string | string[];
|
|
74
|
+
notRegex?: string | string[];
|
|
75
|
+
max_length?: number;
|
|
76
|
+
maxLength?: number;
|
|
77
|
+
};
|
|
78
|
+
|
|
79
|
+
export type EvalToolsExpectation =
|
|
80
|
+
| ToolExpectation
|
|
81
|
+
| ToolExpectation[]
|
|
82
|
+
| EvalToolsContainerExpectation;
|
|
83
|
+
|
|
84
|
+
export type EvalToolsContainerExpectation = {
|
|
85
|
+
persisted?: ToolExpectation | ToolExpectation[];
|
|
53
86
|
persisted_tool_call?: ToolExpectation | ToolExpectation[];
|
|
54
87
|
persistedToolCall?: ToolExpectation | ToolExpectation[];
|
|
55
88
|
persisted_tool_calls?: ToolExpectation | ToolExpectation[];
|
|
56
89
|
persistedToolCalls?: ToolExpectation | ToolExpectation[];
|
|
90
|
+
called?: string | string[];
|
|
91
|
+
called_once?: string | string[];
|
|
92
|
+
calledOnce?: string | string[];
|
|
93
|
+
count?: number;
|
|
94
|
+
order?: string[];
|
|
57
95
|
};
|
|
58
96
|
|
|
59
|
-
type ToolExpectation =
|
|
97
|
+
export type ToolExpectation =
|
|
60
98
|
| string
|
|
61
99
|
| {
|
|
62
100
|
name?: string;
|
|
@@ -83,6 +121,10 @@ export type EvalFromConversationResult = {
|
|
|
83
121
|
turns: number;
|
|
84
122
|
};
|
|
85
123
|
|
|
124
|
+
export function defineEval<const T extends EvalCase>(evalCase: T): T {
|
|
125
|
+
return evalCase;
|
|
126
|
+
}
|
|
127
|
+
|
|
86
128
|
export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
|
|
87
129
|
const root = await findAgentCapsuleRoot(cwd);
|
|
88
130
|
const evalFiles = await findEvalFiles(join(root, "evals"));
|
|
@@ -94,15 +136,29 @@ export async function runEvalsFromCwd(cwd: string): Promise<EvalRunSummary> {
|
|
|
94
136
|
const results: EvalResult[] = [];
|
|
95
137
|
|
|
96
138
|
for (const file of evalFiles) {
|
|
97
|
-
const
|
|
98
|
-
|
|
139
|
+
const evalFile = relative(root, file);
|
|
140
|
+
let name = evalFile;
|
|
141
|
+
let failures: string[];
|
|
142
|
+
let output: string;
|
|
143
|
+
|
|
144
|
+
try {
|
|
145
|
+
const evalCase = await loadEvalCase(file);
|
|
146
|
+
const run = await runEvalCase(root, evalFile, evalCase);
|
|
147
|
+
|
|
148
|
+
name = evalCase.name ?? evalFile;
|
|
149
|
+
failures = run.failures;
|
|
150
|
+
output = run.output;
|
|
151
|
+
} catch (error) {
|
|
152
|
+
failures = [formatEvalFailure(error)];
|
|
153
|
+
output = "";
|
|
154
|
+
}
|
|
99
155
|
|
|
100
156
|
results.push({
|
|
101
|
-
name
|
|
102
|
-
file:
|
|
103
|
-
passed:
|
|
104
|
-
failures
|
|
105
|
-
output
|
|
157
|
+
name,
|
|
158
|
+
file: evalFile,
|
|
159
|
+
passed: failures.length === 0,
|
|
160
|
+
failures,
|
|
161
|
+
output,
|
|
106
162
|
});
|
|
107
163
|
}
|
|
108
164
|
|
|
@@ -136,24 +192,29 @@ export async function writeEvalFromConversation(
|
|
|
136
192
|
);
|
|
137
193
|
}
|
|
138
194
|
|
|
139
|
-
const file =
|
|
195
|
+
const file = resolveEvalOutputPath(root, input.out, trace.title ?? trace.id);
|
|
196
|
+
const source = `import { defineEval } from "@andreprado/agentkit";
|
|
140
197
|
|
|
141
|
-
|
|
142
|
-
throw new AgentKitError(
|
|
143
|
-
"validation_error",
|
|
144
|
-
`${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
|
|
145
|
-
);
|
|
146
|
-
}
|
|
147
|
-
|
|
148
|
-
await mkdir(dirname(file), { recursive: true });
|
|
149
|
-
await writeFile(
|
|
150
|
-
file,
|
|
151
|
-
`export default {
|
|
198
|
+
export default defineEval({
|
|
152
199
|
name: ${JSON.stringify(input.name ?? `replay ${trace.title ?? trace.id}`)},
|
|
153
200
|
turns: ${formatEvalTurns(turns)},
|
|
154
|
-
};
|
|
155
|
-
|
|
156
|
-
|
|
201
|
+
});
|
|
202
|
+
`;
|
|
203
|
+
|
|
204
|
+
await mkdir(dirname(file), { recursive: true });
|
|
205
|
+
|
|
206
|
+
try {
|
|
207
|
+
await writeFile(file, source, { flag: input.force ? "w" : "wx" });
|
|
208
|
+
} catch (error) {
|
|
209
|
+
if (isNodeError(error) && error.code === "EEXIST") {
|
|
210
|
+
throw new AgentKitError(
|
|
211
|
+
"validation_error",
|
|
212
|
+
`${relative(root, file)} already exists. Re-run with --force or choose --out <path>.`,
|
|
213
|
+
);
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
throw error;
|
|
217
|
+
}
|
|
157
218
|
|
|
158
219
|
return {
|
|
159
220
|
root,
|
|
@@ -180,29 +241,61 @@ async function runEvalCase(
|
|
|
180
241
|
const conversationId = `eval_${crypto.randomUUID()}`;
|
|
181
242
|
const failures: string[] = [];
|
|
182
243
|
let output = "";
|
|
244
|
+
const now = resolveEvalNow(file, evalCase.now);
|
|
183
245
|
|
|
184
246
|
for (const [index, turn] of turns.entries()) {
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
247
|
+
try {
|
|
248
|
+
const run = await runAgentMessageFromCwd(root, {
|
|
249
|
+
message: turn.input,
|
|
250
|
+
conversationId,
|
|
251
|
+
now,
|
|
252
|
+
runtime: {
|
|
253
|
+
environment: "eval",
|
|
254
|
+
invocation: "eval",
|
|
255
|
+
},
|
|
256
|
+
});
|
|
257
|
+
const persistedToolCalls = await loadPersistedToolCalls(root, run.runId);
|
|
258
|
+
const turnFailures = evaluateExpectations(turn.expect ?? {}, run, persistedToolCalls);
|
|
259
|
+
|
|
260
|
+
for (const failure of turnFailures) {
|
|
261
|
+
failures.push(turns.length === 1 ? failure : `turn ${index + 1}: ${failure}`);
|
|
262
|
+
}
|
|
195
263
|
|
|
196
|
-
|
|
197
|
-
|
|
264
|
+
output = run.message.content;
|
|
265
|
+
} catch (error) {
|
|
266
|
+
failures.push(turns.length === 1 ? formatEvalFailure(error) : `turn ${index + 1}: ${formatEvalFailure(error)}`);
|
|
267
|
+
break;
|
|
198
268
|
}
|
|
199
|
-
|
|
200
|
-
output = run.message.content;
|
|
201
269
|
}
|
|
202
270
|
|
|
203
271
|
return { failures, output };
|
|
204
272
|
}
|
|
205
273
|
|
|
274
|
+
function resolveEvalNow(file: string, value: EvalCase["now"]): Date | undefined {
|
|
275
|
+
if (value === undefined) {
|
|
276
|
+
return undefined;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
if (typeof value === "string" && !hasExplicitIsoOffset(value)) {
|
|
280
|
+
throw new AgentKitError(
|
|
281
|
+
"validation_error",
|
|
282
|
+
`${file} now must be an ISO timestamp with an explicit timezone offset, such as "2026-02-04T02:30:00.000Z" or "2026-02-04T02:30:00-05:00".`,
|
|
283
|
+
);
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
const date = value instanceof Date ? new Date(value.getTime()) : new Date(value);
|
|
287
|
+
|
|
288
|
+
if (Number.isNaN(date.getTime())) {
|
|
289
|
+
throw new AgentKitError("validation_error", `${file} now must be a valid ISO timestamp or Date when provided.`);
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
return date;
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
function hasExplicitIsoOffset(value: string): boolean {
|
|
296
|
+
return /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(?:\.\d+)?(?:Z|[+-]\d{2}:\d{2})$/.test(value.trim());
|
|
297
|
+
}
|
|
298
|
+
|
|
206
299
|
function normalizeEvalTurns(evalCase: EvalCase): Array<{ input: string; expect?: EvalExpect }> {
|
|
207
300
|
if (evalCase.turns !== undefined) {
|
|
208
301
|
if (!Array.isArray(evalCase.turns)) {
|
|
@@ -303,12 +396,89 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
|
|
|
303
396
|
const failures: string[] = [];
|
|
304
397
|
const content = run.message.content;
|
|
305
398
|
|
|
306
|
-
for (const
|
|
399
|
+
for (const responseExpectation of responseExpectations(expect)) {
|
|
400
|
+
failures.push(...evaluateResponseExpectation(responseExpectation, content));
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
const toolAssertions = collectToolAssertions(expect);
|
|
404
|
+
|
|
405
|
+
for (const toolExpectation of toolAssertions.persisted) {
|
|
406
|
+
const match = persistedToolCalls.find((toolCall) => matchesToolExpectation(toolCall, toolExpectation));
|
|
407
|
+
|
|
408
|
+
if (!match) {
|
|
409
|
+
failures.push(`expected persisted tool call ${formatToolExpectation(toolExpectation)}`);
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
for (const expectedName of toolAssertions.called) {
|
|
414
|
+
if (!persistedToolCalls.some((toolCall) => toolCall.name === expectedName)) {
|
|
415
|
+
failures.push(`expected persisted tool call named ${JSON.stringify(expectedName)}`);
|
|
416
|
+
}
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
for (const expectedName of toolAssertions.calledOnce) {
|
|
420
|
+
const count = persistedToolCalls.filter((toolCall) => toolCall.name === expectedName).length;
|
|
421
|
+
|
|
422
|
+
if (count !== 1) {
|
|
423
|
+
failures.push(`expected persisted tool call ${JSON.stringify(expectedName)} exactly once, got ${count}`);
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
for (const expectedCount of toolAssertions.counts) {
|
|
428
|
+
if (!Number.isInteger(expectedCount) || expectedCount < 0) {
|
|
429
|
+
failures.push(`expected tool call count to be a non-negative integer, got ${JSON.stringify(expectedCount)}`);
|
|
430
|
+
continue;
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
if (persistedToolCalls.length !== expectedCount) {
|
|
434
|
+
failures.push(`expected ${expectedCount} persisted tool call(s), got ${persistedToolCalls.length}`);
|
|
435
|
+
}
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
for (const expectedOrder of toolAssertions.orders) {
|
|
439
|
+
if (!namesContainSubsequence(persistedToolNames(persistedToolCalls), expectedOrder)) {
|
|
440
|
+
failures.push(
|
|
441
|
+
`expected persisted tool call order ${JSON.stringify(expectedOrder)}, got ${JSON.stringify(
|
|
442
|
+
persistedToolNames(persistedToolCalls),
|
|
443
|
+
)}`,
|
|
444
|
+
);
|
|
445
|
+
}
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
return failures;
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
function responseExpectations(expect: EvalExpect): EvalResponseExpectation[] {
|
|
452
|
+
return [expect, expect.response].filter((value): value is EvalResponseExpectation => value !== undefined);
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
function evaluateResponseExpectation(expect: EvalResponseExpectation, content: string): string[] {
|
|
456
|
+
const failures: string[] = [];
|
|
457
|
+
|
|
458
|
+
for (const expected of [
|
|
459
|
+
...list(expect.contains),
|
|
460
|
+
...list(expect.contains_all),
|
|
461
|
+
...list(expect.containsAll),
|
|
462
|
+
]) {
|
|
307
463
|
if (!content.includes(expected)) {
|
|
308
464
|
failures.push(`expected output to contain ${JSON.stringify(expected)}`);
|
|
309
465
|
}
|
|
310
466
|
}
|
|
311
467
|
|
|
468
|
+
const containsAny = [...list(expect.contains_any), ...list(expect.containsAny)];
|
|
469
|
+
|
|
470
|
+
if (containsAny.length > 0 && !containsAny.some((expected) => content.includes(expected))) {
|
|
471
|
+
failures.push(`expected output to contain any of ${JSON.stringify(containsAny)}`);
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
const lowerContent = content.toLowerCase();
|
|
475
|
+
|
|
476
|
+
for (const expected of [...list(expect.case_insensitive_contains), ...list(expect.caseInsensitiveContains)]) {
|
|
477
|
+
if (!lowerContent.includes(expected.toLowerCase())) {
|
|
478
|
+
failures.push(`expected output to contain ${JSON.stringify(expected)} case-insensitively`);
|
|
479
|
+
}
|
|
480
|
+
}
|
|
481
|
+
|
|
312
482
|
for (const forbidden of [...list(expect.not_contains), ...list(expect.notContains)]) {
|
|
313
483
|
if (content.includes(forbidden)) {
|
|
314
484
|
failures.push(`expected output not to contain ${JSON.stringify(forbidden)}`);
|
|
@@ -321,28 +491,140 @@ function evaluateExpectations(expect: EvalExpect, run: AgentRunResult, persisted
|
|
|
321
491
|
}
|
|
322
492
|
}
|
|
323
493
|
|
|
324
|
-
const
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
...toolList(expect.persisted_tool_call),
|
|
330
|
-
...toolList(expect.persistedToolCall),
|
|
331
|
-
...toolList(expect.persisted_tool_calls),
|
|
332
|
-
...toolList(expect.persistedToolCalls),
|
|
333
|
-
];
|
|
494
|
+
for (const pattern of [...list(expect.not_regex), ...list(expect.notRegex)]) {
|
|
495
|
+
if (new RegExp(pattern).test(content)) {
|
|
496
|
+
failures.push(`expected output not to match /${pattern}/`);
|
|
497
|
+
}
|
|
498
|
+
}
|
|
334
499
|
|
|
335
|
-
for (const
|
|
336
|
-
|
|
500
|
+
for (const expectedMaxLength of [expect.max_length, expect.maxLength]) {
|
|
501
|
+
if (expectedMaxLength === undefined) {
|
|
502
|
+
continue;
|
|
503
|
+
}
|
|
337
504
|
|
|
338
|
-
if (!
|
|
339
|
-
failures.push(`expected
|
|
505
|
+
if (!Number.isInteger(expectedMaxLength) || expectedMaxLength < 0) {
|
|
506
|
+
failures.push(`expected response maxLength to be a non-negative integer, got ${JSON.stringify(expectedMaxLength)}`);
|
|
507
|
+
continue;
|
|
508
|
+
}
|
|
509
|
+
|
|
510
|
+
if (content.length > expectedMaxLength) {
|
|
511
|
+
failures.push(`expected output length to be <= ${expectedMaxLength}, got ${content.length}`);
|
|
340
512
|
}
|
|
341
513
|
}
|
|
342
514
|
|
|
343
515
|
return failures;
|
|
344
516
|
}
|
|
345
517
|
|
|
518
|
+
type ToolAssertions = {
|
|
519
|
+
persisted: ToolExpectation[];
|
|
520
|
+
called: string[];
|
|
521
|
+
calledOnce: string[];
|
|
522
|
+
counts: number[];
|
|
523
|
+
orders: string[][];
|
|
524
|
+
};
|
|
525
|
+
|
|
526
|
+
function collectToolAssertions(expect: EvalExpect): ToolAssertions {
|
|
527
|
+
const assertions: ToolAssertions = {
|
|
528
|
+
persisted: [
|
|
529
|
+
...toolList(expect.tool_call),
|
|
530
|
+
...toolList(expect.toolCall),
|
|
531
|
+
...toolList(expect.tool_calls),
|
|
532
|
+
...toolList(expect.toolCalls),
|
|
533
|
+
...toolList(expect.persisted_tool_call),
|
|
534
|
+
...toolList(expect.persistedToolCall),
|
|
535
|
+
...toolList(expect.persisted_tool_calls),
|
|
536
|
+
...toolList(expect.persistedToolCalls),
|
|
537
|
+
],
|
|
538
|
+
called: [],
|
|
539
|
+
calledOnce: [],
|
|
540
|
+
counts: [],
|
|
541
|
+
orders: [],
|
|
542
|
+
};
|
|
543
|
+
|
|
544
|
+
if (expect.tool_call_count !== undefined) {
|
|
545
|
+
assertions.counts.push(expect.tool_call_count);
|
|
546
|
+
}
|
|
547
|
+
|
|
548
|
+
if (expect.toolCallCount !== undefined) {
|
|
549
|
+
assertions.counts.push(expect.toolCallCount);
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
if (expect.tool_call_order !== undefined) {
|
|
553
|
+
assertions.orders.push(expect.tool_call_order);
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
if (expect.toolCallOrder !== undefined) {
|
|
557
|
+
assertions.orders.push(expect.toolCallOrder);
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
appendNestedToolAssertions(assertions, expect.tools);
|
|
561
|
+
|
|
562
|
+
return assertions;
|
|
563
|
+
}
|
|
564
|
+
|
|
565
|
+
function appendNestedToolAssertions(assertions: ToolAssertions, tools: EvalToolsExpectation | undefined): void {
|
|
566
|
+
if (tools === undefined) {
|
|
567
|
+
return;
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
if (!isToolsContainer(tools)) {
|
|
571
|
+
assertions.persisted.push(...toolList(tools));
|
|
572
|
+
return;
|
|
573
|
+
}
|
|
574
|
+
|
|
575
|
+
assertions.persisted.push(
|
|
576
|
+
...toolList(tools.persisted),
|
|
577
|
+
...toolList(tools.persisted_tool_call),
|
|
578
|
+
...toolList(tools.persistedToolCall),
|
|
579
|
+
...toolList(tools.persisted_tool_calls),
|
|
580
|
+
...toolList(tools.persistedToolCalls),
|
|
581
|
+
);
|
|
582
|
+
assertions.called.push(...list(tools.called));
|
|
583
|
+
assertions.calledOnce.push(...list(tools.called_once), ...list(tools.calledOnce));
|
|
584
|
+
|
|
585
|
+
if (tools.count !== undefined) {
|
|
586
|
+
assertions.counts.push(tools.count);
|
|
587
|
+
}
|
|
588
|
+
|
|
589
|
+
if (tools.order !== undefined) {
|
|
590
|
+
assertions.orders.push(tools.order);
|
|
591
|
+
}
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
function isToolsContainer(value: EvalToolsExpectation): value is EvalToolsContainerExpectation {
|
|
595
|
+
if (!isRecord(value)) {
|
|
596
|
+
return false;
|
|
597
|
+
}
|
|
598
|
+
|
|
599
|
+
if (
|
|
600
|
+
"name" in value ||
|
|
601
|
+
"input" in value ||
|
|
602
|
+
"output" in value ||
|
|
603
|
+
"rendered" in value ||
|
|
604
|
+
"status" in value ||
|
|
605
|
+
"visibility" in value
|
|
606
|
+
) {
|
|
607
|
+
return false;
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
if (Object.keys(value).length === 0) {
|
|
611
|
+
return true;
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
return (
|
|
615
|
+
"persisted" in value ||
|
|
616
|
+
"persisted_tool_call" in value ||
|
|
617
|
+
"persistedToolCall" in value ||
|
|
618
|
+
"persisted_tool_calls" in value ||
|
|
619
|
+
"persistedToolCalls" in value ||
|
|
620
|
+
"called" in value ||
|
|
621
|
+
"called_once" in value ||
|
|
622
|
+
"calledOnce" in value ||
|
|
623
|
+
"count" in value ||
|
|
624
|
+
"order" in value
|
|
625
|
+
);
|
|
626
|
+
}
|
|
627
|
+
|
|
346
628
|
function list(value: string | string[] | undefined): string[] {
|
|
347
629
|
if (value === undefined) {
|
|
348
630
|
return [];
|
|
@@ -351,6 +633,30 @@ function list(value: string | string[] | undefined): string[] {
|
|
|
351
633
|
return Array.isArray(value) ? value : [value];
|
|
352
634
|
}
|
|
353
635
|
|
|
636
|
+
function persistedToolNames(toolCalls: ToolCallSnapshot[]): string[] {
|
|
637
|
+
return toolCalls.map((toolCall) => String(toolCall.name ?? ""));
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
function namesContainSubsequence(actual: string[], expected: string[]): boolean {
|
|
641
|
+
if (expected.length === 0) {
|
|
642
|
+
return true;
|
|
643
|
+
}
|
|
644
|
+
|
|
645
|
+
let actualIndex = 0;
|
|
646
|
+
|
|
647
|
+
for (const expectedName of expected) {
|
|
648
|
+
actualIndex = actual.findIndex((actualName, index) => index >= actualIndex && actualName === expectedName);
|
|
649
|
+
|
|
650
|
+
if (actualIndex === -1) {
|
|
651
|
+
return false;
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
actualIndex += 1;
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
return true;
|
|
658
|
+
}
|
|
659
|
+
|
|
354
660
|
function toolList(value: ToolExpectation | ToolExpectation[] | undefined): ToolExpectation[] {
|
|
355
661
|
if (value === undefined) {
|
|
356
662
|
return [];
|
|
@@ -428,7 +734,9 @@ function replayTurnsFromMessages(
|
|
|
428
734
|
...(nextAssistant
|
|
429
735
|
? {
|
|
430
736
|
expect: {
|
|
431
|
-
|
|
737
|
+
response: {
|
|
738
|
+
contains: nextAssistant.content,
|
|
739
|
+
},
|
|
432
740
|
},
|
|
433
741
|
}
|
|
434
742
|
: {}),
|
|
@@ -455,13 +763,40 @@ function slugify(value: string): string {
|
|
|
455
763
|
return slug || "conversation";
|
|
456
764
|
}
|
|
457
765
|
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
766
|
+
function resolveEvalOutputPath(root: string, out: string | undefined, title: string): string {
|
|
767
|
+
const outputPath = out ?? join("evals", `replay-${slugify(title)}.eval.ts`);
|
|
768
|
+
|
|
769
|
+
if (isAbsolute(outputPath)) {
|
|
770
|
+
throw new AgentKitError("validation_error", "eval --out must be a relative path inside the Agent Capsule.");
|
|
771
|
+
}
|
|
772
|
+
|
|
773
|
+
const file = resolve(root, outputPath);
|
|
774
|
+
const relativePath = relative(root, file);
|
|
775
|
+
|
|
776
|
+
if (relativePath === "" || relativePath.startsWith("..") || isAbsolute(relativePath)) {
|
|
777
|
+
throw new AgentKitError("validation_error", "eval --out must stay inside the Agent Capsule.");
|
|
778
|
+
}
|
|
779
|
+
|
|
780
|
+
if (!/\.(eval|spec)\.[cm]?[tj]s$/.test(file)) {
|
|
781
|
+
throw new AgentKitError(
|
|
782
|
+
"validation_error",
|
|
783
|
+
"eval --out must end with .eval.ts, .eval.js, .spec.ts, or another AgentKit eval/spec extension.",
|
|
784
|
+
);
|
|
464
785
|
}
|
|
786
|
+
|
|
787
|
+
return file;
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
function formatEvalFailure(error: unknown): string {
|
|
791
|
+
if (isAgentKitError(error)) {
|
|
792
|
+
return `${error.code}: ${error.message}`;
|
|
793
|
+
}
|
|
794
|
+
|
|
795
|
+
if (error instanceof Error) {
|
|
796
|
+
return `runtime_error: ${error.message}`;
|
|
797
|
+
}
|
|
798
|
+
|
|
799
|
+
return `runtime_error: ${String(error)}`;
|
|
465
800
|
}
|
|
466
801
|
|
|
467
802
|
function jsonContains(actual: unknown, expected: unknown): boolean {
|
package/src/runtime/inspect.ts
CHANGED
|
@@ -6,6 +6,7 @@ import { loadAgentCapsule } from "./config";
|
|
|
6
6
|
import { loadCapsuleEnv } from "./env";
|
|
7
7
|
import { resolveKnowledgeConfig } from "./knowledge/config";
|
|
8
8
|
import { KNOWLEDGE_SEARCH_TOOL_NAME } from "./knowledge/tool";
|
|
9
|
+
import { resolveRuntimeTimeZone } from "./prompt-context";
|
|
9
10
|
|
|
10
11
|
export type SecretState = "set" | "missing";
|
|
11
12
|
|
|
@@ -13,6 +14,10 @@ export type AgentInspectState = {
|
|
|
13
14
|
agent: string;
|
|
14
15
|
runtime: AgentRuntime;
|
|
15
16
|
provider: AgentProvider;
|
|
17
|
+
timeZone: {
|
|
18
|
+
configured: string | null;
|
|
19
|
+
effective: string;
|
|
20
|
+
};
|
|
16
21
|
prompt: string;
|
|
17
22
|
tools: string[];
|
|
18
23
|
channels: AgentInspectChannel[];
|
|
@@ -122,6 +127,13 @@ export function buildInspectState(
|
|
|
122
127
|
agent: capsule.config.name,
|
|
123
128
|
runtime: capsule.config.runtime,
|
|
124
129
|
provider: capsule.config.provider,
|
|
130
|
+
timeZone: {
|
|
131
|
+
configured: capsule.config.timeZone ?? null,
|
|
132
|
+
effective: resolveRuntimeTimeZone({
|
|
133
|
+
timeZone: capsule.config.timeZone,
|
|
134
|
+
env,
|
|
135
|
+
}),
|
|
136
|
+
},
|
|
125
137
|
prompt: toCapsulePath(capsule.root, capsule.instructionsPath),
|
|
126
138
|
tools: (capsule.config.tools ?? []).map((tool) => tool.name),
|
|
127
139
|
channels: (capsule.config.channels ?? []).map((channel) => ({
|