@cursor/july 0.1.19 → 0.1.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +10 -0
- package/dist/bin/agent-serve.js +0 -0
- package/dist/channels/github/instrument.d.ts +20 -0
- package/dist/channels/github/instrument.d.ts.map +1 -0
- package/dist/channels/slack/api.d.ts +2 -0
- package/dist/channels/slack/api.d.ts.map +1 -1
- package/dist/channels/slack/api.js +3 -0
- package/dist/channels/slack/bot-mentions.d.ts +106 -0
- package/dist/channels/slack/bot-mentions.d.ts.map +1 -0
- package/dist/channels/slack/bot-mentions.js +243 -0
- package/dist/channels/slack/dispatch.d.ts +6 -0
- package/dist/channels/slack/dispatch.d.ts.map +1 -1
- package/dist/channels/slack/dispatch.js +24 -3
- package/dist/channels/slack/inbound.d.ts +10 -1
- package/dist/channels/slack/inbound.d.ts.map +1 -1
- package/dist/channels/slack/inbound.js +12 -4
- package/dist/channels/slack/index.d.ts +1 -0
- package/dist/channels/slack/index.d.ts.map +1 -1
- package/dist/channels/slack/index.js +1 -0
- package/dist/channels/slack/slack-channel.d.ts.map +1 -1
- package/dist/channels/slack/slack-channel.js +61 -22
- package/dist/channels/slack/types.d.ts +58 -0
- package/dist/channels/slack/types.d.ts.map +1 -1
- package/dist/docs/404.html +1 -1
- package/dist/docs/ab.html +2 -2
- package/dist/docs/assets/{app.CrsWMchO.js → app.jDxLzWv4.js} +1 -1
- package/dist/docs/assets/chunks/@localSearchIndexroot.DoJHJjqF.js +1 -0
- package/dist/docs/assets/chunks/{VPLocalSearchBox.D1JqzSh8.js → VPLocalSearchBox.Y6bDR1-a.js} +1 -1
- package/dist/docs/assets/chunks/{theme.DaBvZYwl.js → theme.CLazCWlJ.js} +2 -2
- package/dist/docs/building-with-agents.html +2 -2
- package/dist/docs/concepts.html +2 -2
- package/dist/docs/deployment.html +2 -2
- package/dist/docs/evals.html +2 -2
- package/dist/docs/example-agents/approval-buddy.html +2 -2
- package/dist/docs/example-agents/benny.html +2 -2
- package/dist/docs/example-agents/bugbot.html +2 -2
- package/dist/docs/example-agents/codebase-wiki.html +2 -2
- package/dist/docs/example-agents/codeowners-review.html +2 -2
- package/dist/docs/example-agents/concierge.html +2 -2
- package/dist/docs/example-agents/fsd.html +2 -2
- package/dist/docs/example-agents/index.html +2 -2
- package/dist/docs/example-agents/knowledge-base.html +2 -2
- package/dist/docs/example-agents/oncall.html +2 -2
- package/dist/docs/example-agents/security-reviewer.html +2 -2
- package/dist/docs/example-agents/slack-agent.html +2 -2
- package/dist/docs/example-agents/weather-agent.html +2 -2
- package/dist/docs/guides/agent-to-agent.html +2 -2
- package/dist/docs/guides/cloud-runtime.html +2 -2
- package/dist/docs/guides/github.html +2 -2
- package/dist/docs/guides/human-in-the-loop.html +2 -2
- package/dist/docs/guides/mcp-oauth.html +2 -2
- package/dist/docs/guides/slack.html +2 -2
- package/dist/docs/guides/webhooks.html +2 -2
- package/dist/docs/hillclimbing.html +2 -2
- package/dist/docs/index.html +2 -2
- package/dist/docs/quickstart.html +2 -2
- package/dist/docs/reference/agent-config.html +2 -2
- package/dist/docs/reference/channels.html +2 -2
- package/dist/docs/reference/cli.html +2 -2
- package/dist/docs/reference/connections.html +2 -2
- package/dist/docs/reference/hooks.html +2 -2
- package/dist/docs/reference/http-api.html +2 -2
- package/dist/docs/reference/instructions.html +2 -2
- package/dist/docs/reference/playground.html +2 -2
- package/dist/docs/reference/project-layout.html +2 -2
- package/dist/docs/reference/prompt.html +2 -2
- package/dist/docs/reference/schedules.html +2 -2
- package/dist/docs/reference/sessions.html +2 -2
- package/dist/docs/reference/skills.html +2 -2
- package/dist/docs/reference/subagents.html +2 -2
- package/dist/docs/reference/tools.html +2 -2
- package/dist/docs/scaffolding-agents.html +2 -2
- package/dist/docs/storage.html +2 -2
- package/dist/docs/troubleshooting.html +2 -2
- package/dist/internal/json-dir-store.d.ts +32 -0
- package/dist/internal/json-dir-store.d.ts.map +1 -0
- package/dist/internal/json-dir-store.js +100 -0
- package/dist/internal/resolved-connections.d.ts +1 -1
- package/dist/internal/resolved-connections.d.ts.map +1 -1
- package/dist/internal/resolved-connections.js +10 -0
- package/dist/internal/session-engine.d.ts +4 -1
- package/dist/internal/session-engine.d.ts.map +1 -1
- package/dist/internal/session-engine.js +13 -2
- package/dist/playground/assets/index-CjOQ4hN9.css +1 -0
- package/dist/playground/assets/{index-C2SU2xV5.js → index-dshZQJCp.js} +46 -46
- package/dist/playground/index.html +2 -2
- package/dist/types.d.ts +7 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +9 -0
- package/package.json +25 -25
- package/skills/create-agent/SKILL.md +36 -4
- package/skills/evals/SKILL.md +3 -0
- package/skills/framework-map/SKILL.md +14 -0
- package/skills/github/SKILL.md +6 -1
- package/skills/hillclimb/SKILL.md +28 -7
- package/src/bin/agent-serve.version.test.ts +62 -0
- package/src/channels/github/api.test.ts +64 -0
- package/src/channels/github/auth.test.ts +105 -0
- package/src/channels/github/cursor-account.test.ts +204 -0
- package/src/channels/github/forward.test.ts +457 -0
- package/src/channels/github/github.test.ts +937 -0
- package/src/channels/github/replay.test.ts +179 -0
- package/src/channels/slack/api.post-message.test.ts +148 -0
- package/src/channels/slack/api.ts +5 -0
- package/src/channels/slack/approvals.test.ts +328 -0
- package/src/channels/slack/block-actions.test.ts +452 -0
- package/src/channels/slack/bot-mentions.test.ts +217 -0
- package/src/channels/slack/bot-mentions.ts +315 -0
- package/src/channels/slack/channel-watch.test.ts +363 -0
- package/src/channels/slack/cursor-account.test.ts +253 -0
- package/src/channels/slack/defaults.final-post.test.ts +182 -0
- package/src/channels/slack/dispatch.test.ts +795 -0
- package/src/channels/slack/dispatch.ts +27 -1
- package/src/channels/slack/eval-directive.test.ts +273 -0
- package/src/channels/slack/inbound.ts +25 -4
- package/src/channels/slack/index.ts +1 -0
- package/src/channels/slack/message-body.test.ts +54 -0
- package/src/channels/slack/nudge-store.test.ts +143 -0
- package/src/channels/slack/slack-channel.ts +66 -8
- package/src/channels/slack/slack.test.ts +391 -0
- package/src/channels/slack/stop.test.ts +23 -0
- package/src/channels/slack/thread-context.test.ts +202 -0
- package/src/channels/slack/types.ts +60 -0
- package/src/evals/assertions.test.ts +580 -0
- package/src/evals/expect.test.ts +144 -0
- package/src/evals/judge.test.ts +181 -0
- package/src/evals/loaders.test.ts +132 -0
- package/src/evals/matchers.test.ts +95 -0
- package/src/evals/reporters.test.ts +303 -0
- package/src/evals/run-facts.test.ts +259 -0
- package/src/internal/ab-snapshot.test.ts +325 -0
- package/src/internal/approval-gate.test.ts +49 -0
- package/src/internal/approvals.integration.test.ts +383 -0
- package/src/internal/authored-loaders.test.ts +31 -0
- package/src/internal/builtin-tools/reminders.test.ts +201 -0
- package/src/internal/channel-route-schema.test.ts +294 -0
- package/src/internal/chat-attach.test.ts +262 -0
- package/src/internal/cli-deploy.test.ts +1991 -0
- package/src/internal/cli-mcp.test.ts +789 -0
- package/src/internal/cli-skills.test.ts +133 -0
- package/src/internal/cli-slack.test.ts +1647 -0
- package/src/internal/cloud-merge.test.ts +74 -0
- package/src/internal/cron.test.ts +22 -0
- package/src/internal/cursor/account-mcp.test.ts +807 -0
- package/src/internal/cursor/backend-client.test.ts +591 -0
- package/src/internal/cursor/credentials.test.ts +351 -0
- package/src/internal/cursor/github-credentials.test.ts +136 -0
- package/src/internal/cursor-account-mcp-auth.test.ts +310 -0
- package/src/internal/cursor-account.integration.test.ts +441 -0
- package/src/internal/cursor-event-relay.test.ts +746 -0
- package/src/internal/cursor-github-credentials.integration.test.ts +271 -0
- package/src/internal/cursor-slack-relay.test.ts +525 -0
- package/src/internal/deploy-source.test.ts +111 -0
- package/src/internal/discovery.builtin-tools.test.ts +94 -0
- package/src/internal/discovery.concurrency.test.ts +60 -0
- package/src/internal/discovery.cursor-account.test.ts +133 -0
- package/src/internal/discovery.cwd.test.ts +83 -0
- package/src/internal/discovery.hosting.test.ts +80 -0
- package/src/internal/discovery.identity.test.ts +44 -0
- package/src/internal/docs-site.test.ts +66 -0
- package/src/internal/duration.test.ts +29 -0
- package/src/internal/eval-judge-model.test.ts +187 -0
- package/src/internal/eval-run-store.cancel.test.ts +142 -0
- package/src/internal/eval-run-store.storage.test.ts +211 -0
- package/src/internal/eval-runner.http.test.ts +403 -0
- package/src/internal/eval-runner.run.test.ts +928 -0
- package/src/internal/evals-client.test.ts +307 -0
- package/src/internal/event-mapper.test.ts +243 -0
- package/src/internal/github-fanout.test.ts +213 -0
- package/src/internal/handleAgentServeTrigger.test.ts +179 -0
- package/src/internal/host-kv.test.ts +82 -0
- package/src/internal/host-platforms.test.ts +126 -0
- package/src/internal/http-channel.test.ts +402 -0
- package/src/internal/init-project.test.ts +269 -0
- package/src/internal/install-cursor-skills.test.ts +262 -0
- package/src/internal/local-env.test.ts +100 -0
- package/src/internal/log-ring.test.ts +31 -0
- package/src/internal/logs-client.test.ts +350 -0
- package/src/internal/mcp-endpoint.test.ts +436 -0
- package/src/internal/mcp-host.test.ts +298 -0
- package/src/internal/mcp-oauth.test.ts +148 -0
- package/src/internal/net.test.ts +17 -0
- package/src/internal/peer-connections.test.ts +128 -0
- package/src/internal/peer-mcp.integration.test.ts +289 -0
- package/src/internal/playground/toolchain.test.ts +53 -0
- package/src/internal/playground-cli.test.ts +187 -0
- package/src/internal/playground-proxy.test.ts +376 -0
- package/src/internal/prompt-context.integration.test.ts +232 -0
- package/src/internal/prompt-context.test.ts +127 -0
- package/src/internal/reminder-runner.test.ts +390 -0
- package/src/internal/reminder-store.test.ts +53 -0
- package/src/internal/request-headers.test.ts +27 -0
- package/src/internal/resolve-prod-target.test.ts +787 -0
- package/src/internal/resolved-connections.test.ts +295 -0
- package/src/internal/resolved-connections.ts +18 -1
- package/src/internal/router.test.ts +57 -0
- package/src/internal/sdk-runner.test.ts +290 -0
- package/src/internal/session-engine.coalesce.test.ts +169 -0
- package/src/internal/session-engine.concurrency.test.ts +250 -0
- package/src/internal/session-engine.host-oauth-mcp.test.ts +110 -0
- package/src/internal/session-engine.interrupt.test.ts +577 -0
- package/src/internal/session-engine.storage.test.ts +547 -0
- package/src/internal/session-engine.ts +15 -1
- package/src/internal/session-urls.test.ts +28 -0
- package/src/internal/sessions-client.test.ts +518 -0
- package/src/internal/storage-coordinator.test.ts +517 -0
- package/src/internal/tool-call.test.ts +458 -0
- package/src/internal/tool-result.test.ts +52 -0
- package/src/internal/trajectory.approvals.test.ts +83 -0
- package/src/internal/trajectory.subagents.test.ts +198 -0
- package/src/internal/turn-governor.test.ts +137 -0
- package/src/internal/update-check.test.ts +485 -0
- package/src/internal/workspace.test.ts +81 -0
- package/src/storage-backends/cursor-hosted.test.ts +121 -0
- package/src/types.ts +12 -0
- package/dist/docs/assets/chunks/@localSearchIndexroot.BnHRjfoe.js +0 -1
- package/dist/playground/assets/index-CidizGZv.css +0 -1
|
@@ -0,0 +1,303 @@
|
|
|
1
|
+
import { mkdtemp, readFile, rm } from "node:fs/promises";
|
|
2
|
+
import { tmpdir } from "node:os";
|
|
3
|
+
import { join } from "node:path";
|
|
4
|
+
import { afterEach, describe, expect, it, vi } from "vitest";
|
|
5
|
+
import {
|
|
6
|
+
Artifacts,
|
|
7
|
+
combineReporters,
|
|
8
|
+
JUnit,
|
|
9
|
+
renderJUnitXml,
|
|
10
|
+
sanitizeCaseId,
|
|
11
|
+
} from "./reporters.js";
|
|
12
|
+
import type { EvalReporter, EvalRunResult, EvalRunSummary } from "./results.js";
|
|
13
|
+
import { summarizeEvalResults } from "./results.js";
|
|
14
|
+
|
|
15
|
+
const dirs: string[] = [];
|
|
16
|
+
|
|
17
|
+
async function tempDir(): Promise<string> {
|
|
18
|
+
const dir = await mkdtemp(join(tmpdir(), "agentkit-eval-reporters-"));
|
|
19
|
+
dirs.push(dir);
|
|
20
|
+
return dir;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
afterEach(async () => {
|
|
24
|
+
await Promise.all(
|
|
25
|
+
dirs.splice(0).map((dir) => rm(dir, { recursive: true, force: true }))
|
|
26
|
+
);
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
function result(overrides: Partial<EvalRunResult> = {}): EvalRunResult {
|
|
30
|
+
return {
|
|
31
|
+
id: "weather/nyc",
|
|
32
|
+
path: "/repo/evals/weather.eval.ts",
|
|
33
|
+
ok: true,
|
|
34
|
+
verdict: "passed",
|
|
35
|
+
assertions: [],
|
|
36
|
+
inputs: ["What is the weather in NYC?"],
|
|
37
|
+
logs: [],
|
|
38
|
+
metrics: {},
|
|
39
|
+
durationMs: 1500,
|
|
40
|
+
...overrides,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function summary(results: EvalRunResult[], strict = false): EvalRunSummary {
|
|
45
|
+
return summarizeEvalResults(results, {
|
|
46
|
+
strict,
|
|
47
|
+
startedAt: "2026-01-01T00:00:00.000Z",
|
|
48
|
+
finishedAt: "2026-01-01T00:00:05.000Z",
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
describe("summarizeEvalResults", () => {
|
|
53
|
+
it("tallies every verdict and the wall time", () => {
|
|
54
|
+
const totals = summary([
|
|
55
|
+
result(),
|
|
56
|
+
result({ id: "b", verdict: "failed", ok: false }),
|
|
57
|
+
result({ id: "c", verdict: "scored" }),
|
|
58
|
+
result({ id: "d", verdict: "skipped", skipReason: "no judge model" }),
|
|
59
|
+
]);
|
|
60
|
+
expect(totals).toMatchObject({
|
|
61
|
+
total: 4,
|
|
62
|
+
passed: 1,
|
|
63
|
+
failed: 1,
|
|
64
|
+
scored: 1,
|
|
65
|
+
skipped: 1,
|
|
66
|
+
durationMs: 5000,
|
|
67
|
+
});
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
it("infers a verdict from ok for results written before verdicts existed", () => {
|
|
71
|
+
const totals = summary([
|
|
72
|
+
{ ...result(), verdict: undefined },
|
|
73
|
+
{ ...result({ id: "b", ok: false }), verdict: undefined },
|
|
74
|
+
]);
|
|
75
|
+
expect(totals).toMatchObject({ passed: 1, failed: 1 });
|
|
76
|
+
});
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
describe("renderJUnitXml", () => {
|
|
80
|
+
it("emits one testcase per eval with scores on system-out", () => {
|
|
81
|
+
const xml = renderJUnitXml(
|
|
82
|
+
summary([
|
|
83
|
+
result({
|
|
84
|
+
assertions: [
|
|
85
|
+
{ name: "succeeded", passed: true },
|
|
86
|
+
{
|
|
87
|
+
name: "judge.closedQA(cites a source)",
|
|
88
|
+
passed: true,
|
|
89
|
+
severity: "soft",
|
|
90
|
+
score: 0.75,
|
|
91
|
+
threshold: 0.6,
|
|
92
|
+
},
|
|
93
|
+
],
|
|
94
|
+
metrics: { recall: "40.0% (4/10)" },
|
|
95
|
+
logs: ["matched=[a, b]"],
|
|
96
|
+
}),
|
|
97
|
+
]),
|
|
98
|
+
"agentkit-evals"
|
|
99
|
+
);
|
|
100
|
+
expect(xml).toContain('<testcase name="weather/nyc"');
|
|
101
|
+
expect(xml).not.toContain("<failure");
|
|
102
|
+
expect(xml).toContain("judge.closedQA(cites a source)=0.750 (>= 0.6)");
|
|
103
|
+
expect(xml).toContain("recall=40.0% (4/10)");
|
|
104
|
+
expect(xml).toContain("matched=[a, b]");
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
it("reports a failed gate as a failure with the assertion detail", () => {
|
|
108
|
+
const xml = renderJUnitXml(
|
|
109
|
+
summary([
|
|
110
|
+
result({
|
|
111
|
+
verdict: "failed",
|
|
112
|
+
ok: false,
|
|
113
|
+
assertions: [
|
|
114
|
+
{
|
|
115
|
+
name: "calledTool(get_weather)",
|
|
116
|
+
passed: false,
|
|
117
|
+
detail: "called []",
|
|
118
|
+
},
|
|
119
|
+
],
|
|
120
|
+
}),
|
|
121
|
+
]),
|
|
122
|
+
"suite"
|
|
123
|
+
);
|
|
124
|
+
expect(xml).toContain('failures="1"');
|
|
125
|
+
expect(xml).toContain(
|
|
126
|
+
'<failure message="calledTool(get_weather) (called [])" type="failed"/>'
|
|
127
|
+
);
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
it("prefers the execution error message when the test body threw", () => {
|
|
131
|
+
const xml = renderJUnitXml(
|
|
132
|
+
summary([
|
|
133
|
+
result({ verdict: "failed", ok: false, error: "turn timed out" }),
|
|
134
|
+
]),
|
|
135
|
+
"suite"
|
|
136
|
+
);
|
|
137
|
+
expect(xml).toContain('<failure message="turn timed out"');
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
it("marks a skipped eval skipped and keeps it out of the failure count", () => {
|
|
141
|
+
const xml = renderJUnitXml(
|
|
142
|
+
summary([
|
|
143
|
+
result({ verdict: "skipped", skipReason: "no judge credentials" }),
|
|
144
|
+
]),
|
|
145
|
+
"suite"
|
|
146
|
+
);
|
|
147
|
+
expect(xml).toContain('<skipped message="no judge credentials"/>');
|
|
148
|
+
expect(xml).toContain('failures="0"');
|
|
149
|
+
expect(xml).toContain('skipped="1"');
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
it("only fails a scored eval under strict", () => {
|
|
153
|
+
const scored = [result({ verdict: "scored" })];
|
|
154
|
+
expect(renderJUnitXml(summary(scored, false), "s")).not.toContain(
|
|
155
|
+
"<failure"
|
|
156
|
+
);
|
|
157
|
+
const strict = renderJUnitXml(summary(scored, true), "s");
|
|
158
|
+
expect(strict).toContain('type="scored"');
|
|
159
|
+
expect(strict).toContain('failures="1"');
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
it("escapes XML metacharacters and strips illegal control characters", () => {
|
|
163
|
+
const xml = renderJUnitXml(
|
|
164
|
+
summary([
|
|
165
|
+
result({
|
|
166
|
+
id: 'weird & "id" <x>',
|
|
167
|
+
verdict: "failed",
|
|
168
|
+
ok: false,
|
|
169
|
+
error: "bad\u0000output",
|
|
170
|
+
}),
|
|
171
|
+
]),
|
|
172
|
+
"suite"
|
|
173
|
+
);
|
|
174
|
+
expect(xml).toContain('name="weird & "id" <x>"');
|
|
175
|
+
expect(xml).toContain('message="badoutput"');
|
|
176
|
+
expect(xml).not.toContain("\u0000");
|
|
177
|
+
});
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
describe("JUnit reporter", () => {
|
|
181
|
+
it("writes the XML file, creating parent directories", async () => {
|
|
182
|
+
const dir = await tempDir();
|
|
183
|
+
const filePath = join(dir, "nested", "junit.xml");
|
|
184
|
+
const reporter = JUnit({ filePath });
|
|
185
|
+
await reporter.onRunComplete?.(summary([result()]));
|
|
186
|
+
const xml = await readFile(filePath, "utf8");
|
|
187
|
+
expect(xml).toContain('<testcase name="weather/nyc"');
|
|
188
|
+
});
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
describe("Artifacts reporter", () => {
|
|
192
|
+
it("writes a summary, an index, and per-case detail", async () => {
|
|
193
|
+
const dir = await tempDir();
|
|
194
|
+
const reporter = Artifacts({ dir });
|
|
195
|
+
const passing = result();
|
|
196
|
+
const failing = result({
|
|
197
|
+
id: "weather/paris",
|
|
198
|
+
verdict: "failed",
|
|
199
|
+
ok: false,
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
await reporter.onRunStart?.(
|
|
203
|
+
[
|
|
204
|
+
{ id: "weather/nyc", fileId: "weather" },
|
|
205
|
+
{ id: "weather/paris", fileId: "weather" },
|
|
206
|
+
],
|
|
207
|
+
{ baseUrl: "http://127.0.0.1:3000", mode: "local" }
|
|
208
|
+
);
|
|
209
|
+
await reporter.onEvalComplete?.(passing);
|
|
210
|
+
await reporter.onEvalComplete?.(failing);
|
|
211
|
+
await reporter.onRunComplete?.(summary([passing, failing]));
|
|
212
|
+
|
|
213
|
+
const runSummary = JSON.parse(
|
|
214
|
+
await readFile(join(dir, "summary.json"), "utf8")
|
|
215
|
+
) as Record<string, unknown>;
|
|
216
|
+
expect(runSummary).toMatchObject({
|
|
217
|
+
total: 2,
|
|
218
|
+
passed: 1,
|
|
219
|
+
failed: 1,
|
|
220
|
+
caseCount: 2,
|
|
221
|
+
target: { baseUrl: "http://127.0.0.1:3000", mode: "local" },
|
|
222
|
+
discovered: ["weather/nyc", "weather/paris"],
|
|
223
|
+
});
|
|
224
|
+
// The full result list stays in the per-case files, not the summary.
|
|
225
|
+
expect(runSummary.results).toBeUndefined();
|
|
226
|
+
|
|
227
|
+
const index = (await readFile(join(dir, "results.jsonl"), "utf8"))
|
|
228
|
+
.trim()
|
|
229
|
+
.split("\n")
|
|
230
|
+
.map((line) => JSON.parse(line) as Record<string, unknown>);
|
|
231
|
+
expect(index).toEqual([
|
|
232
|
+
{ id: "weather/nyc", verdict: "passed", ok: true, durationMs: 1500 },
|
|
233
|
+
{ id: "weather/paris", verdict: "failed", ok: false, durationMs: 1500 },
|
|
234
|
+
]);
|
|
235
|
+
|
|
236
|
+
const detail = JSON.parse(
|
|
237
|
+
await readFile(join(dir, "evals", "weather", "nyc.json"), "utf8")
|
|
238
|
+
) as EvalRunResult;
|
|
239
|
+
expect(detail.inputs).toEqual(["What is the weather in NYC?"]);
|
|
240
|
+
});
|
|
241
|
+
|
|
242
|
+
it("keeps a hand-built case id inside the run directory", () => {
|
|
243
|
+
expect(sanitizeCaseId("weather/nyc")).toBe("weather/nyc");
|
|
244
|
+
expect(sanitizeCaseId("../../etc/passwd")).toBe("etc/passwd");
|
|
245
|
+
expect(sanitizeCaseId("a/../b")).toBe("a/b");
|
|
246
|
+
expect(sanitizeCaseId("weird name!")).toBe("weird_name_");
|
|
247
|
+
});
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
describe("combineReporters", () => {
|
|
251
|
+
it("fans every hook out to each reporter", async () => {
|
|
252
|
+
const calls: string[] = [];
|
|
253
|
+
const make = (name: string): EvalReporter => ({
|
|
254
|
+
onRunStart: () => {
|
|
255
|
+
calls.push(`${name}:start`);
|
|
256
|
+
},
|
|
257
|
+
onEvalComplete: () => {
|
|
258
|
+
calls.push(`${name}:case`);
|
|
259
|
+
},
|
|
260
|
+
onRunComplete: () => {
|
|
261
|
+
calls.push(`${name}:complete`);
|
|
262
|
+
},
|
|
263
|
+
});
|
|
264
|
+
const combined = combineReporters([make("a"), make("b")]);
|
|
265
|
+
await combined.onRunStart?.([], { baseUrl: "x", mode: "local" });
|
|
266
|
+
await combined.onEvalComplete?.(result());
|
|
267
|
+
await combined.onRunComplete?.(summary([result()]));
|
|
268
|
+
expect(calls).toEqual([
|
|
269
|
+
"a:start",
|
|
270
|
+
"b:start",
|
|
271
|
+
"a:case",
|
|
272
|
+
"b:case",
|
|
273
|
+
"a:complete",
|
|
274
|
+
"b:complete",
|
|
275
|
+
]);
|
|
276
|
+
});
|
|
277
|
+
|
|
278
|
+
it("isolates a throwing reporter so the run still finishes", async () => {
|
|
279
|
+
const onError = vi.fn();
|
|
280
|
+
const healthy = vi.fn();
|
|
281
|
+
const combined = combineReporters(
|
|
282
|
+
[
|
|
283
|
+
{
|
|
284
|
+
onEvalComplete: () => {
|
|
285
|
+
throw new Error("upload failed");
|
|
286
|
+
},
|
|
287
|
+
},
|
|
288
|
+
{ onEvalComplete: healthy },
|
|
289
|
+
],
|
|
290
|
+
onError
|
|
291
|
+
);
|
|
292
|
+
await combined.onEvalComplete?.(result());
|
|
293
|
+
expect(onError).toHaveBeenCalledTimes(1);
|
|
294
|
+
expect(healthy).toHaveBeenCalledTimes(1);
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
it("tolerates reporters that implement only some hooks", async () => {
|
|
298
|
+
const combined = combineReporters([{}]);
|
|
299
|
+
await expect(
|
|
300
|
+
combined.onRunComplete?.(summary([result()]))
|
|
301
|
+
).resolves.toBeUndefined();
|
|
302
|
+
});
|
|
303
|
+
});
|
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
import { describe, expect, it } from "vitest";
|
|
2
|
+
import type { SessionEvent } from "../types.js";
|
|
3
|
+
import { deriveRunFacts } from "./run-facts.js";
|
|
4
|
+
|
|
5
|
+
let sequence = 0;
|
|
6
|
+
|
|
7
|
+
function event(
|
|
8
|
+
type: string,
|
|
9
|
+
data: unknown,
|
|
10
|
+
extra: { turnId?: string } = {}
|
|
11
|
+
): SessionEvent {
|
|
12
|
+
sequence++;
|
|
13
|
+
return {
|
|
14
|
+
sessionId: "ses_1",
|
|
15
|
+
seq: sequence,
|
|
16
|
+
at: new Date(sequence * 1000).toISOString(),
|
|
17
|
+
turnId: extra.turnId ?? "turn_1",
|
|
18
|
+
type,
|
|
19
|
+
data,
|
|
20
|
+
} as unknown as SessionEvent;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function toolTurn(): SessionEvent[] {
|
|
24
|
+
return [
|
|
25
|
+
event("message.received", { text: "weather in NYC?" }),
|
|
26
|
+
event("actions.requested", {
|
|
27
|
+
calls: [{ callId: "c1", toolName: "get_weather", args: { city: "NYC" } }],
|
|
28
|
+
}),
|
|
29
|
+
event("action.result", {
|
|
30
|
+
callId: "c1",
|
|
31
|
+
toolName: "get_weather",
|
|
32
|
+
output: { tempF: 72 },
|
|
33
|
+
isError: false,
|
|
34
|
+
}),
|
|
35
|
+
event("message.completed", { text: "Sunny, 72F", finishReason: "stop" }),
|
|
36
|
+
event("turn.completed", {}),
|
|
37
|
+
event("session.waiting", {}),
|
|
38
|
+
];
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
describe("deriveRunFacts tool lifecycle", () => {
|
|
42
|
+
it("resolves a requested call with its result instead of duplicating it", () => {
|
|
43
|
+
const facts = deriveRunFacts(toolTurn());
|
|
44
|
+
expect(facts.toolCalls).toHaveLength(1);
|
|
45
|
+
expect(facts.toolCalls[0]).toMatchObject({
|
|
46
|
+
callId: "c1",
|
|
47
|
+
toolName: "get_weather",
|
|
48
|
+
input: { city: "NYC" },
|
|
49
|
+
output: { tempF: 72 },
|
|
50
|
+
status: "completed",
|
|
51
|
+
index: 0,
|
|
52
|
+
});
|
|
53
|
+
expect(facts.ok).toBe(true);
|
|
54
|
+
expect(facts.failedToolCalls).toEqual([]);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
it("marks a call pending until its result arrives", () => {
|
|
58
|
+
const events = toolTurn().filter((e) => e.type !== "action.result");
|
|
59
|
+
const facts = deriveRunFacts(events);
|
|
60
|
+
expect(facts.toolCalls[0]?.status).toBe("pending");
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
it("marks an errored call failed and lists it under failedToolCalls", () => {
|
|
64
|
+
const events = [
|
|
65
|
+
event("actions.requested", {
|
|
66
|
+
calls: [{ callId: "c1", toolName: "bash", args: { command: "false" } }],
|
|
67
|
+
}),
|
|
68
|
+
event("action.result", {
|
|
69
|
+
callId: "c1",
|
|
70
|
+
toolName: "bash",
|
|
71
|
+
output: "exit 1",
|
|
72
|
+
isError: true,
|
|
73
|
+
}),
|
|
74
|
+
event("turn.completed", {}),
|
|
75
|
+
];
|
|
76
|
+
const facts = deriveRunFacts(events);
|
|
77
|
+
expect(facts.toolCalls[0]?.status).toBe("failed");
|
|
78
|
+
expect(facts.failedToolCalls.map((c) => c.toolName)).toEqual(["bash"]);
|
|
79
|
+
// A failed tool does not by itself fail the turn.
|
|
80
|
+
expect(facts.ok).toBe(true);
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
it("records a result-only call so a dropped request event still counts", () => {
|
|
84
|
+
const facts = deriveRunFacts([
|
|
85
|
+
event("action.result", {
|
|
86
|
+
callId: "c9",
|
|
87
|
+
toolName: "read_file",
|
|
88
|
+
output: "hi",
|
|
89
|
+
isError: false,
|
|
90
|
+
}),
|
|
91
|
+
event("turn.completed", {}),
|
|
92
|
+
]);
|
|
93
|
+
expect(facts.toolCalls.map((c) => c.toolName)).toEqual(["read_file"]);
|
|
94
|
+
expect(facts.toolCalls[0]?.status).toBe("completed");
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
it("keeps calls in request order across turns", () => {
|
|
98
|
+
const facts = deriveRunFacts([
|
|
99
|
+
event("actions.requested", {
|
|
100
|
+
calls: [{ callId: "a", toolName: "first", args: {} }],
|
|
101
|
+
}),
|
|
102
|
+
event(
|
|
103
|
+
"actions.requested",
|
|
104
|
+
{ calls: [{ callId: "b", toolName: "second", args: {} }] },
|
|
105
|
+
{ turnId: "turn_2" }
|
|
106
|
+
),
|
|
107
|
+
event("turn.completed", {}, { turnId: "turn_2" }),
|
|
108
|
+
]);
|
|
109
|
+
expect(facts.toolCalls.map((c) => c.toolName)).toEqual(["first", "second"]);
|
|
110
|
+
expect(facts.turns).toBe(2);
|
|
111
|
+
});
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
describe("deriveRunFacts human-in-the-loop", () => {
|
|
115
|
+
it("parks on an unanswered approval request", () => {
|
|
116
|
+
const facts = deriveRunFacts([
|
|
117
|
+
event("actions.requested", {
|
|
118
|
+
calls: [{ callId: "c1", toolName: "deploy", args: { env: "prod" } }],
|
|
119
|
+
}),
|
|
120
|
+
event("action.approval_requested", {
|
|
121
|
+
callId: "c1",
|
|
122
|
+
toolName: "deploy",
|
|
123
|
+
args: { env: "prod" },
|
|
124
|
+
}),
|
|
125
|
+
event("session.waiting", {}),
|
|
126
|
+
]);
|
|
127
|
+
expect(facts.parked).toBe(true);
|
|
128
|
+
expect(facts.pendingInputRequests).toHaveLength(1);
|
|
129
|
+
expect(facts.pendingInputRequests[0]).toMatchObject({
|
|
130
|
+
toolName: "deploy",
|
|
131
|
+
args: { env: "prod" },
|
|
132
|
+
resolved: false,
|
|
133
|
+
});
|
|
134
|
+
expect(facts.toolCalls[0]?.status).toBe("pending");
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
it("is not parked once the approval resolves", () => {
|
|
138
|
+
const facts = deriveRunFacts([
|
|
139
|
+
event("actions.requested", {
|
|
140
|
+
calls: [{ callId: "c1", toolName: "deploy", args: {} }],
|
|
141
|
+
}),
|
|
142
|
+
event("action.approval_requested", { callId: "c1", toolName: "deploy" }),
|
|
143
|
+
event("action.approval_resolved", {
|
|
144
|
+
callId: "c1",
|
|
145
|
+
toolName: "deploy",
|
|
146
|
+
decision: "approve",
|
|
147
|
+
}),
|
|
148
|
+
event("action.result", {
|
|
149
|
+
callId: "c1",
|
|
150
|
+
toolName: "deploy",
|
|
151
|
+
output: "ok",
|
|
152
|
+
isError: false,
|
|
153
|
+
}),
|
|
154
|
+
event("turn.completed", {}),
|
|
155
|
+
]);
|
|
156
|
+
expect(facts.parked).toBe(false);
|
|
157
|
+
expect(facts.toolCalls[0]).toMatchObject({
|
|
158
|
+
status: "completed",
|
|
159
|
+
approvalDecision: "approve",
|
|
160
|
+
});
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
it("keeps a denied call rejected even after its synthetic error result", () => {
|
|
164
|
+
const facts = deriveRunFacts([
|
|
165
|
+
event("actions.requested", {
|
|
166
|
+
calls: [{ callId: "c1", toolName: "deploy", args: {} }],
|
|
167
|
+
}),
|
|
168
|
+
event("action.approval_requested", { callId: "c1", toolName: "deploy" }),
|
|
169
|
+
event("action.approval_resolved", {
|
|
170
|
+
callId: "c1",
|
|
171
|
+
toolName: "deploy",
|
|
172
|
+
decision: "deny",
|
|
173
|
+
}),
|
|
174
|
+
event("action.result", {
|
|
175
|
+
callId: "c1",
|
|
176
|
+
toolName: "deploy",
|
|
177
|
+
output: "denied by user",
|
|
178
|
+
isError: true,
|
|
179
|
+
}),
|
|
180
|
+
event("turn.completed", {}),
|
|
181
|
+
]);
|
|
182
|
+
expect(facts.toolCalls[0]?.status).toBe("rejected");
|
|
183
|
+
// A rejection is a policy outcome, not a broken tool.
|
|
184
|
+
expect(facts.failedToolCalls).toEqual([]);
|
|
185
|
+
});
|
|
186
|
+
});
|
|
187
|
+
|
|
188
|
+
describe("deriveRunFacts subagents and failures", () => {
|
|
189
|
+
it("tracks a subagent delegation and its output", () => {
|
|
190
|
+
const facts = deriveRunFacts([
|
|
191
|
+
event("actions.requested", {
|
|
192
|
+
calls: [{ callId: "t1", toolName: "task", args: { name: "research" } }],
|
|
193
|
+
}),
|
|
194
|
+
event("subagent.called", {
|
|
195
|
+
callId: "t1",
|
|
196
|
+
name: "research",
|
|
197
|
+
description: "look it up",
|
|
198
|
+
}),
|
|
199
|
+
event("message.completed", {
|
|
200
|
+
text: "nested chatter",
|
|
201
|
+
finishReason: "stop",
|
|
202
|
+
parentCallId: "t1",
|
|
203
|
+
}),
|
|
204
|
+
event("subagent.completed", { callId: "t1", name: "research" }),
|
|
205
|
+
event("action.result", {
|
|
206
|
+
callId: "t1",
|
|
207
|
+
toolName: "task",
|
|
208
|
+
output: "found it",
|
|
209
|
+
isError: false,
|
|
210
|
+
}),
|
|
211
|
+
event("message.completed", { text: "Done.", finishReason: "stop" }),
|
|
212
|
+
event("turn.completed", {}),
|
|
213
|
+
]);
|
|
214
|
+
expect(facts.subagents).toHaveLength(1);
|
|
215
|
+
expect(facts.subagents[0]).toMatchObject({
|
|
216
|
+
name: "research",
|
|
217
|
+
status: "completed",
|
|
218
|
+
output: "found it",
|
|
219
|
+
});
|
|
220
|
+
// Nested subagent text is not part of the agent's own reply.
|
|
221
|
+
expect(facts.assistantMessages).toEqual(["Done."]);
|
|
222
|
+
});
|
|
223
|
+
|
|
224
|
+
it("fails the run on a failed turn", () => {
|
|
225
|
+
const facts = deriveRunFacts([
|
|
226
|
+
event("message.received", { text: "hi" }),
|
|
227
|
+
event("turn.failed", { message: "model error" }),
|
|
228
|
+
]);
|
|
229
|
+
expect(facts.ok).toBe(false);
|
|
230
|
+
expect(facts.turnsFailed).toBe(1);
|
|
231
|
+
expect(facts.failureMessage).toBe("model error");
|
|
232
|
+
});
|
|
233
|
+
|
|
234
|
+
it("fails the run on a session failure with no turn id", () => {
|
|
235
|
+
const events = [
|
|
236
|
+
event("message.received", { text: "hi" }),
|
|
237
|
+
{
|
|
238
|
+
...event("session.failed", { message: "transport died" }),
|
|
239
|
+
turnId: undefined,
|
|
240
|
+
},
|
|
241
|
+
] as SessionEvent[];
|
|
242
|
+
const facts = deriveRunFacts(events);
|
|
243
|
+
expect(facts.ok).toBe(false);
|
|
244
|
+
expect(facts.failureMessage).toBe("transport died");
|
|
245
|
+
});
|
|
246
|
+
|
|
247
|
+
it("is not ok with no turns at all", () => {
|
|
248
|
+
expect(deriveRunFacts([]).ok).toBe(false);
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
it("joins assistant messages in order", () => {
|
|
252
|
+
const facts = deriveRunFacts([
|
|
253
|
+
event("message.completed", { text: "first", finishReason: "tool_call" }),
|
|
254
|
+
event("message.completed", { text: "second", finishReason: "stop" }),
|
|
255
|
+
event("turn.completed", {}),
|
|
256
|
+
]);
|
|
257
|
+
expect(facts.assistantText).toBe("first\nsecond");
|
|
258
|
+
});
|
|
259
|
+
});
|