@cursor/july 0.1.19 → 0.1.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (217) hide show
  1. package/AGENTS.md +10 -0
  2. package/dist/bin/agent-serve.js +0 -0
  3. package/dist/channels/github/instrument.d.ts +20 -0
  4. package/dist/channels/github/instrument.d.ts.map +1 -0
  5. package/dist/channels/slack/api.d.ts +2 -0
  6. package/dist/channels/slack/api.d.ts.map +1 -1
  7. package/dist/channels/slack/api.js +3 -0
  8. package/dist/channels/slack/bot-mentions.d.ts +106 -0
  9. package/dist/channels/slack/bot-mentions.d.ts.map +1 -0
  10. package/dist/channels/slack/bot-mentions.js +243 -0
  11. package/dist/channels/slack/dispatch.d.ts +6 -0
  12. package/dist/channels/slack/dispatch.d.ts.map +1 -1
  13. package/dist/channels/slack/dispatch.js +24 -3
  14. package/dist/channels/slack/inbound.d.ts +10 -1
  15. package/dist/channels/slack/inbound.d.ts.map +1 -1
  16. package/dist/channels/slack/inbound.js +12 -4
  17. package/dist/channels/slack/index.d.ts +1 -0
  18. package/dist/channels/slack/index.d.ts.map +1 -1
  19. package/dist/channels/slack/index.js +1 -0
  20. package/dist/channels/slack/slack-channel.d.ts.map +1 -1
  21. package/dist/channels/slack/slack-channel.js +61 -22
  22. package/dist/channels/slack/types.d.ts +58 -0
  23. package/dist/channels/slack/types.d.ts.map +1 -1
  24. package/dist/docs/404.html +1 -1
  25. package/dist/docs/ab.html +2 -2
  26. package/dist/docs/assets/{app.CrsWMchO.js → app.jDxLzWv4.js} +1 -1
  27. package/dist/docs/assets/chunks/@localSearchIndexroot.DoJHJjqF.js +1 -0
  28. package/dist/docs/assets/chunks/{VPLocalSearchBox.D1JqzSh8.js → VPLocalSearchBox.Y6bDR1-a.js} +1 -1
  29. package/dist/docs/assets/chunks/{theme.DaBvZYwl.js → theme.CLazCWlJ.js} +2 -2
  30. package/dist/docs/building-with-agents.html +2 -2
  31. package/dist/docs/concepts.html +2 -2
  32. package/dist/docs/deployment.html +2 -2
  33. package/dist/docs/evals.html +2 -2
  34. package/dist/docs/example-agents/approval-buddy.html +2 -2
  35. package/dist/docs/example-agents/benny.html +2 -2
  36. package/dist/docs/example-agents/bugbot.html +2 -2
  37. package/dist/docs/example-agents/codebase-wiki.html +2 -2
  38. package/dist/docs/example-agents/codeowners-review.html +2 -2
  39. package/dist/docs/example-agents/concierge.html +2 -2
  40. package/dist/docs/example-agents/fsd.html +2 -2
  41. package/dist/docs/example-agents/index.html +2 -2
  42. package/dist/docs/example-agents/knowledge-base.html +2 -2
  43. package/dist/docs/example-agents/oncall.html +2 -2
  44. package/dist/docs/example-agents/security-reviewer.html +2 -2
  45. package/dist/docs/example-agents/slack-agent.html +2 -2
  46. package/dist/docs/example-agents/weather-agent.html +2 -2
  47. package/dist/docs/guides/agent-to-agent.html +2 -2
  48. package/dist/docs/guides/cloud-runtime.html +2 -2
  49. package/dist/docs/guides/github.html +2 -2
  50. package/dist/docs/guides/human-in-the-loop.html +2 -2
  51. package/dist/docs/guides/mcp-oauth.html +2 -2
  52. package/dist/docs/guides/slack.html +2 -2
  53. package/dist/docs/guides/webhooks.html +2 -2
  54. package/dist/docs/hillclimbing.html +2 -2
  55. package/dist/docs/index.html +2 -2
  56. package/dist/docs/quickstart.html +2 -2
  57. package/dist/docs/reference/agent-config.html +2 -2
  58. package/dist/docs/reference/channels.html +2 -2
  59. package/dist/docs/reference/cli.html +2 -2
  60. package/dist/docs/reference/connections.html +2 -2
  61. package/dist/docs/reference/hooks.html +2 -2
  62. package/dist/docs/reference/http-api.html +2 -2
  63. package/dist/docs/reference/instructions.html +2 -2
  64. package/dist/docs/reference/playground.html +2 -2
  65. package/dist/docs/reference/project-layout.html +2 -2
  66. package/dist/docs/reference/prompt.html +2 -2
  67. package/dist/docs/reference/schedules.html +2 -2
  68. package/dist/docs/reference/sessions.html +2 -2
  69. package/dist/docs/reference/skills.html +2 -2
  70. package/dist/docs/reference/subagents.html +2 -2
  71. package/dist/docs/reference/tools.html +2 -2
  72. package/dist/docs/scaffolding-agents.html +2 -2
  73. package/dist/docs/storage.html +2 -2
  74. package/dist/docs/troubleshooting.html +2 -2
  75. package/dist/internal/json-dir-store.d.ts +32 -0
  76. package/dist/internal/json-dir-store.d.ts.map +1 -0
  77. package/dist/internal/json-dir-store.js +100 -0
  78. package/dist/internal/resolved-connections.d.ts +1 -1
  79. package/dist/internal/resolved-connections.d.ts.map +1 -1
  80. package/dist/internal/resolved-connections.js +10 -0
  81. package/dist/internal/session-engine.d.ts +4 -1
  82. package/dist/internal/session-engine.d.ts.map +1 -1
  83. package/dist/internal/session-engine.js +13 -2
  84. package/dist/playground/assets/index-CjOQ4hN9.css +1 -0
  85. package/dist/playground/assets/{index-C2SU2xV5.js → index-dshZQJCp.js} +46 -46
  86. package/dist/playground/index.html +2 -2
  87. package/dist/types.d.ts +7 -0
  88. package/dist/types.d.ts.map +1 -1
  89. package/dist/types.js +9 -0
  90. package/package.json +25 -25
  91. package/skills/create-agent/SKILL.md +36 -4
  92. package/skills/evals/SKILL.md +3 -0
  93. package/skills/framework-map/SKILL.md +14 -0
  94. package/skills/github/SKILL.md +6 -1
  95. package/skills/hillclimb/SKILL.md +28 -7
  96. package/src/bin/agent-serve.version.test.ts +62 -0
  97. package/src/channels/github/api.test.ts +64 -0
  98. package/src/channels/github/auth.test.ts +105 -0
  99. package/src/channels/github/cursor-account.test.ts +204 -0
  100. package/src/channels/github/forward.test.ts +457 -0
  101. package/src/channels/github/github.test.ts +937 -0
  102. package/src/channels/github/replay.test.ts +179 -0
  103. package/src/channels/slack/api.post-message.test.ts +148 -0
  104. package/src/channels/slack/api.ts +5 -0
  105. package/src/channels/slack/approvals.test.ts +328 -0
  106. package/src/channels/slack/block-actions.test.ts +452 -0
  107. package/src/channels/slack/bot-mentions.test.ts +217 -0
  108. package/src/channels/slack/bot-mentions.ts +315 -0
  109. package/src/channels/slack/channel-watch.test.ts +363 -0
  110. package/src/channels/slack/cursor-account.test.ts +253 -0
  111. package/src/channels/slack/defaults.final-post.test.ts +182 -0
  112. package/src/channels/slack/dispatch.test.ts +795 -0
  113. package/src/channels/slack/dispatch.ts +27 -1
  114. package/src/channels/slack/eval-directive.test.ts +273 -0
  115. package/src/channels/slack/inbound.ts +25 -4
  116. package/src/channels/slack/index.ts +1 -0
  117. package/src/channels/slack/message-body.test.ts +54 -0
  118. package/src/channels/slack/nudge-store.test.ts +143 -0
  119. package/src/channels/slack/slack-channel.ts +66 -8
  120. package/src/channels/slack/slack.test.ts +391 -0
  121. package/src/channels/slack/stop.test.ts +23 -0
  122. package/src/channels/slack/thread-context.test.ts +202 -0
  123. package/src/channels/slack/types.ts +60 -0
  124. package/src/evals/assertions.test.ts +580 -0
  125. package/src/evals/expect.test.ts +144 -0
  126. package/src/evals/judge.test.ts +181 -0
  127. package/src/evals/loaders.test.ts +132 -0
  128. package/src/evals/matchers.test.ts +95 -0
  129. package/src/evals/reporters.test.ts +303 -0
  130. package/src/evals/run-facts.test.ts +259 -0
  131. package/src/internal/ab-snapshot.test.ts +325 -0
  132. package/src/internal/approval-gate.test.ts +49 -0
  133. package/src/internal/approvals.integration.test.ts +383 -0
  134. package/src/internal/authored-loaders.test.ts +31 -0
  135. package/src/internal/builtin-tools/reminders.test.ts +201 -0
  136. package/src/internal/channel-route-schema.test.ts +294 -0
  137. package/src/internal/chat-attach.test.ts +262 -0
  138. package/src/internal/cli-deploy.test.ts +1991 -0
  139. package/src/internal/cli-mcp.test.ts +789 -0
  140. package/src/internal/cli-skills.test.ts +133 -0
  141. package/src/internal/cli-slack.test.ts +1647 -0
  142. package/src/internal/cloud-merge.test.ts +74 -0
  143. package/src/internal/cron.test.ts +22 -0
  144. package/src/internal/cursor/account-mcp.test.ts +807 -0
  145. package/src/internal/cursor/backend-client.test.ts +591 -0
  146. package/src/internal/cursor/credentials.test.ts +351 -0
  147. package/src/internal/cursor/github-credentials.test.ts +136 -0
  148. package/src/internal/cursor-account-mcp-auth.test.ts +310 -0
  149. package/src/internal/cursor-account.integration.test.ts +441 -0
  150. package/src/internal/cursor-event-relay.test.ts +746 -0
  151. package/src/internal/cursor-github-credentials.integration.test.ts +271 -0
  152. package/src/internal/cursor-slack-relay.test.ts +525 -0
  153. package/src/internal/deploy-source.test.ts +111 -0
  154. package/src/internal/discovery.builtin-tools.test.ts +94 -0
  155. package/src/internal/discovery.concurrency.test.ts +60 -0
  156. package/src/internal/discovery.cursor-account.test.ts +133 -0
  157. package/src/internal/discovery.cwd.test.ts +83 -0
  158. package/src/internal/discovery.hosting.test.ts +80 -0
  159. package/src/internal/discovery.identity.test.ts +44 -0
  160. package/src/internal/docs-site.test.ts +66 -0
  161. package/src/internal/duration.test.ts +29 -0
  162. package/src/internal/eval-judge-model.test.ts +187 -0
  163. package/src/internal/eval-run-store.cancel.test.ts +142 -0
  164. package/src/internal/eval-run-store.storage.test.ts +211 -0
  165. package/src/internal/eval-runner.http.test.ts +403 -0
  166. package/src/internal/eval-runner.run.test.ts +928 -0
  167. package/src/internal/evals-client.test.ts +307 -0
  168. package/src/internal/event-mapper.test.ts +243 -0
  169. package/src/internal/github-fanout.test.ts +213 -0
  170. package/src/internal/handleAgentServeTrigger.test.ts +179 -0
  171. package/src/internal/host-kv.test.ts +82 -0
  172. package/src/internal/host-platforms.test.ts +126 -0
  173. package/src/internal/http-channel.test.ts +402 -0
  174. package/src/internal/init-project.test.ts +269 -0
  175. package/src/internal/install-cursor-skills.test.ts +262 -0
  176. package/src/internal/local-env.test.ts +100 -0
  177. package/src/internal/log-ring.test.ts +31 -0
  178. package/src/internal/logs-client.test.ts +350 -0
  179. package/src/internal/mcp-endpoint.test.ts +436 -0
  180. package/src/internal/mcp-host.test.ts +298 -0
  181. package/src/internal/mcp-oauth.test.ts +148 -0
  182. package/src/internal/net.test.ts +17 -0
  183. package/src/internal/peer-connections.test.ts +128 -0
  184. package/src/internal/peer-mcp.integration.test.ts +289 -0
  185. package/src/internal/playground/toolchain.test.ts +53 -0
  186. package/src/internal/playground-cli.test.ts +187 -0
  187. package/src/internal/playground-proxy.test.ts +376 -0
  188. package/src/internal/prompt-context.integration.test.ts +232 -0
  189. package/src/internal/prompt-context.test.ts +127 -0
  190. package/src/internal/reminder-runner.test.ts +390 -0
  191. package/src/internal/reminder-store.test.ts +53 -0
  192. package/src/internal/request-headers.test.ts +27 -0
  193. package/src/internal/resolve-prod-target.test.ts +787 -0
  194. package/src/internal/resolved-connections.test.ts +295 -0
  195. package/src/internal/resolved-connections.ts +18 -1
  196. package/src/internal/router.test.ts +57 -0
  197. package/src/internal/sdk-runner.test.ts +290 -0
  198. package/src/internal/session-engine.coalesce.test.ts +169 -0
  199. package/src/internal/session-engine.concurrency.test.ts +250 -0
  200. package/src/internal/session-engine.host-oauth-mcp.test.ts +110 -0
  201. package/src/internal/session-engine.interrupt.test.ts +577 -0
  202. package/src/internal/session-engine.storage.test.ts +547 -0
  203. package/src/internal/session-engine.ts +15 -1
  204. package/src/internal/session-urls.test.ts +28 -0
  205. package/src/internal/sessions-client.test.ts +518 -0
  206. package/src/internal/storage-coordinator.test.ts +517 -0
  207. package/src/internal/tool-call.test.ts +458 -0
  208. package/src/internal/tool-result.test.ts +52 -0
  209. package/src/internal/trajectory.approvals.test.ts +83 -0
  210. package/src/internal/trajectory.subagents.test.ts +198 -0
  211. package/src/internal/turn-governor.test.ts +137 -0
  212. package/src/internal/update-check.test.ts +485 -0
  213. package/src/internal/workspace.test.ts +81 -0
  214. package/src/storage-backends/cursor-hosted.test.ts +121 -0
  215. package/src/types.ts +12 -0
  216. package/dist/docs/assets/chunks/@localSearchIndexroot.BnHRjfoe.js +0 -1
  217. package/dist/playground/assets/index-CidizGZv.css +0 -1
@@ -0,0 +1,187 @@
1
+ import { beforeEach, describe, expect, it, vi } from "vitest";
2
+ import { EvalJudgeUnavailableError } from "../evals/judge.js";
3
+ import {
4
+ assertJudgeBackendMatch,
5
+ callJudgeModel,
6
+ resolveJudgeModel,
7
+ } from "./eval-judge-model.js";
8
+
9
+ const create = vi.hoisted(() => vi.fn());
10
+ const resolveApiKey = vi.hoisted(() => vi.fn());
11
+
12
+ vi.mock("@cursor/sdk", () => ({ Agent: { create } }));
13
+
14
+ vi.mock("./cursor/credentials.js", async (importOriginal) => {
15
+ const actual =
16
+ await importOriginal<typeof import("./cursor/credentials.js")>();
17
+ return { ...actual, resolveApiKey };
18
+ });
19
+
20
+ beforeEach(() => {
21
+ create.mockReset();
22
+ resolveApiKey.mockReset();
23
+ resolveApiKey.mockResolvedValue({
24
+ apiKey: "key_1",
25
+ source: "env",
26
+ backendUrl: "https://api2.cursor.sh",
27
+ });
28
+ create.mockResolvedValue({
29
+ send: async () => ({
30
+ wait: async () => ({ status: "completed", result: "CHOICE: Y" }),
31
+ }),
32
+ close: () => undefined,
33
+ });
34
+ });
35
+
36
+ describe("resolveJudgeModel", () => {
37
+ it("resolves innermost-first and ignores a blank env value", () => {
38
+ expect(
39
+ resolveJudgeModel({
40
+ call: "call-model",
41
+ evalLevel: "eval-model",
42
+ config: "config-model",
43
+ env: "env-model",
44
+ })
45
+ ).toBe("call-model");
46
+ expect(
47
+ resolveJudgeModel({ evalLevel: "eval-model", config: "config-model" })
48
+ ).toBe("eval-model");
49
+ expect(resolveJudgeModel({ config: "config-model" })).toBe("config-model");
50
+ expect(resolveJudgeModel({ env: " env-model " })).toBe("env-model");
51
+ expect(resolveJudgeModel({ env: " " })).toBeUndefined();
52
+ expect(resolveJudgeModel({})).toBeUndefined();
53
+ });
54
+ });
55
+
56
+ describe("assertJudgeBackendMatch", () => {
57
+ it("allows a credential scoped to the backend the SDK will use", () => {
58
+ expect(() =>
59
+ assertJudgeBackendMatch(
60
+ "https://api2.cursor.sh",
61
+ "https://api2.cursor.sh"
62
+ )
63
+ ).not.toThrow();
64
+ });
65
+
66
+ it("ignores a trailing-slash difference", () => {
67
+ expect(() =>
68
+ assertJudgeBackendMatch(
69
+ "https://api2.cursor.sh/",
70
+ "https://api2.cursor.sh"
71
+ )
72
+ ).not.toThrow();
73
+ });
74
+
75
+ it("declines when the credential belongs to a different backend", () => {
76
+ // The SDK takes no backend option, so sending the key anyway would ship it
77
+ // (and the judge prompt) to a host it was not minted for.
78
+ expect(() =>
79
+ assertJudgeBackendMatch(
80
+ "https://staging.example.com",
81
+ "https://api2.cursor.sh"
82
+ )
83
+ ).toThrow(EvalJudgeUnavailableError);
84
+ expect(() =>
85
+ assertJudgeBackendMatch(
86
+ "https://staging.example.com",
87
+ "https://api2.cursor.sh"
88
+ )
89
+ ).toThrow(/does not read AGENT_SERVE_CURSOR_BACKEND_URL/);
90
+ });
91
+ });
92
+
93
+ describe("callJudgeModel", () => {
94
+ it("skips visibly when there are no credentials", async () => {
95
+ resolveApiKey.mockResolvedValue(undefined);
96
+ await expect(
97
+ callJudgeModel({ prompt: "grade this", model: "gpt-5.4-mini" })
98
+ ).rejects.toThrow(EvalJudgeUnavailableError);
99
+ expect(create).not.toHaveBeenCalled();
100
+ });
101
+
102
+ it("confines the grading session so an injected reply cannot reach the host", async () => {
103
+ await callJudgeModel({ prompt: "grade this", model: "gpt-5.4-mini" });
104
+
105
+ expect(create).toHaveBeenCalledTimes(1);
106
+ const options = create.mock.calls[0]![0] as {
107
+ mcpServers: Record<string, unknown>;
108
+ local: {
109
+ cwd: string;
110
+ settingSources: string[];
111
+ sandboxOptions: { enabled: boolean };
112
+ };
113
+ };
114
+ // Grading embeds untrusted agent output, so the judge gets no MCP surface,
115
+ // no ambient settings, an empty scratch cwd, and a sandbox.
116
+ expect(options.mcpServers).toEqual({});
117
+ expect(options.local.settingSources).toEqual([]);
118
+ expect(options.local.sandboxOptions).toEqual({ enabled: true });
119
+ expect(options.local.cwd).toMatch(/agentkit-eval-judge-/);
120
+ });
121
+
122
+ it("never opens a session when the credential's backend disagrees", async () => {
123
+ resolveApiKey.mockResolvedValue({
124
+ apiKey: "key_1",
125
+ source: "login",
126
+ backendUrl: "https://staging.example.com",
127
+ });
128
+ await expect(
129
+ callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
130
+ ).rejects.toThrow(EvalJudgeUnavailableError);
131
+ // The point of failing closed: the key never reaches the SDK.
132
+ expect(create).not.toHaveBeenCalled();
133
+ });
134
+
135
+ it("passes a bare model id through as a selection object", async () => {
136
+ await callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" });
137
+ expect(create.mock.calls[0]![0]).toMatchObject({
138
+ apiKey: "key_1",
139
+ model: { id: "gpt-5.4-mini" },
140
+ });
141
+ });
142
+
143
+ it("preserves model params when given a selection object", async () => {
144
+ await callJudgeModel({
145
+ prompt: "p",
146
+ model: { id: "gpt-5.4-mini", params: [{ id: "effort", value: "low" }] },
147
+ });
148
+ expect(create.mock.calls[0]![0]).toMatchObject({
149
+ model: { id: "gpt-5.4-mini", params: [{ id: "effort", value: "low" }] },
150
+ });
151
+ });
152
+
153
+ it("returns the judge reply", async () => {
154
+ await expect(
155
+ callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
156
+ ).resolves.toBe("CHOICE: Y");
157
+ });
158
+
159
+ it("surfaces a failed grading run as an error", async () => {
160
+ create.mockResolvedValue({
161
+ send: async () => ({
162
+ wait: async () => ({
163
+ status: "error",
164
+ error: { message: "provider unavailable" },
165
+ }),
166
+ }),
167
+ close: () => undefined,
168
+ });
169
+ await expect(
170
+ callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
171
+ ).rejects.toThrow(/eval judge run failed: provider unavailable/);
172
+ });
173
+
174
+ it("closes the session even when grading throws", async () => {
175
+ const close = vi.fn();
176
+ create.mockResolvedValue({
177
+ send: async () => {
178
+ throw new Error("boom");
179
+ },
180
+ close,
181
+ });
182
+ await expect(
183
+ callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
184
+ ).rejects.toThrow("boom");
185
+ expect(close).toHaveBeenCalledTimes(1);
186
+ });
187
+ });
@@ -0,0 +1,142 @@
1
+ import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
2
+ import { tmpdir } from "node:os";
3
+ import { join } from "node:path";
4
+ import { afterEach, describe, expect, it, vi } from "vitest";
5
+ import {
6
+ EvalRunNotRunningError,
7
+ EvalRunStore,
8
+ EvalRunUnknownError,
9
+ } from "./eval-run-store.js";
10
+
11
+ vi.mock("./eval-runner.js", async (importOriginal) => {
12
+ const actual = await importOriginal<typeof import("./eval-runner.js")>();
13
+ return {
14
+ ...actual,
15
+ runDiscoveredEvals: vi.fn(
16
+ async (options: {
17
+ signal?: AbortSignal;
18
+ onCaseStart?: (id: string) => void;
19
+ onCaseDone?: (result: {
20
+ id: string;
21
+ path: string;
22
+ ok: boolean;
23
+ assertions: [];
24
+ logs: string[];
25
+ metrics: Record<string, string | number>;
26
+ inputs: string[];
27
+ toolCalls: [];
28
+ tools: string[];
29
+ durationMs: number;
30
+ }) => void;
31
+ discovered?: {
32
+ evals: Array<{ id: string; path: string }>;
33
+ };
34
+ }) => {
35
+ const first = options.discovered?.evals[0];
36
+ if (first !== undefined) {
37
+ options.onCaseStart?.(first.id);
38
+ }
39
+ await new Promise<void>((resolve, reject) => {
40
+ const onAbort = (): void => {
41
+ options.signal?.removeEventListener("abort", onAbort);
42
+ reject(new Error("cancelled"));
43
+ };
44
+ if (options.signal?.aborted) {
45
+ onAbort();
46
+ return;
47
+ }
48
+ options.signal?.addEventListener("abort", onAbort, { once: true });
49
+ // Keep the batch "running" until cancel aborts.
50
+ setTimeout(() => {
51
+ options.signal?.removeEventListener("abort", onAbort);
52
+ if (first !== undefined) {
53
+ options.onCaseDone?.({
54
+ id: first.id,
55
+ path: first.path,
56
+ ok: true,
57
+ assertions: [],
58
+ logs: [],
59
+ metrics: {},
60
+ inputs: [],
61
+ toolCalls: [],
62
+ tools: [],
63
+ durationMs: 1,
64
+ });
65
+ }
66
+ resolve();
67
+ }, 5_000);
68
+ });
69
+ return [];
70
+ }
71
+ ),
72
+ };
73
+ });
74
+
75
+ describe("EvalRunStore.cancel", () => {
76
+ let dir: string | undefined;
77
+
78
+ afterEach(async () => {
79
+ if (dir !== undefined) {
80
+ await rm(dir, { recursive: true, force: true });
81
+ dir = undefined;
82
+ }
83
+ vi.clearAllMocks();
84
+ });
85
+
86
+ it("cancels a running batch by Eval ID and clears the active run", async () => {
87
+ dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-cancel-"));
88
+ await mkdir(join(dir, "evals"), { recursive: true });
89
+ await writeFile(
90
+ join(dir, "evals", "evals.config.js"),
91
+ "export default { maxConcurrency: 1 };\n",
92
+ "utf8"
93
+ );
94
+ await writeFile(
95
+ join(dir, "evals", "smoke.eval.js"),
96
+ `export default {
97
+ __agentServe: "eval",
98
+ async test() {}
99
+ };
100
+ `,
101
+ "utf8"
102
+ );
103
+
104
+ const store = new EvalRunStore(dir, () => {});
105
+ store.loopbackUrl = "http://127.0.0.1:9/smoke";
106
+ const started = await store.start();
107
+ expect(started.status).toBe("running");
108
+ expect(store.getActiveRunId()).toBe(started.runId);
109
+
110
+ const cancelled = await store.cancel(started.runId);
111
+ expect(cancelled.status).toBe("cancelled");
112
+ expect(cancelled.error).toContain("cancelled");
113
+ expect(store.getActiveRunId()).toBeUndefined();
114
+ expect(await store.cancel(started.runId)).toMatchObject({
115
+ status: "cancelled",
116
+ });
117
+ await expect(store.cancel("evalrun_missing")).rejects.toBeInstanceOf(
118
+ EvalRunUnknownError
119
+ );
120
+ await expect(store.cancel(started.runId)).resolves.toMatchObject({
121
+ status: "cancelled",
122
+ });
123
+
124
+ // After cancel, a completed-looking batch cannot be cancelled again as running.
125
+ const completedId = started.runId;
126
+ const snap = await store.get(completedId);
127
+ expect(snap?.status).toBe("cancelled");
128
+ // Force a non-running status to exercise NotRunningError via a fresh id.
129
+ const fakeId = "evalrun_done_only";
130
+ (
131
+ store as unknown as {
132
+ runs: Map<string, { runId: string; status: string }>;
133
+ }
134
+ ).runs.set(fakeId, {
135
+ runId: fakeId,
136
+ status: "completed",
137
+ });
138
+ await expect(store.cancel(fakeId)).rejects.toBeInstanceOf(
139
+ EvalRunNotRunningError
140
+ );
141
+ });
142
+ });
@@ -0,0 +1,211 @@
1
+ import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
2
+ import { tmpdir } from "node:os";
3
+ import { join } from "node:path";
4
+ import { afterEach, describe, expect, it } from "vitest";
5
+ import type { EvalRunSnapshot } from "../evals.js";
6
+ import { defineStorage } from "../storage.js";
7
+ import { EvalRunStore } from "./eval-run-store.js";
8
+ import { StorageCoordinator } from "./storage-coordinator.js";
9
+
10
+ function completedRun(runId: string): EvalRunSnapshot {
11
+ return {
12
+ runId,
13
+ status: "completed",
14
+ startedAt: "2026-01-01T00:00:00.000Z",
15
+ finishedAt: "2026-01-01T00:00:01.000Z",
16
+ summary: { passed: 1, failed: 0, total: 1, done: 1 },
17
+ cases: [],
18
+ config: { maxPlaygroundRuns: 20, durableRuns: true },
19
+ };
20
+ }
21
+
22
+ describe("EvalRunStore defineStorage hydrate", () => {
23
+ let dir: string | undefined;
24
+
25
+ afterEach(async () => {
26
+ if (dir !== undefined) {
27
+ await rm(dir, { recursive: true, force: true });
28
+ dir = undefined;
29
+ }
30
+ });
31
+
32
+ it("hydrates playground history from the defineStorage evals table", async () => {
33
+ dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
34
+ await mkdir(join(dir, "evals"), { recursive: true });
35
+ await writeFile(
36
+ join(dir, "evals", "evals.config.js"),
37
+ "export default { maxConcurrency: 1 };\n",
38
+ "utf8"
39
+ );
40
+
41
+ const table = new Map<string, EvalRunSnapshot>();
42
+ const coordinator = new StorageCoordinator({
43
+ definition: defineStorage({
44
+ put: () => {},
45
+ evals: {
46
+ put: (run) => {
47
+ table.set(run.runId, run);
48
+ },
49
+ delete: (runId) => {
50
+ table.delete(runId);
51
+ },
52
+ list: () => [...table.values()],
53
+ },
54
+ }),
55
+ agentName: "test-agent",
56
+ projectRoot: dir,
57
+ logger: () => {},
58
+ });
59
+
60
+ const storage = coordinator.evalRuns();
61
+ expect(storage).toBeDefined();
62
+ storage!.save(completedRun("evalrun_from_sink"));
63
+ await coordinator.whenIdle();
64
+
65
+ const store = new EvalRunStore(dir, () => {}, coordinator.evalRuns());
66
+ await store.hydrate();
67
+ const listed = await store.listRuns();
68
+ expect(listed).toHaveLength(1);
69
+ expect(listed[0]?.runId).toBe("evalrun_from_sink");
70
+ expect(listed[0]?.status).toBe("completed");
71
+ });
72
+
73
+ it("marks interrupted running batches as failed on hydrate", async () => {
74
+ dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
75
+ await mkdir(join(dir, "evals"), { recursive: true });
76
+ await writeFile(
77
+ join(dir, "evals", "evals.config.js"),
78
+ "export default { maxConcurrency: 1 };\n",
79
+ "utf8"
80
+ );
81
+
82
+ const interrupted: EvalRunSnapshot = {
83
+ ...completedRun("evalrun_interrupted"),
84
+ status: "running",
85
+ finishedAt: undefined,
86
+ cases: [
87
+ { id: "smoke", fileId: "smoke", status: "running" },
88
+ { id: "other", fileId: "other", status: "pending" },
89
+ ],
90
+ };
91
+ const saved: EvalRunSnapshot[] = [];
92
+ const store = new EvalRunStore(dir, () => {}, {
93
+ save: (run) => {
94
+ saved.push(structuredClone(run));
95
+ },
96
+ delete: () => {},
97
+ list: async () => [interrupted],
98
+ });
99
+ await store.hydrate();
100
+
101
+ const listed = await store.listRuns();
102
+ expect(listed[0]?.status).toBe("failed");
103
+ expect(listed[0]?.cases.every((c) => c.status === "done")).toBe(true);
104
+ // The summary is reconciled with the repaired case rows (both cases
105
+ // were interrupted, so nothing passed) instead of keeping stale counts.
106
+ expect(listed[0]?.summary).toEqual({
107
+ passed: 0,
108
+ failed: 2,
109
+ scored: 0,
110
+ skipped: 0,
111
+ total: 2,
112
+ done: 2,
113
+ });
114
+ // The repaired snapshot was written back to storage.
115
+ expect(saved.map((run) => run.runId)).toContain("evalrun_interrupted");
116
+ expect(saved.at(-1)?.status).toBe("failed");
117
+ });
118
+
119
+ it("prunes hydrated history even when eval discovery fails", async () => {
120
+ dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
121
+ await mkdir(join(dir, "evals"), { recursive: true });
122
+ // Discovery throws at hydrate time; pruning must still run with the
123
+ // default playground-history window (20).
124
+ await writeFile(
125
+ join(dir, "evals", "broken.eval.js"),
126
+ "throw new Error('boom');\n",
127
+ "utf8"
128
+ );
129
+
130
+ const stored = Array.from({ length: 25 }, (_, i) => ({
131
+ ...completedRun(`evalrun_${String(i).padStart(2, "0")}`),
132
+ startedAt: `2026-01-01T00:${String(i).padStart(2, "0")}:00.000Z`,
133
+ }));
134
+ const deleted: string[] = [];
135
+ const store = new EvalRunStore(dir, () => {}, {
136
+ save: () => {},
137
+ delete: (runId) => {
138
+ deleted.push(runId);
139
+ },
140
+ list: async () => stored,
141
+ });
142
+ await store.hydrate();
143
+
144
+ const listed = await store.listRuns();
145
+ expect(listed).toHaveLength(20);
146
+ expect(deleted).toHaveLength(5);
147
+ // Oldest five (by startedAt) were pruned.
148
+ expect(deleted.sort()).toEqual([
149
+ "evalrun_00",
150
+ "evalrun_01",
151
+ "evalrun_02",
152
+ "evalrun_03",
153
+ "evalrun_04",
154
+ ]);
155
+ });
156
+
157
+ it("close() durably marks an in-flight batch as interrupted", async () => {
158
+ dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
159
+ await mkdir(join(dir, "evals"), { recursive: true });
160
+ await writeFile(
161
+ join(dir, "evals", "evals.config.js"),
162
+ "export default { maxConcurrency: 1 };\n",
163
+ "utf8"
164
+ );
165
+
166
+ const running: EvalRunSnapshot = {
167
+ ...completedRun("evalrun_active"),
168
+ status: "running",
169
+ finishedAt: undefined,
170
+ summary: { passed: 1, failed: 0, total: 2, done: 1 },
171
+ cases: [
172
+ { id: "done", fileId: "done", status: "done", ok: true },
173
+ { id: "midway", fileId: "midway", status: "running" },
174
+ ],
175
+ };
176
+ const saved: EvalRunSnapshot[] = [];
177
+ const store = new EvalRunStore(dir, () => {}, {
178
+ save: (run) => {
179
+ saved.push(structuredClone(run));
180
+ },
181
+ delete: () => {},
182
+ list: async () => [running],
183
+ });
184
+ await store.hydrate();
185
+ // hydrate already repaired it; reset to a live in-flight shape.
186
+ const run = (await store.listRuns())[0]!;
187
+ run.status = "running";
188
+ run.error = undefined;
189
+ run.cases[1]!.status = "running";
190
+ run.cases[1]!.ok = undefined;
191
+ run.cases[1]!.error = undefined;
192
+ (store as unknown as { activeRunId?: string }).activeRunId = run.runId;
193
+ saved.length = 0;
194
+
195
+ store.close();
196
+
197
+ expect(saved).toHaveLength(1);
198
+ expect(saved[0]?.status).toBe("failed");
199
+ expect(saved[0]?.summary).toEqual({
200
+ passed: 1,
201
+ failed: 1,
202
+ scored: 0,
203
+ skipped: 0,
204
+ total: 2,
205
+ done: 2,
206
+ });
207
+ // Idempotent: a second close has nothing left to write.
208
+ store.close();
209
+ expect(saved).toHaveLength(1);
210
+ });
211
+ });