@cursor/july 0.1.23 → 0.1.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/dist/bin/agent-serve.js +0 -0
  2. package/dist/docs/404.html +1 -1
  3. package/dist/docs/ab.html +2 -2
  4. package/dist/docs/assets/{app.DYcC9FY-.js → app.CvSGaAxk.js} +1 -1
  5. package/dist/docs/assets/chunks/@localSearchIndexroot.CL2Y0Zmh.js +1 -0
  6. package/dist/docs/assets/chunks/{VPLocalSearchBox.o4N_knTV.js → VPLocalSearchBox.J1jJbCvs.js} +1 -1
  7. package/dist/docs/assets/chunks/{theme.DQ-njyo0.js → theme.C7vfzr7h.js} +2 -2
  8. package/dist/docs/building-with-agents.html +2 -2
  9. package/dist/docs/concepts.html +2 -2
  10. package/dist/docs/deployment.html +2 -2
  11. package/dist/docs/evals.html +2 -2
  12. package/dist/docs/example-agents/approval-buddy.html +2 -2
  13. package/dist/docs/example-agents/benny.html +2 -2
  14. package/dist/docs/example-agents/bugbot.html +2 -2
  15. package/dist/docs/example-agents/codebase-wiki.html +2 -2
  16. package/dist/docs/example-agents/codeowners-review.html +2 -2
  17. package/dist/docs/example-agents/concierge.html +2 -2
  18. package/dist/docs/example-agents/fsd.html +2 -2
  19. package/dist/docs/example-agents/index.html +2 -2
  20. package/dist/docs/example-agents/knowledge-base.html +2 -2
  21. package/dist/docs/example-agents/oncall.html +2 -2
  22. package/dist/docs/example-agents/security-reviewer.html +2 -2
  23. package/dist/docs/example-agents/slack-agent.html +2 -2
  24. package/dist/docs/example-agents/weather-agent.html +2 -2
  25. package/dist/docs/guides/agent-to-agent.html +2 -2
  26. package/dist/docs/guides/cloud-runtime.html +2 -2
  27. package/dist/docs/guides/github.html +2 -2
  28. package/dist/docs/guides/human-in-the-loop.html +2 -2
  29. package/dist/docs/guides/mcp-oauth.html +2 -2
  30. package/dist/docs/guides/slack.html +2 -2
  31. package/dist/docs/guides/webhooks.html +2 -2
  32. package/dist/docs/hillclimbing.html +2 -2
  33. package/dist/docs/index.html +2 -2
  34. package/dist/docs/quickstart.html +2 -2
  35. package/dist/docs/reference/agent-config.html +2 -2
  36. package/dist/docs/reference/channels.html +2 -2
  37. package/dist/docs/reference/cli.html +2 -2
  38. package/dist/docs/reference/connections.html +2 -2
  39. package/dist/docs/reference/hooks.html +2 -2
  40. package/dist/docs/reference/http-api.html +2 -2
  41. package/dist/docs/reference/instructions.html +2 -2
  42. package/dist/docs/reference/playground.html +2 -2
  43. package/dist/docs/reference/project-layout.html +2 -2
  44. package/dist/docs/reference/prompt.html +2 -2
  45. package/dist/docs/reference/schedules.html +2 -2
  46. package/dist/docs/reference/sessions.html +2 -2
  47. package/dist/docs/reference/skills.html +2 -2
  48. package/dist/docs/reference/subagents.html +2 -2
  49. package/dist/docs/reference/tools.html +2 -2
  50. package/dist/docs/scaffolding-agents.html +2 -2
  51. package/dist/docs/storage.html +2 -2
  52. package/dist/docs/troubleshooting.html +2 -2
  53. package/dist/playground/assets/index-CidizGZv.css +1 -0
  54. package/dist/playground/assets/{index-DqXdAFGa.js → index-DTG9OsPV.js} +41 -41
  55. package/dist/playground/index.html +2 -2
  56. package/package.json +24 -24
  57. package/dist/channels/github/instrument.d.ts +0 -20
  58. package/dist/channels/github/instrument.d.ts.map +0 -1
  59. package/dist/docs/assets/chunks/@localSearchIndexroot.BQTzJjR_.js +0 -1
  60. package/dist/internal/json-dir-store.d.ts +0 -32
  61. package/dist/internal/json-dir-store.d.ts.map +0 -1
  62. package/dist/internal/json-dir-store.js +0 -100
  63. package/dist/playground/assets/index-CjOQ4hN9.css +0 -1
  64. package/src/bin/agent-serve.version.test.ts +0 -64
  65. package/src/channels/github/api.test.ts +0 -64
  66. package/src/channels/github/auth.test.ts +0 -105
  67. package/src/channels/github/cursor-account.test.ts +0 -204
  68. package/src/channels/github/forward.test.ts +0 -457
  69. package/src/channels/github/github.test.ts +0 -937
  70. package/src/channels/github/replay.test.ts +0 -179
  71. package/src/channels/slack/api.post-message.test.ts +0 -148
  72. package/src/channels/slack/approvals.test.ts +0 -328
  73. package/src/channels/slack/block-actions.test.ts +0 -452
  74. package/src/channels/slack/bot-mentions.test.ts +0 -267
  75. package/src/channels/slack/channel-watch.test.ts +0 -363
  76. package/src/channels/slack/cursor-account.test.ts +0 -253
  77. package/src/channels/slack/defaults.final-post.test.ts +0 -182
  78. package/src/channels/slack/dispatch.test.ts +0 -795
  79. package/src/channels/slack/eval-directive.test.ts +0 -273
  80. package/src/channels/slack/message-body.test.ts +0 -54
  81. package/src/channels/slack/nudge-store.test.ts +0 -143
  82. package/src/channels/slack/slack.test.ts +0 -391
  83. package/src/channels/slack/stop.test.ts +0 -23
  84. package/src/channels/slack/thread-context.test.ts +0 -202
  85. package/src/evals/assertions.test.ts +0 -580
  86. package/src/evals/expect.test.ts +0 -144
  87. package/src/evals/judge.test.ts +0 -181
  88. package/src/evals/loaders.test.ts +0 -132
  89. package/src/evals/matchers.test.ts +0 -95
  90. package/src/evals/reporters.test.ts +0 -303
  91. package/src/evals/run-facts.test.ts +0 -259
  92. package/src/internal/ab-snapshot.test.ts +0 -325
  93. package/src/internal/approval-gate.test.ts +0 -49
  94. package/src/internal/approvals.integration.test.ts +0 -383
  95. package/src/internal/authored-loaders.test.ts +0 -31
  96. package/src/internal/builtin-tools/reminders.test.ts +0 -201
  97. package/src/internal/channel-route-schema.test.ts +0 -294
  98. package/src/internal/chat-attach.test.ts +0 -262
  99. package/src/internal/cli-deploy.test.ts +0 -1991
  100. package/src/internal/cli-docs.test.ts +0 -161
  101. package/src/internal/cli-mcp.test.ts +0 -789
  102. package/src/internal/cli-skills.test.ts +0 -133
  103. package/src/internal/cli-slack.test.ts +0 -1647
  104. package/src/internal/cloud-merge.test.ts +0 -74
  105. package/src/internal/cron.test.ts +0 -22
  106. package/src/internal/cursor/account-mcp.test.ts +0 -807
  107. package/src/internal/cursor/backend-client.test.ts +0 -591
  108. package/src/internal/cursor/credentials.test.ts +0 -351
  109. package/src/internal/cursor/github-credentials.test.ts +0 -136
  110. package/src/internal/cursor-account-mcp-auth.test.ts +0 -310
  111. package/src/internal/cursor-account.integration.test.ts +0 -441
  112. package/src/internal/cursor-event-relay.test.ts +0 -746
  113. package/src/internal/cursor-github-credentials.integration.test.ts +0 -271
  114. package/src/internal/cursor-slack-relay.test.ts +0 -525
  115. package/src/internal/deploy-source.test.ts +0 -111
  116. package/src/internal/discovery.builtin-tools.test.ts +0 -94
  117. package/src/internal/discovery.concurrency.test.ts +0 -60
  118. package/src/internal/discovery.cursor-account.test.ts +0 -133
  119. package/src/internal/discovery.cwd.test.ts +0 -83
  120. package/src/internal/discovery.hosting.test.ts +0 -80
  121. package/src/internal/discovery.identity.test.ts +0 -44
  122. package/src/internal/docs-site.test.ts +0 -66
  123. package/src/internal/duration.test.ts +0 -29
  124. package/src/internal/eval-judge-model.test.ts +0 -187
  125. package/src/internal/eval-run-store.cancel.test.ts +0 -142
  126. package/src/internal/eval-run-store.storage.test.ts +0 -211
  127. package/src/internal/eval-runner.http.test.ts +0 -403
  128. package/src/internal/eval-runner.run.test.ts +0 -928
  129. package/src/internal/evals-client.test.ts +0 -307
  130. package/src/internal/event-mapper.test.ts +0 -243
  131. package/src/internal/github-fanout.test.ts +0 -213
  132. package/src/internal/handleAgentServeTrigger.test.ts +0 -179
  133. package/src/internal/host-kv.test.ts +0 -82
  134. package/src/internal/host-platforms.test.ts +0 -126
  135. package/src/internal/http-channel.test.ts +0 -402
  136. package/src/internal/init-project.test.ts +0 -270
  137. package/src/internal/install-cursor-skills.test.ts +0 -262
  138. package/src/internal/local-env.test.ts +0 -120
  139. package/src/internal/log-ring.test.ts +0 -31
  140. package/src/internal/logs-client.test.ts +0 -350
  141. package/src/internal/mcp-endpoint.test.ts +0 -436
  142. package/src/internal/mcp-host.test.ts +0 -298
  143. package/src/internal/mcp-oauth.test.ts +0 -148
  144. package/src/internal/net.test.ts +0 -17
  145. package/src/internal/peer-connections.test.ts +0 -128
  146. package/src/internal/peer-mcp.integration.test.ts +0 -289
  147. package/src/internal/playground/toolchain.test.ts +0 -53
  148. package/src/internal/playground-cli.test.ts +0 -187
  149. package/src/internal/playground-proxy.test.ts +0 -376
  150. package/src/internal/prompt-context.integration.test.ts +0 -232
  151. package/src/internal/prompt-context.test.ts +0 -127
  152. package/src/internal/reminder-runner.test.ts +0 -390
  153. package/src/internal/reminder-store.test.ts +0 -53
  154. package/src/internal/request-headers.test.ts +0 -27
  155. package/src/internal/resolve-prod-target.test.ts +0 -787
  156. package/src/internal/resolved-connections.test.ts +0 -295
  157. package/src/internal/router.test.ts +0 -57
  158. package/src/internal/sdk-runner.test.ts +0 -290
  159. package/src/internal/session-engine.coalesce.test.ts +0 -169
  160. package/src/internal/session-engine.concurrency.test.ts +0 -250
  161. package/src/internal/session-engine.host-oauth-mcp.test.ts +0 -110
  162. package/src/internal/session-engine.interrupt.test.ts +0 -577
  163. package/src/internal/session-engine.storage.test.ts +0 -547
  164. package/src/internal/session-urls.test.ts +0 -28
  165. package/src/internal/sessions-client.test.ts +0 -518
  166. package/src/internal/storage-coordinator.test.ts +0 -517
  167. package/src/internal/tool-call.test.ts +0 -458
  168. package/src/internal/tool-result.test.ts +0 -52
  169. package/src/internal/trajectory.approvals.test.ts +0 -83
  170. package/src/internal/trajectory.subagents.test.ts +0 -198
  171. package/src/internal/turn-governor.test.ts +0 -137
  172. package/src/internal/update-check.test.ts +0 -485
  173. package/src/internal/workspace.test.ts +0 -207
  174. package/src/storage-backends/cursor-hosted.test.ts +0 -121
@@ -1,187 +0,0 @@
1
- import { beforeEach, describe, expect, it, vi } from "vitest";
2
- import { EvalJudgeUnavailableError } from "../evals/judge.js";
3
- import {
4
- assertJudgeBackendMatch,
5
- callJudgeModel,
6
- resolveJudgeModel,
7
- } from "./eval-judge-model.js";
8
-
9
- const create = vi.hoisted(() => vi.fn());
10
- const resolveApiKey = vi.hoisted(() => vi.fn());
11
-
12
- vi.mock("@cursor/sdk", () => ({ Agent: { create } }));
13
-
14
- vi.mock("./cursor/credentials.js", async (importOriginal) => {
15
- const actual =
16
- await importOriginal<typeof import("./cursor/credentials.js")>();
17
- return { ...actual, resolveApiKey };
18
- });
19
-
20
- beforeEach(() => {
21
- create.mockReset();
22
- resolveApiKey.mockReset();
23
- resolveApiKey.mockResolvedValue({
24
- apiKey: "key_1",
25
- source: "env",
26
- backendUrl: "https://api2.cursor.sh",
27
- });
28
- create.mockResolvedValue({
29
- send: async () => ({
30
- wait: async () => ({ status: "completed", result: "CHOICE: Y" }),
31
- }),
32
- close: () => undefined,
33
- });
34
- });
35
-
36
- describe("resolveJudgeModel", () => {
37
- it("resolves innermost-first and ignores a blank env value", () => {
38
- expect(
39
- resolveJudgeModel({
40
- call: "call-model",
41
- evalLevel: "eval-model",
42
- config: "config-model",
43
- env: "env-model",
44
- })
45
- ).toBe("call-model");
46
- expect(
47
- resolveJudgeModel({ evalLevel: "eval-model", config: "config-model" })
48
- ).toBe("eval-model");
49
- expect(resolveJudgeModel({ config: "config-model" })).toBe("config-model");
50
- expect(resolveJudgeModel({ env: " env-model " })).toBe("env-model");
51
- expect(resolveJudgeModel({ env: " " })).toBeUndefined();
52
- expect(resolveJudgeModel({})).toBeUndefined();
53
- });
54
- });
55
-
56
- describe("assertJudgeBackendMatch", () => {
57
- it("allows a credential scoped to the backend the SDK will use", () => {
58
- expect(() =>
59
- assertJudgeBackendMatch(
60
- "https://api2.cursor.sh",
61
- "https://api2.cursor.sh"
62
- )
63
- ).not.toThrow();
64
- });
65
-
66
- it("ignores a trailing-slash difference", () => {
67
- expect(() =>
68
- assertJudgeBackendMatch(
69
- "https://api2.cursor.sh/",
70
- "https://api2.cursor.sh"
71
- )
72
- ).not.toThrow();
73
- });
74
-
75
- it("declines when the credential belongs to a different backend", () => {
76
- // The SDK takes no backend option, so sending the key anyway would ship it
77
- // (and the judge prompt) to a host it was not minted for.
78
- expect(() =>
79
- assertJudgeBackendMatch(
80
- "https://staging.example.com",
81
- "https://api2.cursor.sh"
82
- )
83
- ).toThrow(EvalJudgeUnavailableError);
84
- expect(() =>
85
- assertJudgeBackendMatch(
86
- "https://staging.example.com",
87
- "https://api2.cursor.sh"
88
- )
89
- ).toThrow(/does not read AGENT_SERVE_CURSOR_BACKEND_URL/);
90
- });
91
- });
92
-
93
- describe("callJudgeModel", () => {
94
- it("skips visibly when there are no credentials", async () => {
95
- resolveApiKey.mockResolvedValue(undefined);
96
- await expect(
97
- callJudgeModel({ prompt: "grade this", model: "gpt-5.4-mini" })
98
- ).rejects.toThrow(EvalJudgeUnavailableError);
99
- expect(create).not.toHaveBeenCalled();
100
- });
101
-
102
- it("confines the grading session so an injected reply cannot reach the host", async () => {
103
- await callJudgeModel({ prompt: "grade this", model: "gpt-5.4-mini" });
104
-
105
- expect(create).toHaveBeenCalledTimes(1);
106
- const options = create.mock.calls[0]![0] as {
107
- mcpServers: Record<string, unknown>;
108
- local: {
109
- cwd: string;
110
- settingSources: string[];
111
- sandboxOptions: { enabled: boolean };
112
- };
113
- };
114
- // Grading embeds untrusted agent output, so the judge gets no MCP surface,
115
- // no ambient settings, an empty scratch cwd, and a sandbox.
116
- expect(options.mcpServers).toEqual({});
117
- expect(options.local.settingSources).toEqual([]);
118
- expect(options.local.sandboxOptions).toEqual({ enabled: true });
119
- expect(options.local.cwd).toMatch(/agentkit-eval-judge-/);
120
- });
121
-
122
- it("never opens a session when the credential's backend disagrees", async () => {
123
- resolveApiKey.mockResolvedValue({
124
- apiKey: "key_1",
125
- source: "login",
126
- backendUrl: "https://staging.example.com",
127
- });
128
- await expect(
129
- callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
130
- ).rejects.toThrow(EvalJudgeUnavailableError);
131
- // The point of failing closed: the key never reaches the SDK.
132
- expect(create).not.toHaveBeenCalled();
133
- });
134
-
135
- it("passes a bare model id through as a selection object", async () => {
136
- await callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" });
137
- expect(create.mock.calls[0]![0]).toMatchObject({
138
- apiKey: "key_1",
139
- model: { id: "gpt-5.4-mini" },
140
- });
141
- });
142
-
143
- it("preserves model params when given a selection object", async () => {
144
- await callJudgeModel({
145
- prompt: "p",
146
- model: { id: "gpt-5.4-mini", params: [{ id: "effort", value: "low" }] },
147
- });
148
- expect(create.mock.calls[0]![0]).toMatchObject({
149
- model: { id: "gpt-5.4-mini", params: [{ id: "effort", value: "low" }] },
150
- });
151
- });
152
-
153
- it("returns the judge reply", async () => {
154
- await expect(
155
- callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
156
- ).resolves.toBe("CHOICE: Y");
157
- });
158
-
159
- it("surfaces a failed grading run as an error", async () => {
160
- create.mockResolvedValue({
161
- send: async () => ({
162
- wait: async () => ({
163
- status: "error",
164
- error: { message: "provider unavailable" },
165
- }),
166
- }),
167
- close: () => undefined,
168
- });
169
- await expect(
170
- callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
171
- ).rejects.toThrow(/eval judge run failed: provider unavailable/);
172
- });
173
-
174
- it("closes the session even when grading throws", async () => {
175
- const close = vi.fn();
176
- create.mockResolvedValue({
177
- send: async () => {
178
- throw new Error("boom");
179
- },
180
- close,
181
- });
182
- await expect(
183
- callJudgeModel({ prompt: "p", model: "gpt-5.4-mini" })
184
- ).rejects.toThrow("boom");
185
- expect(close).toHaveBeenCalledTimes(1);
186
- });
187
- });
@@ -1,142 +0,0 @@
1
- import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
2
- import { tmpdir } from "node:os";
3
- import { join } from "node:path";
4
- import { afterEach, describe, expect, it, vi } from "vitest";
5
- import {
6
- EvalRunNotRunningError,
7
- EvalRunStore,
8
- EvalRunUnknownError,
9
- } from "./eval-run-store.js";
10
-
11
- vi.mock("./eval-runner.js", async (importOriginal) => {
12
- const actual = await importOriginal<typeof import("./eval-runner.js")>();
13
- return {
14
- ...actual,
15
- runDiscoveredEvals: vi.fn(
16
- async (options: {
17
- signal?: AbortSignal;
18
- onCaseStart?: (id: string) => void;
19
- onCaseDone?: (result: {
20
- id: string;
21
- path: string;
22
- ok: boolean;
23
- assertions: [];
24
- logs: string[];
25
- metrics: Record<string, string | number>;
26
- inputs: string[];
27
- toolCalls: [];
28
- tools: string[];
29
- durationMs: number;
30
- }) => void;
31
- discovered?: {
32
- evals: Array<{ id: string; path: string }>;
33
- };
34
- }) => {
35
- const first = options.discovered?.evals[0];
36
- if (first !== undefined) {
37
- options.onCaseStart?.(first.id);
38
- }
39
- await new Promise<void>((resolve, reject) => {
40
- const onAbort = (): void => {
41
- options.signal?.removeEventListener("abort", onAbort);
42
- reject(new Error("cancelled"));
43
- };
44
- if (options.signal?.aborted) {
45
- onAbort();
46
- return;
47
- }
48
- options.signal?.addEventListener("abort", onAbort, { once: true });
49
- // Keep the batch "running" until cancel aborts.
50
- setTimeout(() => {
51
- options.signal?.removeEventListener("abort", onAbort);
52
- if (first !== undefined) {
53
- options.onCaseDone?.({
54
- id: first.id,
55
- path: first.path,
56
- ok: true,
57
- assertions: [],
58
- logs: [],
59
- metrics: {},
60
- inputs: [],
61
- toolCalls: [],
62
- tools: [],
63
- durationMs: 1,
64
- });
65
- }
66
- resolve();
67
- }, 5_000);
68
- });
69
- return [];
70
- }
71
- ),
72
- };
73
- });
74
-
75
- describe("EvalRunStore.cancel", () => {
76
- let dir: string | undefined;
77
-
78
- afterEach(async () => {
79
- if (dir !== undefined) {
80
- await rm(dir, { recursive: true, force: true });
81
- dir = undefined;
82
- }
83
- vi.clearAllMocks();
84
- });
85
-
86
- it("cancels a running batch by Eval ID and clears the active run", async () => {
87
- dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-cancel-"));
88
- await mkdir(join(dir, "evals"), { recursive: true });
89
- await writeFile(
90
- join(dir, "evals", "evals.config.js"),
91
- "export default { maxConcurrency: 1 };\n",
92
- "utf8"
93
- );
94
- await writeFile(
95
- join(dir, "evals", "smoke.eval.js"),
96
- `export default {
97
- __agentServe: "eval",
98
- async test() {}
99
- };
100
- `,
101
- "utf8"
102
- );
103
-
104
- const store = new EvalRunStore(dir, () => {});
105
- store.loopbackUrl = "http://127.0.0.1:9/smoke";
106
- const started = await store.start();
107
- expect(started.status).toBe("running");
108
- expect(store.getActiveRunId()).toBe(started.runId);
109
-
110
- const cancelled = await store.cancel(started.runId);
111
- expect(cancelled.status).toBe("cancelled");
112
- expect(cancelled.error).toContain("cancelled");
113
- expect(store.getActiveRunId()).toBeUndefined();
114
- expect(await store.cancel(started.runId)).toMatchObject({
115
- status: "cancelled",
116
- });
117
- await expect(store.cancel("evalrun_missing")).rejects.toBeInstanceOf(
118
- EvalRunUnknownError
119
- );
120
- await expect(store.cancel(started.runId)).resolves.toMatchObject({
121
- status: "cancelled",
122
- });
123
-
124
- // After cancel, a completed-looking batch cannot be cancelled again as running.
125
- const completedId = started.runId;
126
- const snap = await store.get(completedId);
127
- expect(snap?.status).toBe("cancelled");
128
- // Force a non-running status to exercise NotRunningError via a fresh id.
129
- const fakeId = "evalrun_done_only";
130
- (
131
- store as unknown as {
132
- runs: Map<string, { runId: string; status: string }>;
133
- }
134
- ).runs.set(fakeId, {
135
- runId: fakeId,
136
- status: "completed",
137
- });
138
- await expect(store.cancel(fakeId)).rejects.toBeInstanceOf(
139
- EvalRunNotRunningError
140
- );
141
- });
142
- });
@@ -1,211 +0,0 @@
1
- import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
2
- import { tmpdir } from "node:os";
3
- import { join } from "node:path";
4
- import { afterEach, describe, expect, it } from "vitest";
5
- import type { EvalRunSnapshot } from "../evals.js";
6
- import { defineStorage } from "../storage.js";
7
- import { EvalRunStore } from "./eval-run-store.js";
8
- import { StorageCoordinator } from "./storage-coordinator.js";
9
-
10
- function completedRun(runId: string): EvalRunSnapshot {
11
- return {
12
- runId,
13
- status: "completed",
14
- startedAt: "2026-01-01T00:00:00.000Z",
15
- finishedAt: "2026-01-01T00:00:01.000Z",
16
- summary: { passed: 1, failed: 0, total: 1, done: 1 },
17
- cases: [],
18
- config: { maxPlaygroundRuns: 20, durableRuns: true },
19
- };
20
- }
21
-
22
- describe("EvalRunStore defineStorage hydrate", () => {
23
- let dir: string | undefined;
24
-
25
- afterEach(async () => {
26
- if (dir !== undefined) {
27
- await rm(dir, { recursive: true, force: true });
28
- dir = undefined;
29
- }
30
- });
31
-
32
- it("hydrates playground history from the defineStorage evals table", async () => {
33
- dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
34
- await mkdir(join(dir, "evals"), { recursive: true });
35
- await writeFile(
36
- join(dir, "evals", "evals.config.js"),
37
- "export default { maxConcurrency: 1 };\n",
38
- "utf8"
39
- );
40
-
41
- const table = new Map<string, EvalRunSnapshot>();
42
- const coordinator = new StorageCoordinator({
43
- definition: defineStorage({
44
- put: () => {},
45
- evals: {
46
- put: (run) => {
47
- table.set(run.runId, run);
48
- },
49
- delete: (runId) => {
50
- table.delete(runId);
51
- },
52
- list: () => [...table.values()],
53
- },
54
- }),
55
- agentName: "test-agent",
56
- projectRoot: dir,
57
- logger: () => {},
58
- });
59
-
60
- const storage = coordinator.evalRuns();
61
- expect(storage).toBeDefined();
62
- storage!.save(completedRun("evalrun_from_sink"));
63
- await coordinator.whenIdle();
64
-
65
- const store = new EvalRunStore(dir, () => {}, coordinator.evalRuns());
66
- await store.hydrate();
67
- const listed = await store.listRuns();
68
- expect(listed).toHaveLength(1);
69
- expect(listed[0]?.runId).toBe("evalrun_from_sink");
70
- expect(listed[0]?.status).toBe("completed");
71
- });
72
-
73
- it("marks interrupted running batches as failed on hydrate", async () => {
74
- dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
75
- await mkdir(join(dir, "evals"), { recursive: true });
76
- await writeFile(
77
- join(dir, "evals", "evals.config.js"),
78
- "export default { maxConcurrency: 1 };\n",
79
- "utf8"
80
- );
81
-
82
- const interrupted: EvalRunSnapshot = {
83
- ...completedRun("evalrun_interrupted"),
84
- status: "running",
85
- finishedAt: undefined,
86
- cases: [
87
- { id: "smoke", fileId: "smoke", status: "running" },
88
- { id: "other", fileId: "other", status: "pending" },
89
- ],
90
- };
91
- const saved: EvalRunSnapshot[] = [];
92
- const store = new EvalRunStore(dir, () => {}, {
93
- save: (run) => {
94
- saved.push(structuredClone(run));
95
- },
96
- delete: () => {},
97
- list: async () => [interrupted],
98
- });
99
- await store.hydrate();
100
-
101
- const listed = await store.listRuns();
102
- expect(listed[0]?.status).toBe("failed");
103
- expect(listed[0]?.cases.every((c) => c.status === "done")).toBe(true);
104
- // The summary is reconciled with the repaired case rows (both cases
105
- // were interrupted, so nothing passed) instead of keeping stale counts.
106
- expect(listed[0]?.summary).toEqual({
107
- passed: 0,
108
- failed: 2,
109
- scored: 0,
110
- skipped: 0,
111
- total: 2,
112
- done: 2,
113
- });
114
- // The repaired snapshot was written back to storage.
115
- expect(saved.map((run) => run.runId)).toContain("evalrun_interrupted");
116
- expect(saved.at(-1)?.status).toBe("failed");
117
- });
118
-
119
- it("prunes hydrated history even when eval discovery fails", async () => {
120
- dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
121
- await mkdir(join(dir, "evals"), { recursive: true });
122
- // Discovery throws at hydrate time; pruning must still run with the
123
- // default playground-history window (20).
124
- await writeFile(
125
- join(dir, "evals", "broken.eval.js"),
126
- "throw new Error('boom');\n",
127
- "utf8"
128
- );
129
-
130
- const stored = Array.from({ length: 25 }, (_, i) => ({
131
- ...completedRun(`evalrun_${String(i).padStart(2, "0")}`),
132
- startedAt: `2026-01-01T00:${String(i).padStart(2, "0")}:00.000Z`,
133
- }));
134
- const deleted: string[] = [];
135
- const store = new EvalRunStore(dir, () => {}, {
136
- save: () => {},
137
- delete: (runId) => {
138
- deleted.push(runId);
139
- },
140
- list: async () => stored,
141
- });
142
- await store.hydrate();
143
-
144
- const listed = await store.listRuns();
145
- expect(listed).toHaveLength(20);
146
- expect(deleted).toHaveLength(5);
147
- // Oldest five (by startedAt) were pruned.
148
- expect(deleted.sort()).toEqual([
149
- "evalrun_00",
150
- "evalrun_01",
151
- "evalrun_02",
152
- "evalrun_03",
153
- "evalrun_04",
154
- ]);
155
- });
156
-
157
- it("close() durably marks an in-flight batch as interrupted", async () => {
158
- dir = await mkdtemp(join(tmpdir(), "agent-serve-eval-storage-"));
159
- await mkdir(join(dir, "evals"), { recursive: true });
160
- await writeFile(
161
- join(dir, "evals", "evals.config.js"),
162
- "export default { maxConcurrency: 1 };\n",
163
- "utf8"
164
- );
165
-
166
- const running: EvalRunSnapshot = {
167
- ...completedRun("evalrun_active"),
168
- status: "running",
169
- finishedAt: undefined,
170
- summary: { passed: 1, failed: 0, total: 2, done: 1 },
171
- cases: [
172
- { id: "done", fileId: "done", status: "done", ok: true },
173
- { id: "midway", fileId: "midway", status: "running" },
174
- ],
175
- };
176
- const saved: EvalRunSnapshot[] = [];
177
- const store = new EvalRunStore(dir, () => {}, {
178
- save: (run) => {
179
- saved.push(structuredClone(run));
180
- },
181
- delete: () => {},
182
- list: async () => [running],
183
- });
184
- await store.hydrate();
185
- // hydrate already repaired it; reset to a live in-flight shape.
186
- const run = (await store.listRuns())[0]!;
187
- run.status = "running";
188
- run.error = undefined;
189
- run.cases[1]!.status = "running";
190
- run.cases[1]!.ok = undefined;
191
- run.cases[1]!.error = undefined;
192
- (store as unknown as { activeRunId?: string }).activeRunId = run.runId;
193
- saved.length = 0;
194
-
195
- store.close();
196
-
197
- expect(saved).toHaveLength(1);
198
- expect(saved[0]?.status).toBe("failed");
199
- expect(saved[0]?.summary).toEqual({
200
- passed: 1,
201
- failed: 1,
202
- scored: 0,
203
- skipped: 0,
204
- total: 2,
205
- done: 2,
206
- });
207
- // Idempotent: a second close has nothing left to write.
208
- store.close();
209
- expect(saved).toHaveLength(1);
210
- });
211
- });