@haystackeditor/cli 0.15.30 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. package/README.md +111 -60
  2. package/dist/assets/hooks/scripts/commit-msg.sh +3 -0
  3. package/dist/assets/hooks/scripts/post-commit.sh +3 -0
  4. package/dist/assets/hooks/scripts/pre-commit.sh +11 -6
  5. package/dist/assets/hooks/scripts/pre-push.sh +3 -0
  6. package/dist/assets/hooks/scripts/prepare-commit-msg.sh +3 -0
  7. package/dist/commands/case-batch-contract.js +608 -0
  8. package/dist/commands/case-batch.js +713 -0
  9. package/dist/commands/cloud-verifier-identity-census.js +5 -2
  10. package/dist/commands/combination-search-hook.js +139 -0
  11. package/dist/commands/design-verify.js +27 -1
  12. package/dist/commands/dismiss.js +2 -1
  13. package/dist/commands/hooks.js +66 -7
  14. package/dist/commands/install-session-hooks.js +131 -44
  15. package/dist/commands/policy.js +32 -45
  16. package/dist/commands/precompute-delivery.js +6 -1
  17. package/dist/commands/scaffold-provisional-universe.js +8 -10
  18. package/dist/commands/setup.js +32 -10
  19. package/dist/commands/submit.js +7 -2
  20. package/dist/commands/telemetry.js +17 -2
  21. package/dist/commands/triage.js +2 -1
  22. package/dist/commands/verify-explore.js +1 -0
  23. package/dist/commands/verify-history.js +152 -0
  24. package/dist/commands/verify-hosted-mcp.js +3 -12
  25. package/dist/commands/verify-hosted-reproducibility.js +49 -547
  26. package/dist/commands/verify-hosted.js +180 -163
  27. package/dist/commands/verify-precompute.js +57 -10
  28. package/dist/commands/verify.js +51 -139
  29. package/dist/index.js +264 -348
  30. package/dist/lazy.js +8 -0
  31. package/dist/schema.js +1 -0
  32. package/dist/tools/detect.js +3 -24
  33. package/dist/triage/prompts.js +46 -45
  34. package/dist/triage/runner.js +119 -33
  35. package/dist/types/verify-history.js +1 -0
  36. package/dist/types.js +3 -3
  37. package/dist/utils/auth.js +14 -2
  38. package/dist/utils/design-verifier-api.js +23 -0
  39. package/dist/utils/design-verifier-history.js +79 -0
  40. package/dist/utils/git.js +60 -29
  41. package/dist/utils/github-api.js +14 -1
  42. package/dist/utils/haystack-api.js +26 -7
  43. package/dist/utils/hooks.js +43 -6
  44. package/dist/utils/prompter.js +25 -10
  45. package/dist/utils/safe-write.js +31 -0
  46. package/dist/utils/secret-paths.js +115 -0
  47. package/dist/utils/secrets.js +0 -1
  48. package/dist/utils/telemetry.js +34 -14
  49. package/dist/utils/update-check.js +151 -0
  50. package/package.json +18 -14
  51. package/schemas/case-batch.v1.json +184 -0
  52. package/schemas/cloud-verifier.v1.json +55 -100
  53. package/dist/commands/ask.d.ts +0 -14
  54. package/dist/commands/cloud-verifier-behaviors.d.ts +0 -27
  55. package/dist/commands/cloud-verifier-data-store-census.d.ts +0 -47
  56. package/dist/commands/cloud-verifier-data-store-drift.d.ts +0 -42
  57. package/dist/commands/cloud-verifier-identity-census.d.ts +0 -88
  58. package/dist/commands/cloud-verifier-materialization.d.ts +0 -16
  59. package/dist/commands/cloud-verifier-pascal-selector-census.d.ts +0 -29
  60. package/dist/commands/cloud-verifier-python-manifest-selector-census.d.ts +0 -27
  61. package/dist/commands/cloud-verifier-specialized-operational-census.d.ts +0 -51
  62. package/dist/commands/cloud-verifier-universe.d.ts +0 -31
  63. package/dist/commands/config.d.ts +0 -46
  64. package/dist/commands/design-verify.d.ts +0 -31
  65. package/dist/commands/dismiss.d.ts +0 -29
  66. package/dist/commands/hooks.d.ts +0 -13
  67. package/dist/commands/inbox.d.ts +0 -65
  68. package/dist/commands/init.d.ts +0 -10
  69. package/dist/commands/install-session-hooks.d.ts +0 -17
  70. package/dist/commands/login.d.ts +0 -8
  71. package/dist/commands/mcp.d.ts +0 -1
  72. package/dist/commands/policy.d.ts +0 -31
  73. package/dist/commands/pr-status.d.ts +0 -144
  74. package/dist/commands/pr.d.ts +0 -40
  75. package/dist/commands/precompute-delivery-contract.d.ts +0 -68
  76. package/dist/commands/precompute-delivery.d.ts +0 -20
  77. package/dist/commands/prepare-universe-review.d.ts +0 -115
  78. package/dist/commands/production-source-deny-policy.d.ts +0 -15
  79. package/dist/commands/request-review.d.ts +0 -26
  80. package/dist/commands/review.d.ts +0 -25
  81. package/dist/commands/rules.d.ts +0 -4
  82. package/dist/commands/scaffold-provisional-universe.d.ts +0 -468
  83. package/dist/commands/schema-cmd.d.ts +0 -2
  84. package/dist/commands/setup.d.ts +0 -28
  85. package/dist/commands/skills.d.ts +0 -8
  86. package/dist/commands/status.d.ts +0 -4
  87. package/dist/commands/submit.d.ts +0 -30
  88. package/dist/commands/system-map.d.ts +0 -42
  89. package/dist/commands/telemetry.d.ts +0 -53
  90. package/dist/commands/tokens.d.ts +0 -14
  91. package/dist/commands/triage.d.ts +0 -35
  92. package/dist/commands/verify-core.d.ts +0 -449
  93. package/dist/commands/verify-core.js +0 -789
  94. package/dist/commands/verify-explore.d.ts +0 -14
  95. package/dist/commands/verify-hosted-mcp.d.ts +0 -12
  96. package/dist/commands/verify-hosted-reproducibility.d.ts +0 -90
  97. package/dist/commands/verify-hosted.d.ts +0 -92
  98. package/dist/commands/verify-mcp.d.ts +0 -8
  99. package/dist/commands/verify-mcp.js +0 -517
  100. package/dist/commands/verify-ops.d.ts +0 -158
  101. package/dist/commands/verify-ops.js +0 -1148
  102. package/dist/commands/verify-precompute.d.ts +0 -30
  103. package/dist/commands/verify-reproducibility.d.ts +0 -85
  104. package/dist/commands/verify-reproducibility.js +0 -494
  105. package/dist/commands/verify-reseal.d.ts +0 -9
  106. package/dist/commands/verify-reseal.js +0 -148
  107. package/dist/commands/verify-sandboxes.d.ts +0 -95
  108. package/dist/commands/verify-sandboxes.js +0 -352
  109. package/dist/commands/verify.d.ts +0 -28
  110. package/dist/commands/webhooks.d.ts +0 -30
  111. package/dist/index.d.ts +0 -22
  112. package/dist/schema.d.ts +0 -28
  113. package/dist/states.d.ts +0 -29
  114. package/dist/tools/detect.d.ts +0 -50
  115. package/dist/triage/prompts.d.ts +0 -24
  116. package/dist/triage/runner.d.ts +0 -34
  117. package/dist/triage/types.d.ts +0 -42
  118. package/dist/types.d.ts +0 -1684
  119. package/dist/utils/action-output.d.ts +0 -24
  120. package/dist/utils/analysis-api.d.ts +0 -187
  121. package/dist/utils/auth.d.ts +0 -79
  122. package/dist/utils/config.d.ts +0 -24
  123. package/dist/utils/design-verifier-api.d.ts +0 -200
  124. package/dist/utils/design-verifier-result.d.ts +0 -74
  125. package/dist/utils/detect.d.ts +0 -43
  126. package/dist/utils/git.d.ts +0 -135
  127. package/dist/utils/github-api.d.ts +0 -104
  128. package/dist/utils/haystack-api.d.ts +0 -37
  129. package/dist/utils/hooks.d.ts +0 -12
  130. package/dist/utils/pending-state.d.ts +0 -40
  131. package/dist/utils/pr-ref.d.ts +0 -27
  132. package/dist/utils/prompter.d.ts +0 -85
  133. package/dist/utils/secrets.d.ts +0 -47
  134. package/dist/utils/telemetry.d.ts +0 -19
  135. /package/dist/commands/{precompute-delivery-worker.d.ts → combination-search-hook-contract.js} +0 -0
@@ -1,789 +0,0 @@
1
- // Shared plumbing for the cloud-verifier operator surface: `haystack verify
2
- // status|watch|list|show|artifacts` and the `haystack verify mcp` tools.
3
- //
4
- // Run state lives in two places with different strengths. The control plane
5
- // (Vite middleware) resolves aliases and knows whether the orchestrator that
6
- // owns a run is still alive — a run.json can sit at `running` forever after a
7
- // dev-server restart, and only the server reports that honestly. Disk
8
- // (.haystack/verify/runs/<runId>/run.json) survives the server being down and
9
- // is the only way to enumerate runs. So: server first, disk fallback, and a
10
- // disk-read of a non-terminal run is reported as stale rather than live.
11
- //
12
- // Types here are a hand-maintained mirror of src/features/cloud-verifier/types.ts
13
- // (the CLI ships standalone and cannot import the web app's sources), same as
14
- // the SPECIMENS list in verify.ts.
15
- import { createHash } from 'node:crypto';
16
- import * as fs from 'node:fs';
17
- import * as path from 'node:path';
18
- export const DEFAULT_CONTROL_PLANE = 'http://127.0.0.1:3000';
19
- export const TERMINAL_RUN_STATUSES = new Set(['complete', 'failed']);
20
- // ---------- locating things ----------
21
- export function resolveServer(explicit) {
22
- return (explicit || process.env.HAYSTACK_VERIFY_SERVER || DEFAULT_CONTROL_PLANE).replace(/\/+$/, '');
23
- }
24
- // ---------- auth ----------
25
- //
26
- // The demo control plane is open on localhost; a shared or cloud control plane
27
- // sets HAYSTACK_VERIFY_AUTH_TOKEN and rejects unauthenticated requests. The
28
- // client sends its token (--token / HAYSTACK_VERIFY_TOKEN) as a Bearer header
29
- // on every control-plane request. The token is only ever sent to the server
30
- // the operator explicitly chose — never inferred from product login state.
31
- let explicitVerifyToken = null;
32
- export function setVerifyToken(token) {
33
- explicitVerifyToken = token || null;
34
- }
35
- export function verifyAuthHeaders() {
36
- const token = explicitVerifyToken ?? process.env.HAYSTACK_VERIFY_TOKEN ?? null;
37
- return token ? { authorization: `Bearer ${token}` } : {};
38
- }
39
- export class VerifyUnauthorizedError extends Error {
40
- constructor(server) {
41
- super(`Control plane at ${server} requires a token (HTTP 401). Pass --token or set HAYSTACK_VERIFY_TOKEN to the value of the server's HAYSTACK_VERIFY_AUTH_TOKEN.`);
42
- }
43
- }
44
- /**
45
- * Find the repository that holds .haystack/verify — explicit flag, then
46
- * HAYSTACK_VERIFY_REPO, then walking up from cwd. Null when nothing matches;
47
- * server-backed commands still work without it.
48
- */
49
- export function discoverVerifyRoot(explicit) {
50
- const candidates = [explicit, process.env.HAYSTACK_VERIFY_REPO].filter((v) => Boolean(v));
51
- for (const candidate of candidates) {
52
- const resolved = path.resolve(candidate);
53
- if (fs.existsSync(path.join(resolved, '.haystack', 'verify')))
54
- return resolved;
55
- // A configured root is an instruction, not a hint: report it unusable
56
- // rather than silently walking somewhere else.
57
- throw new Error(`${resolved} has no .haystack/verify directory`);
58
- }
59
- let dir = process.cwd();
60
- for (;;) {
61
- if (fs.existsSync(path.join(dir, '.haystack', 'verify')))
62
- return dir;
63
- const parent = path.dirname(dir);
64
- if (parent === dir)
65
- return null;
66
- dir = parent;
67
- }
68
- }
69
- export function runsRoot(repoRoot) {
70
- return path.join(repoRoot, '.haystack', 'verify', 'runs');
71
- }
72
- export function runDir(repoRoot, runId) {
73
- return path.join(runsRoot(repoRoot), runId);
74
- }
75
- function isMissingFileError(error) {
76
- return typeof error === 'object'
77
- && error !== null
78
- && 'code' in error
79
- && error.code === 'ENOENT';
80
- }
81
- function runRecordReadError(runJsonPath, error) {
82
- return new Error(`Could not read verifier run record ${runJsonPath}: ${error instanceof Error ? error.message : String(error)}`, { cause: error });
83
- }
84
- /**
85
- * Read when a run began from the record itself. updatedAt and mtime both move
86
- * when cleanup or a migration rewrites old runs and therefore cannot define
87
- * "latest".
88
- */
89
- export function runStartedAtMs(runJsonPath) {
90
- try {
91
- const run = JSON.parse(fs.readFileSync(runJsonPath, 'utf8'));
92
- if (typeof run.createdAt === 'string') {
93
- const recorded = Date.parse(run.createdAt);
94
- if (Number.isFinite(recorded))
95
- return recorded;
96
- }
97
- }
98
- catch (error) {
99
- if (isMissingFileError(error))
100
- return null;
101
- throw runRecordReadError(runJsonPath, error);
102
- }
103
- return null;
104
- }
105
- export function listRunDirs(repoRoot) {
106
- let entries;
107
- try {
108
- entries = fs.readdirSync(runsRoot(repoRoot));
109
- }
110
- catch {
111
- return [];
112
- }
113
- return entries
114
- .filter(name => name.startsWith('run-'))
115
- .map(name => ({
116
- runId: name,
117
- startedAt: runStartedAtMs(path.join(runsRoot(repoRoot), name, 'run.json')),
118
- }))
119
- .filter((entry) => (entry.startedAt !== null))
120
- .sort((a, b) => b.startedAt - a.startedAt);
121
- }
122
- /** Same alias rules as the control plane: run id verbatim, `latest`, or a chapter id. */
123
- export function resolveRunIdOnDisk(repoRoot, requested) {
124
- if (requested.startsWith('run-'))
125
- return requested;
126
- const prefix = requested === 'latest' ? 'run-' : `run-${requested}-`;
127
- const match = listRunDirs(repoRoot).find(entry => entry.runId.startsWith(prefix));
128
- return match?.runId ?? requested;
129
- }
130
- export function loadRunFromDisk(repoRoot, runId) {
131
- const runJsonPath = path.join(runDir(repoRoot, runId), 'run.json');
132
- try {
133
- return JSON.parse(fs.readFileSync(runJsonPath, 'utf8'));
134
- }
135
- catch (error) {
136
- if (isMissingFileError(error))
137
- return null;
138
- throw runRecordReadError(runJsonPath, error);
139
- }
140
- }
141
- export async function fetchRunFromServer(server, idOrAlias) {
142
- let response;
143
- try {
144
- response = await fetch(`${server}/api/cloud-verifier/runs/${encodeURIComponent(idOrAlias)}`, {
145
- headers: verifyAuthHeaders(),
146
- });
147
- }
148
- catch (error) {
149
- return { outcome: 'unreachable', reason: error instanceof Error ? error.message : String(error) };
150
- }
151
- if (response.status === 404)
152
- return { outcome: 'not_found' };
153
- if (response.status === 401)
154
- return { outcome: 'unauthorized' };
155
- if (!response.ok)
156
- return { outcome: 'unreachable', reason: `control plane returned ${response.status}` };
157
- return { outcome: 'ok', run: (await response.json()) };
158
- }
159
- export async function loadRunAuto(options) {
160
- const fetched = await fetchRunFromServer(options.server, options.selector);
161
- if (fetched.outcome === 'ok')
162
- return { run: fetched.run, source: 'server', stale: false };
163
- // A 401 is a misconfiguration, not an outage — falling back to disk would
164
- // silently mask it (and a remote control plane has no local disk anyway).
165
- if (fetched.outcome === 'unauthorized')
166
- throw new VerifyUnauthorizedError(options.server);
167
- if (options.repoRoot) {
168
- const runId = resolveRunIdOnDisk(options.repoRoot, options.selector);
169
- const run = loadRunFromDisk(options.repoRoot, runId);
170
- if (run) {
171
- return { run, source: 'disk', stale: !TERMINAL_RUN_STATUSES.has(run.status) };
172
- }
173
- }
174
- if (fetched.outcome === 'not_found') {
175
- throw new Error(`Run "${options.selector}" not found on the control plane${options.repoRoot ? ' or on disk' : ''}.`);
176
- }
177
- throw new Error(`Run "${options.selector}" is not readable: control plane at ${options.server} is unreachable (${fetched.reason})`
178
- + (options.repoRoot ? ` and no matching run exists under ${runsRoot(options.repoRoot)}.`
179
- : ' and no .haystack/verify directory was found from the current directory (run from the repository, or set HAYSTACK_VERIFY_REPO).'));
180
- }
181
- export function isTerminal(run) {
182
- return TERMINAL_RUN_STATUSES.has(run.status);
183
- }
184
- function rowOfRun(run) {
185
- const bisect = run.chapters.find(chapter => chapter.bisect)?.bisect;
186
- return {
187
- run_id: run.runId,
188
- chapters: run.chapters.map(chapter => chapter.chapterId),
189
- status: run.status,
190
- age: formatAge(run.createdAt),
191
- created_at: run.createdAt,
192
- updated_at: run.updatedAt,
193
- tests: run.summary.total,
194
- passed: run.summary.passed,
195
- failed: run.summary.failed,
196
- could_not_run: run.summary.infraErrors + (run.summary.inconclusive ?? 0),
197
- culprit_sha: bisect?.culpritSha ?? null,
198
- };
199
- }
200
- /**
201
- * List runs, newest first. Prefers the control plane's GET /runs (present on
202
- * servers with the operator API; required for a remote control plane) and
203
- * falls back to enumerating the runs directory.
204
- */
205
- export async function listRunRows(options) {
206
- let serverReason = 'unreachable';
207
- try {
208
- const response = await fetch(`${options.server}/api/cloud-verifier/runs?limit=${options.limit}`, {
209
- headers: verifyAuthHeaders(),
210
- });
211
- if (response.status === 401)
212
- throw new VerifyUnauthorizedError(options.server);
213
- if (response.ok) {
214
- const body = (await response.json());
215
- if (Array.isArray(body.runs)) {
216
- // Recompute ages locally so they do not drift with server response caching.
217
- const rows = body.runs.map(row => ({
218
- ...row,
219
- age: formatAge(row.created_at || row.updated_at) ?? row.age,
220
- }));
221
- return { rows, source: 'server', total: body.total ?? rows.length };
222
- }
223
- }
224
- // A 404 is an older control plane without the list route; use disk.
225
- serverReason = `HTTP ${response.status}`;
226
- }
227
- catch (error) {
228
- if (error instanceof VerifyUnauthorizedError)
229
- throw error;
230
- serverReason = error instanceof Error ? error.message : String(error);
231
- }
232
- if (!options.repoRoot) {
233
- throw new Error(`Cannot list runs: the control plane at ${options.server} has no run-list endpoint (${serverReason}) `
234
- + 'and no .haystack/verify directory was found locally (run from the repository or set HAYSTACK_VERIFY_REPO).');
235
- }
236
- const entries = listRunDirs(options.repoRoot);
237
- const rows = entries.slice(0, options.limit)
238
- .map(entry => loadRunFromDisk(options.repoRoot, entry.runId))
239
- .filter((run) => run !== null)
240
- .map(rowOfRun);
241
- return { rows, source: 'disk', total: entries.length };
242
- }
243
- // ---------- summarisation (findings, not run.json) ----------
244
- export function riskCellIdFor(chapterId, cellKey) {
245
- return `cell-${createHash('sha256').update(`${chapterId}|${cellKey}`).digest('hex').slice(0, 12)}`;
246
- }
247
- /**
248
- * Recover the planner's cellKey from artifact filenames (`<cellKey>-<side>-…`).
249
- * run.json stores only the hashed riskCellId; the key is friendlier to type.
250
- */
251
- /**
252
- * Basename of an evidence artifactUrl. The executor URL-encodes the whole
253
- * relative path (`<chapter>%2Fartifacts%2F<file>`), so decode before taking
254
- * the last path segment.
255
- */
256
- export function artifactFileOf(artifactUrl) {
257
- const decoded = decodeURIComponent(artifactUrl);
258
- return decoded.split('/').pop() ?? decoded;
259
- }
260
- /** Relative artifact path under the run's chapters/ dir, from an evidence URL. */
261
- export function artifactRelOf(artifactUrl) {
262
- const decoded = decodeURIComponent(artifactUrl);
263
- const match = decoded.match(/\/artifacts\/(.+)$/);
264
- return match ? match[1] : null;
265
- }
266
- export function cellKeyOf(cell) {
267
- for (const universe of [cell.head, cell.base]) {
268
- for (const evidence of universe.evidence) {
269
- if (!evidence.artifactUrl)
270
- continue;
271
- const file = artifactFileOf(evidence.artifactUrl);
272
- const marker = `-${universe.side}-`;
273
- const index = file.indexOf(marker);
274
- if (index > 0)
275
- return file.slice(0, index);
276
- }
277
- }
278
- return null;
279
- }
280
- function liveEnvironment(cell) {
281
- for (const universe of [cell.head, cell.base]) {
282
- if (!universe.browserUrl)
283
- continue;
284
- if (universe.status !== 'running')
285
- continue;
286
- const expiresAt = universe.expiresAt ?? null;
287
- const expired = expiresAt ? Date.parse(expiresAt) < Date.now() : false;
288
- return { url: universe.browserUrl, expiresAt, expired };
289
- }
290
- return null;
291
- }
292
- export function formatDurationMs(ms) {
293
- if (typeof ms !== 'number' || !Number.isFinite(ms))
294
- return null;
295
- if (ms < 1000)
296
- return `${Math.round(ms)}ms`;
297
- const seconds = Math.round(ms / 1000);
298
- if (seconds < 60)
299
- return `${seconds}s`;
300
- const minutes = Math.floor(seconds / 60);
301
- if (minutes < 60)
302
- return `${minutes}m${String(seconds % 60).padStart(2, '0')}s`;
303
- const hours = Math.floor(minutes / 60);
304
- if (hours < 48)
305
- return `${hours}h${String(minutes % 60).padStart(2, '0')}m`;
306
- return `${Math.floor(hours / 24)}d${hours % 24}h`;
307
- }
308
- export function formatAge(iso) {
309
- if (!iso)
310
- return null;
311
- const delta = Date.now() - Date.parse(iso);
312
- if (Number.isNaN(delta))
313
- return null;
314
- return formatDurationMs(Math.abs(delta));
315
- }
316
- export function formatUntil(iso) {
317
- if (!iso)
318
- return null;
319
- const delta = Date.parse(iso) - Date.now();
320
- if (Number.isNaN(delta))
321
- return null;
322
- return delta <= 0 ? 'expired' : `in ${formatDurationMs(delta)}`;
323
- }
324
- /**
325
- * Parse an assertion summary of the executor's fixed shape:
326
- * `<comparison> — base=<json> head=<json> → PASS|FAIL|UNMEASURABLE`.
327
- * Fallback for runs whose executor did not yet attach structured details to
328
- * assertion evidence; the structured path always wins when present.
329
- */
330
- function readingFromSummary(label, summary) {
331
- const arrow = summary.lastIndexOf(' → ');
332
- if (arrow < 0)
333
- return null;
334
- const result = summary.slice(arrow + ' → '.length).trim();
335
- if (!['PASS', 'FAIL', 'UNMEASURABLE'].includes(result))
336
- return null;
337
- const head = summary.slice(0, arrow);
338
- const baseAt = head.indexOf(' — base=');
339
- if (baseAt < 0)
340
- return null;
341
- const comparison = head.slice(0, baseAt);
342
- const rest = head.slice(baseAt + ' — base='.length);
343
- const headAt = rest.indexOf(' head=');
344
- if (headAt < 0)
345
- return null;
346
- const parse = (raw) => {
347
- try {
348
- return JSON.parse(raw);
349
- }
350
- catch {
351
- return raw;
352
- }
353
- };
354
- const baseValue = parse(rest.slice(0, headAt));
355
- const headValue = parse(rest.slice(headAt + ' head='.length));
356
- return {
357
- metric: label,
358
- comparison,
359
- base: baseValue ?? null,
360
- head: headValue ?? null,
361
- changed: JSON.stringify(baseValue ?? null) !== JSON.stringify(headValue ?? null),
362
- result: result === 'UNMEASURABLE' ? 'unmeasurable' : result === 'PASS' ? 'pass' : 'fail',
363
- };
364
- }
365
- /** Frozen-oracle readings for one cell, extracted from its assertion evidence. */
366
- export function oracleReadingsOf(cell) {
367
- const readings = [];
368
- const seen = new Set();
369
- for (const universe of [cell.head, cell.base]) {
370
- for (const evidence of universe.evidence) {
371
- if (evidence.kind !== 'assertion')
372
- continue;
373
- const details = (evidence.details ?? null);
374
- const reading = details && 'pass' in details
375
- ? {
376
- metric: String(details.metric ?? evidence.label),
377
- comparison: String(details.comparison ?? ''),
378
- base: details.baseValue ?? null,
379
- head: details.headValue ?? null,
380
- changed: JSON.stringify(details.baseValue ?? null) !== JSON.stringify(details.headValue ?? null),
381
- result: details.missing ? 'unmeasurable' : details.pass ? 'pass' : 'fail',
382
- }
383
- : readingFromSummary(evidence.label, evidence.summary);
384
- if (!reading)
385
- continue;
386
- const key = `${reading.metric}|${reading.comparison}`;
387
- if (seen.has(key))
388
- continue;
389
- seen.add(key);
390
- readings.push(reading);
391
- }
392
- }
393
- return readings;
394
- }
395
- function findingOf(chapter, cell) {
396
- const live = liveEnvironment(cell);
397
- return {
398
- chapter_id: chapter.chapterId,
399
- ordinal: cell.ordinal,
400
- title: cell.title,
401
- severity: cell.severity ?? null,
402
- verdict: cell.agentJudgment?.verdict ?? null,
403
- summary: cell.agentJudgment ? cell.agentJudgment.explanation.split('\n\n')[0] : null,
404
- changed_readings: oracleReadingsOf(cell)
405
- .filter(reading => reading.result !== 'pass')
406
- .map(reading => reading.result === 'unmeasurable'
407
- ? `${reading.metric}: unmeasurable`
408
- : `${reading.metric}: ${JSON.stringify(reading.base)} → ${JSON.stringify(reading.head)}`),
409
- live_url: live && !live.expired ? live.url : null,
410
- };
411
- }
412
- export function summarizeRun(loaded, server) {
413
- const { run } = loaded;
414
- const liveEnvironments = [];
415
- const findings = [];
416
- const problems = [];
417
- // The control plane marks an orphaned run by rewriting its status to
418
- // `failed` and appending a synthetic evt-stale event; surface that the same
419
- // way as our own disk-side stale detection. The marking can be a false
420
- // positive (a dev-server restart loses the in-memory registry while the old
421
- // orchestrator keeps running), so a recently-advancing run.json softens it.
422
- const staleMarked = run.events.some(event => event.eventId.startsWith('evt-stale-'));
423
- const stale = loaded.stale || staleMarked;
424
- const possiblyAlive = stale && Date.now() - Date.parse(run.updatedAt) < 60_000;
425
- if (stale) {
426
- problems.push(possiblyAlive
427
- ? 'The control plane lost track of this run (dev-server restart), but run.json advanced within the last minute — it may still be progressing. `haystack verify watch` will confirm.'
428
- : 'Run is not terminal but no orchestrator is driving it (control plane down or restarted mid-run). Start a new run.');
429
- }
430
- // Once nothing can drive the run further, a cell still in a live phase will
431
- // never finish — that is a could-not-run, not a verdict.
432
- const runDead = (stale && !possiblyAlive) || (TERMINAL_RUN_STATUSES.has(run.status) && !stale);
433
- const cellTerminal = new Set(['passed', 'failed', 'infra_error', 'inconclusive']);
434
- let stuckCells = 0;
435
- const chapters = run.chapters.map(chapter => {
436
- const mode = chapter.executionMode === 'bisect' ? 'bisect' : 'risk_cells';
437
- const tests = (chapter.riskCells ?? []).map(cell => {
438
- if (runDead && !cellTerminal.has(cell.status)) {
439
- stuckCells += 1;
440
- problems.push(`Test ${cell.ordinal} ("${cell.title}") never finished (still ${cell.status} when the run died) — not a verdict about the change.`);
441
- }
442
- // Older executors report a cell whose oracles could not be measured as
443
- // `inconclusive` (newer ones fail the run instead). Either way it is our
444
- // gap, never a verdict.
445
- if (cell.status === 'inconclusive') {
446
- problems.push(`Test ${cell.ordinal} ("${cell.title}") was inconclusive — one or more readings could not be measured. That is a harness gap, not a verdict about the change.`);
447
- }
448
- const live = liveEnvironment(cell);
449
- if (live && !live.expired) {
450
- liveEnvironments.push({
451
- chapter_id: chapter.chapterId,
452
- ordinal: cell.ordinal,
453
- title: cell.title,
454
- url: live.url,
455
- expires_at: live.expiresAt,
456
- expires: formatUntil(live.expiresAt),
457
- });
458
- }
459
- if (cell.status === 'infra_error') {
460
- problems.push(cell.infrastructureError
461
- ? `Test ${cell.ordinal} ("${cell.title}") could not run — ${cell.infrastructureError.reason}: ${cell.infrastructureError.message}`
462
- : `Test ${cell.ordinal} ("${cell.title}") could not run — infrastructure failure, not a verdict about the change.`);
463
- }
464
- if (cell.status === 'failed') {
465
- findings.push(findingOf(chapter, cell));
466
- }
467
- return {
468
- ordinal: cell.ordinal,
469
- cell_id: cell.riskCellId,
470
- cell_key: cellKeyOf(cell),
471
- title: cell.title,
472
- status: cell.status,
473
- severity: cell.severity ?? null,
474
- verdict: cell.agentJudgment?.verdict ?? null,
475
- duration: formatDurationMs(cell.durationMs),
476
- live_url: live && !live.expired ? live.url : null,
477
- live_expires_at: live && !live.expired ? live.expiresAt : null,
478
- live_expires: live && !live.expired ? formatUntil(live.expiresAt) : null,
479
- };
480
- });
481
- let bisect = null;
482
- if (chapter.bisect) {
483
- const byStatus = {};
484
- for (const probe of chapter.bisect.probes) {
485
- byStatus[probe.status] = (byStatus[probe.status] ?? 0) + 1;
486
- }
487
- bisect = {
488
- status: chapter.bisect.status,
489
- signal: chapter.bisect.signal,
490
- total_candidates: chapter.bisect.totalCandidates,
491
- probes_total: chapter.bisect.probes.length,
492
- probes_by_status: byStatus,
493
- culprit_sha: chapter.bisect.culpritSha ?? null,
494
- culprit_subject: chapter.bisect.culpritSubject ?? null,
495
- parent_sha: chapter.bisect.parentSha ?? null,
496
- };
497
- if (['inconclusive', 'error'].includes(chapter.bisect.status)) {
498
- problems.push(`Bisect chapter ${chapter.chapterId} ended ${chapter.bisect.status} — no culprit isolated.`);
499
- }
500
- if (chapter.bisect.culpritSha) {
501
- findings.push({
502
- chapter_id: chapter.chapterId,
503
- ordinal: 0,
504
- title: `First bad commit: ${chapter.bisect.culpritSha.slice(0, 8)} ${chapter.bisect.culpritSubject ?? ''}`.trim(),
505
- severity: null,
506
- verdict: 'culprit',
507
- summary: chapter.bisect.signal,
508
- changed_readings: chapter.bisect.parentSha ? [`clean parent: ${chapter.bisect.parentSha.slice(0, 8)}`] : [],
509
- live_url: null,
510
- });
511
- }
512
- }
513
- if (chapter.status === 'failed' && mode === 'risk_cells' && tests.length === 0) {
514
- problems.push(`Chapter ${chapter.chapterId} failed before any test ran (see events).`);
515
- }
516
- return {
517
- chapter_id: chapter.chapterId,
518
- execution_mode: mode,
519
- analysis_mode: chapter.analysisMode ?? null,
520
- label: chapter.label,
521
- repository: chapter.repository,
522
- pull_request: chapter.pullRequest,
523
- status: chapter.status,
524
- tests,
525
- bisect,
526
- };
527
- });
528
- if (run.summary.infraErrors > 0) {
529
- problems.push(`${run.summary.infraErrors} test(s) could not run at all; they are excluded from pass/fail.`);
530
- }
531
- const lastEvent = run.events.length > 0 ? run.events[run.events.length - 1] : null;
532
- return {
533
- schema_version: '1.0.0',
534
- run_id: run.runId,
535
- status: stale ? 'failed' : run.status,
536
- stale,
537
- source: loaded.source,
538
- created_at: run.createdAt,
539
- updated_at: run.updatedAt,
540
- age: formatAge(run.createdAt),
541
- counts: {
542
- total: run.summary.total,
543
- completed: run.summary.completed,
544
- passed: run.summary.passed,
545
- failed: run.summary.failed,
546
- could_not_run: run.summary.infraErrors + (run.summary.inconclusive ?? 0) + stuckCells,
547
- },
548
- findings,
549
- chapters,
550
- live_environments: liveEnvironments,
551
- problems,
552
- last_event: lastEvent ? { at: lastEvent.at, phase: lastEvent.phase, message: lastEvent.message } : null,
553
- results_url: `${server}/dev/verify/${run.runId}`,
554
- };
555
- }
556
- /**
557
- * Exit-code policy shared by `verify watch` and callers of the MCP wait tool:
558
- * 0 — everything that ran passed (a found bisect culprit counts as success);
559
- * 1 — at least one test failed;
560
- * 2 — the run itself broke, went stale, or some test could not run.
561
- */
562
- export function exitCodeFor(summary) {
563
- const bisectBroken = summary.chapters.some(chapter => chapter.bisect && ['inconclusive', 'error'].includes(chapter.bisect.status));
564
- if (summary.stale || summary.counts.could_not_run > 0 || bisectBroken)
565
- return 2;
566
- if (summary.status === 'failed' && summary.counts.total === 0)
567
- return 2;
568
- if (summary.counts.failed > 0)
569
- return 1;
570
- return 0;
571
- }
572
- /** Match a test by ordinal, riskCellId, planner cellKey, or title substring. */
573
- export function findCell(run, selector) {
574
- const cells = run.chapters.flatMap(chapter => (chapter.riskCells ?? []).map(cell => ({ chapter, cell })));
575
- if (cells.length === 0) {
576
- throw new Error(`Run ${run.runId} has no risk cells (execution mode: ${run.chapters.map(c => c.executionMode ?? 'risk_cells').join(', ')}).`);
577
- }
578
- const ordinal = Number(selector);
579
- const matches = cells.filter(({ chapter, cell }) => cell.riskCellId === selector
580
- || (Number.isInteger(ordinal) && ordinal > 0 && cell.ordinal === ordinal)
581
- || riskCellIdFor(chapter.chapterId, selector) === cell.riskCellId
582
- || cellKeyOf(cell) === selector
583
- || cell.title.toLowerCase().includes(selector.toLowerCase()));
584
- if (matches.length === 1)
585
- return matches[0];
586
- if (matches.length === 0) {
587
- throw new Error(`No test matches "${selector}". Tests:\n`
588
- + cells.map(({ cell }) => ` ${cell.ordinal}. ${cell.title} (${cell.riskCellId})`).join('\n'));
589
- }
590
- throw new Error(`"${selector}" is ambiguous. Matches:\n`
591
- + matches.map(({ cell }) => ` ${cell.ordinal}. ${cell.title} (${cell.riskCellId})`).join('\n'));
592
- }
593
- /**
594
- * Resolve a cell for an operator action without using its free-text title.
595
- * Ordinals are scoped by the exact run, while riskCellId and planner cellKey
596
- * are stable structured identifiers carried by the run.
597
- */
598
- export function findCellByStableSelector(run, selector) {
599
- const cells = run.chapters.flatMap(chapter => (chapter.riskCells ?? []).map(cell => ({ chapter, cell })));
600
- if (cells.length === 0) {
601
- throw new Error(`Run ${run.runId} has no risk cells (execution mode: `
602
- + `${run.chapters.map(c => c.executionMode ?? 'risk_cells').join(', ')}).`);
603
- }
604
- const ordinal = Number(selector);
605
- const matches = cells.filter(({ chapter, cell }) => cell.riskCellId === selector
606
- || (Number.isInteger(ordinal) && ordinal > 0 && cell.ordinal === ordinal)
607
- || riskCellIdFor(chapter.chapterId, selector) === cell.riskCellId
608
- || cellKeyOf(cell) === selector);
609
- if (matches.length === 1)
610
- return matches[0];
611
- if (matches.length === 0) {
612
- throw new Error(`No test has stable selector "${selector}". Use an ordinal, riskCellId, `
613
- + `or exact planner cellKey:\n`
614
- + cells
615
- .map(({ cell }) => ` ${cell.ordinal}. ${cell.title} (${cell.riskCellId})`)
616
- .join('\n'));
617
- }
618
- throw new Error(`"${selector}" is ambiguous. Use a riskCellId instead:\n`
619
- + matches
620
- .map(({ cell }) => ` ${cell.ordinal}. ${cell.title} (${cell.riskCellId})`)
621
- .join('\n'));
622
- }
623
- function universeDetail(universe, opts) {
624
- const network = [];
625
- const sdkLog = [];
626
- const driverFailures = [];
627
- const artifacts = [];
628
- for (const evidence of universe.evidence) {
629
- if (evidence.kind === 'network' && opts.timeline) {
630
- const details = (evidence.details ?? {});
631
- network.push({
632
- sequence: typeof details.sequence === 'number' ? details.sequence : null,
633
- offset_ms: typeof details.offsetMs === 'number' ? details.offsetMs : null,
634
- method: typeof details.method === 'string' ? details.method : null,
635
- path: typeof details.path === 'string' ? details.path : evidence.label,
636
- status: typeof evidence.value === 'string' ? evidence.value : String(details.status ?? ''),
637
- channel: typeof details.channel === 'string' ? details.channel : null,
638
- event_name: typeof details.eventName === 'string' ? details.eventName : null,
639
- step: evidence.step ? `${evidence.step.ordinal}. ${evidence.step.name}` : null,
640
- });
641
- }
642
- else if (evidence.kind === 'log') {
643
- if (evidence.label === 'driver failure')
644
- driverFailures.push(evidence.summary);
645
- else
646
- sdkLog.push(evidence.summary);
647
- }
648
- else if (['screenshot', 'dom', 'video'].includes(evidence.kind) && evidence.artifactUrl) {
649
- artifacts.push({
650
- kind: evidence.kind,
651
- label: evidence.label,
652
- url: evidence.artifactUrl,
653
- file: artifactFileOf(evidence.artifactUrl),
654
- });
655
- }
656
- }
657
- network.sort((a, b) => (a.sequence ?? 0) - (b.sequence ?? 0));
658
- return {
659
- universe_id: universe.universeId,
660
- status: universe.status,
661
- sandbox_id: universe.sandboxId ?? null,
662
- state_provider: universe.stateProvider ?? null,
663
- snapshot_id: universe.snapshotId ?? null,
664
- fork_id: universe.forkId ?? null,
665
- execution_id: universe.executionId ?? null,
666
- state_layers: universe.stateLayers ?? [],
667
- network,
668
- sdk_log: sdkLog,
669
- driver_failures: driverFailures,
670
- artifacts,
671
- };
672
- }
673
- export function cellDetail(run, match, opts) {
674
- const { chapter, cell } = match;
675
- const timeline = opts?.timeline !== false;
676
- const readings = oracleReadingsOf(cell);
677
- const live = liveEnvironment(cell);
678
- const verdict = cell.agentJudgment?.verdict ?? null;
679
- const finding = cell.agentJudgment
680
- ? `${cell.title}: ${verdict === 'regression' ? 'regressed' : verdict}. ${cell.agentJudgment.explanation.split(/(?<=\.)\s/)[0]}`
681
- : null;
682
- return {
683
- schema_version: '1.0.0',
684
- run_id: run.runId,
685
- chapter_id: chapter.chapterId,
686
- ordinal: cell.ordinal,
687
- cell_id: cell.riskCellId,
688
- cell_key: cellKeyOf(cell),
689
- title: cell.title,
690
- status: cell.status,
691
- severity: cell.severity ?? null,
692
- duration: formatDurationMs(cell.durationMs),
693
- finding,
694
- verdict,
695
- explanation: cell.agentJudgment?.explanation ?? null,
696
- intent_evidence: cell.agentJudgment?.intentEvidence ?? null,
697
- infrastructure_error: cell.infrastructureError ?? null,
698
- rationale: cell.rationale ?? null,
699
- observable_consequence: cell.observableConsequence ?? null,
700
- oracle_readings: readings,
701
- readings_changed: readings.filter(reading => reading.changed || reading.result !== 'pass'),
702
- live_environment: live && !live.expired
703
- ? { url: live.url, expires_at: live.expiresAt, expires: formatUntil(live.expiresAt) }
704
- : null,
705
- base: universeDetail(cell.base, { timeline }),
706
- head: universeDetail(cell.head, { timeline }),
707
- };
708
- }
709
- /**
710
- * Poll until the run reaches a terminal state. The terminal predicate is the
711
- * run status alone — never a cell's: a cell legitimately sits in
712
- * `provisioning` for minutes while sandbox creation retries.
713
- *
714
- * A stale marking (from the control plane, or from a disk read with no
715
- * reachable server) is treated with suspicion rather than as terminal: a Vite
716
- * restart re-creates the plugin's in-memory run registry while the old
717
- * orchestrator can keep running in the same process, so "no orchestrator is
718
- * driving this run" can be false. As long as run.json keeps advancing, the run
719
- * is alive regardless of what the registry thinks; staleness is only accepted
720
- * after a no-progress grace period.
721
- */
722
- export async function waitForTerminal(options) {
723
- const intervalMs = options.intervalMs ?? 3000;
724
- const staleGraceMs = Math.max(20_000, intervalMs * 4);
725
- const deadline = options.timeoutMs ? Date.now() + options.timeoutMs : null;
726
- const callbacks = options.callbacks ?? {};
727
- // Pin the alias to a concrete run id on first load so `latest` cannot slide
728
- // to a different run started mid-watch.
729
- let pinned = null;
730
- const seenEvents = new Set();
731
- const cellStatuses = new Map();
732
- let first = true;
733
- let lastUpdatedAt = '';
734
- let noProgressSince = null;
735
- for (;;) {
736
- const loaded = await loadRunAuto({
737
- server: options.server,
738
- repoRoot: options.repoRoot,
739
- selector: pinned ?? options.selector,
740
- });
741
- if (!pinned) {
742
- pinned = loaded.run.runId;
743
- callbacks.onResolved?.(pinned);
744
- }
745
- for (const event of loaded.run.events) {
746
- if (seenEvents.has(event.eventId))
747
- continue;
748
- seenEvents.add(event.eventId);
749
- // The backlog before we attached is context, not progress; report it only
750
- // via the first snapshot, then stream genuinely new events.
751
- if (!first)
752
- callbacks.onEvent?.(event);
753
- }
754
- for (const chapter of loaded.run.chapters) {
755
- for (const cell of chapter.riskCells ?? []) {
756
- const previous = cellStatuses.get(cell.riskCellId);
757
- cellStatuses.set(cell.riskCellId, cell.status);
758
- if (!first && previous && previous !== cell.status) {
759
- callbacks.onCellChange?.(chapter.chapterId, cell, previous);
760
- }
761
- }
762
- }
763
- first = false;
764
- const staleMarked = loaded.stale || loaded.run.events.some(event => event.eventId.startsWith('evt-stale-'));
765
- if (isTerminal(loaded.run) && !staleMarked) {
766
- return { loaded, timedOut: false, endReason: 'terminal' };
767
- }
768
- const progressed = loaded.run.updatedAt !== lastUpdatedAt;
769
- lastUpdatedAt = loaded.run.updatedAt;
770
- if (staleMarked) {
771
- if (progressed) {
772
- noProgressSince = null;
773
- }
774
- else {
775
- noProgressSince = noProgressSince ?? Date.now();
776
- if (Date.now() - noProgressSince > staleGraceMs) {
777
- return { loaded, timedOut: false, endReason: 'stale' };
778
- }
779
- }
780
- }
781
- else {
782
- noProgressSince = null;
783
- }
784
- if (deadline && Date.now() > deadline) {
785
- return { loaded, timedOut: true, endReason: 'timeout' };
786
- }
787
- await new Promise(resolve => setTimeout(resolve, intervalMs));
788
- }
789
- }