@haystackeditor/cli 0.15.20 → 0.15.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/dist/commands/verify-core.d.ts +424 -0
- package/dist/commands/verify-core.js +753 -0
- package/dist/commands/verify-mcp.d.ts +8 -0
- package/dist/commands/verify-mcp.js +458 -0
- package/dist/commands/verify-ops.d.ts +158 -0
- package/dist/commands/verify-ops.js +1148 -0
- package/dist/commands/verify-reseal.d.ts +9 -0
- package/dist/commands/verify-reseal.js +148 -0
- package/dist/commands/verify-sandboxes.d.ts +95 -0
- package/dist/commands/verify-sandboxes.js +352 -0
- package/dist/commands/verify.d.ts +11 -0
- package/dist/commands/verify.js +97 -0
- package/dist/index.js +255 -0
- package/dist/types.d.ts +54 -0
- package/dist/types.js +15 -0
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -65,6 +65,20 @@ haystack triage <ref> --json # poll findings later / after --no-wait
|
|
|
65
65
|
`traces_list`, `traces_get`, `dismiss`, `mark_reviewed`, `undismiss`,
|
|
66
66
|
`request_review`, `trigger_review`, and `schema`.
|
|
67
67
|
|
|
68
|
+
**`haystack verify mcp`** runs the cloud-verifier MCP server. Retained
|
|
69
|
+
application state is available through `verify_reopen`, `verify_refork`, and
|
|
70
|
+
`verify_cleanup`. Agents select state by an exact Archil fork ID or a stable
|
|
71
|
+
run/test/side coordinate; refork returns a durable child ID that can be
|
|
72
|
+
reopened or nested again without receiving provider credentials. The matching
|
|
73
|
+
CLI flow is:
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
haystack verify reopen latest 3
|
|
77
|
+
# inspect or mutate the returned sandbox
|
|
78
|
+
haystack verify refork latest 3 attempt:fix-2
|
|
79
|
+
haystack verify reopen archil:<disk-id>:branch:<returned-child>
|
|
80
|
+
```
|
|
81
|
+
|
|
68
82
|
**`haystack setup --json`** speaks NDJSON: events (`question`, `permission`,
|
|
69
83
|
`progress`, `result`) on stdout, replies on stdin keyed by
|
|
70
84
|
`requestID`/`permissionID`. `haystack schema setup` documents the full
|
|
@@ -0,0 +1,424 @@
|
|
|
1
|
+
export declare const DEFAULT_CONTROL_PLANE = "http://127.0.0.1:3000";
|
|
2
|
+
export declare const TERMINAL_RUN_STATUSES: Set<string>;
|
|
3
|
+
export interface UniverseEvidence {
|
|
4
|
+
evidenceId: string;
|
|
5
|
+
kind: string;
|
|
6
|
+
label: string;
|
|
7
|
+
summary: string;
|
|
8
|
+
value?: unknown;
|
|
9
|
+
artifactUrl?: string;
|
|
10
|
+
step?: {
|
|
11
|
+
ordinal: number;
|
|
12
|
+
name: string;
|
|
13
|
+
};
|
|
14
|
+
details?: Record<string, unknown>;
|
|
15
|
+
capturedAt?: string;
|
|
16
|
+
}
|
|
17
|
+
export interface UniverseRef {
|
|
18
|
+
universeId: string;
|
|
19
|
+
side: 'base' | 'head';
|
|
20
|
+
status: string;
|
|
21
|
+
snapshotId?: string;
|
|
22
|
+
forkId?: string;
|
|
23
|
+
executionId?: string;
|
|
24
|
+
stateProvider?: 'e2b-snapshot' | 'archil';
|
|
25
|
+
stateLayers?: Array<{
|
|
26
|
+
layerId: string;
|
|
27
|
+
kind: 'compute-filesystem' | 'application-filesystem';
|
|
28
|
+
provider: 'e2b-snapshot' | 'archil';
|
|
29
|
+
immutableBaseId: string;
|
|
30
|
+
mutableOverlayId?: string;
|
|
31
|
+
mountPath?: string;
|
|
32
|
+
consistencyPoint?: string;
|
|
33
|
+
}>;
|
|
34
|
+
sandboxId?: string;
|
|
35
|
+
browserUrl?: string;
|
|
36
|
+
evidence: UniverseEvidence[];
|
|
37
|
+
startedAt?: string;
|
|
38
|
+
completedAt?: string;
|
|
39
|
+
expiresAt?: string;
|
|
40
|
+
stateExpiresAt?: string;
|
|
41
|
+
}
|
|
42
|
+
export interface RiskCell {
|
|
43
|
+
riskCellId: string;
|
|
44
|
+
ordinal: number;
|
|
45
|
+
title: string;
|
|
46
|
+
rationale?: string;
|
|
47
|
+
observableConsequence?: string;
|
|
48
|
+
activationConditions?: string[];
|
|
49
|
+
causalPath?: string[];
|
|
50
|
+
status: string;
|
|
51
|
+
severity?: string;
|
|
52
|
+
confidence?: number;
|
|
53
|
+
durationMs?: number;
|
|
54
|
+
base: UniverseRef;
|
|
55
|
+
head: UniverseRef;
|
|
56
|
+
agentJudgment?: {
|
|
57
|
+
verdict: string;
|
|
58
|
+
explanation: string;
|
|
59
|
+
intentEvidence?: string;
|
|
60
|
+
};
|
|
61
|
+
infrastructureError?: {
|
|
62
|
+
reason: string;
|
|
63
|
+
message: string;
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
export interface BisectProbe {
|
|
67
|
+
commit: string;
|
|
68
|
+
shortSha: string;
|
|
69
|
+
subject: string;
|
|
70
|
+
position: number;
|
|
71
|
+
status: string;
|
|
72
|
+
wave?: number;
|
|
73
|
+
sandboxId?: string;
|
|
74
|
+
durationMs?: number;
|
|
75
|
+
error?: string;
|
|
76
|
+
}
|
|
77
|
+
export interface BisectResult {
|
|
78
|
+
status: string;
|
|
79
|
+
signal: string;
|
|
80
|
+
predicate: string;
|
|
81
|
+
totalCandidates: number;
|
|
82
|
+
probes: BisectProbe[];
|
|
83
|
+
culpritSha?: string;
|
|
84
|
+
culpritSubject?: string;
|
|
85
|
+
parentSha?: string;
|
|
86
|
+
incidentProbe?: BisectProbe & {
|
|
87
|
+
ref?: string;
|
|
88
|
+
label?: string;
|
|
89
|
+
};
|
|
90
|
+
}
|
|
91
|
+
export interface VerificationChapter {
|
|
92
|
+
chapterId: string;
|
|
93
|
+
executionMode?: 'risk_cells' | 'bisect';
|
|
94
|
+
analysisMode?: string;
|
|
95
|
+
label: string;
|
|
96
|
+
repository: string;
|
|
97
|
+
pullRequest: string;
|
|
98
|
+
pullRequestUrl?: string;
|
|
99
|
+
status: string;
|
|
100
|
+
riskCells: RiskCell[];
|
|
101
|
+
bisect?: BisectResult;
|
|
102
|
+
}
|
|
103
|
+
export interface VerificationRunEvent {
|
|
104
|
+
eventId: string;
|
|
105
|
+
at: string;
|
|
106
|
+
phase: string;
|
|
107
|
+
message: string;
|
|
108
|
+
chapterId?: string;
|
|
109
|
+
}
|
|
110
|
+
export interface VerificationRun {
|
|
111
|
+
schemaVersion: string;
|
|
112
|
+
runId: string;
|
|
113
|
+
status: string;
|
|
114
|
+
command: string;
|
|
115
|
+
createdAt: string;
|
|
116
|
+
updatedAt: string;
|
|
117
|
+
chapters: VerificationChapter[];
|
|
118
|
+
events: VerificationRunEvent[];
|
|
119
|
+
summary: {
|
|
120
|
+
total: number;
|
|
121
|
+
completed: number;
|
|
122
|
+
passed: number;
|
|
123
|
+
failed: number;
|
|
124
|
+
infraErrors: number;
|
|
125
|
+
inconclusive?: number;
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
export declare function resolveServer(explicit?: string): string;
|
|
129
|
+
export declare function setVerifyToken(token: string | undefined): void;
|
|
130
|
+
export declare function verifyAuthHeaders(): Record<string, string>;
|
|
131
|
+
export declare class VerifyUnauthorizedError extends Error {
|
|
132
|
+
constructor(server: string);
|
|
133
|
+
}
|
|
134
|
+
/**
|
|
135
|
+
* Find the repository that holds .haystack/verify — explicit flag, then
|
|
136
|
+
* HAYSTACK_VERIFY_REPO, then walking up from cwd. Null when nothing matches;
|
|
137
|
+
* server-backed commands still work without it.
|
|
138
|
+
*/
|
|
139
|
+
export declare function discoverVerifyRoot(explicit?: string): string | null;
|
|
140
|
+
export declare function runsRoot(repoRoot: string): string;
|
|
141
|
+
export declare function runDir(repoRoot: string, runId: string): string;
|
|
142
|
+
export declare function listRunDirs(repoRoot: string): Array<{
|
|
143
|
+
runId: string;
|
|
144
|
+
changedAt: number;
|
|
145
|
+
}>;
|
|
146
|
+
/** Same alias rules as the control plane: run id verbatim, `latest`, or a chapter id. */
|
|
147
|
+
export declare function resolveRunIdOnDisk(repoRoot: string, requested: string): string;
|
|
148
|
+
export declare function loadRunFromDisk(repoRoot: string, runId: string): VerificationRun | null;
|
|
149
|
+
export type ServerFetch = {
|
|
150
|
+
outcome: 'ok';
|
|
151
|
+
run: VerificationRun;
|
|
152
|
+
} | {
|
|
153
|
+
outcome: 'not_found';
|
|
154
|
+
} | {
|
|
155
|
+
outcome: 'unauthorized';
|
|
156
|
+
} | {
|
|
157
|
+
outcome: 'unreachable';
|
|
158
|
+
reason: string;
|
|
159
|
+
};
|
|
160
|
+
export declare function fetchRunFromServer(server: string, idOrAlias: string): Promise<ServerFetch>;
|
|
161
|
+
export interface LoadedRun {
|
|
162
|
+
run: VerificationRun;
|
|
163
|
+
source: 'server' | 'disk';
|
|
164
|
+
/**
|
|
165
|
+
* True when the run is non-terminal but nothing can still be driving it —
|
|
166
|
+
* a disk read with no reachable control plane. (The server performs the same
|
|
167
|
+
* marking itself by rewriting orphaned runs to `failed`.)
|
|
168
|
+
*/
|
|
169
|
+
stale: boolean;
|
|
170
|
+
}
|
|
171
|
+
export declare function loadRunAuto(options: {
|
|
172
|
+
server: string;
|
|
173
|
+
repoRoot: string | null;
|
|
174
|
+
selector: string;
|
|
175
|
+
}): Promise<LoadedRun>;
|
|
176
|
+
export declare function isTerminal(run: VerificationRun): boolean;
|
|
177
|
+
export interface RunListRow {
|
|
178
|
+
run_id: string;
|
|
179
|
+
chapters: string[];
|
|
180
|
+
status: string;
|
|
181
|
+
age: string | null;
|
|
182
|
+
updated_at: string;
|
|
183
|
+
tests: number;
|
|
184
|
+
passed: number;
|
|
185
|
+
failed: number;
|
|
186
|
+
could_not_run: number;
|
|
187
|
+
culprit_sha: string | null;
|
|
188
|
+
}
|
|
189
|
+
export interface RunListResult {
|
|
190
|
+
rows: RunListRow[];
|
|
191
|
+
source: 'server' | 'disk';
|
|
192
|
+
total: number;
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* List runs, newest first. Prefers the control plane's GET /runs (present on
|
|
196
|
+
* servers with the operator API; required for a remote control plane) and
|
|
197
|
+
* falls back to enumerating the runs directory.
|
|
198
|
+
*/
|
|
199
|
+
export declare function listRunRows(options: {
|
|
200
|
+
server: string;
|
|
201
|
+
repoRoot: string | null;
|
|
202
|
+
limit: number;
|
|
203
|
+
}): Promise<RunListResult>;
|
|
204
|
+
export declare function riskCellIdFor(chapterId: string, cellKey: string): string;
|
|
205
|
+
/**
|
|
206
|
+
* Recover the planner's cellKey from artifact filenames (`<cellKey>-<side>-…`).
|
|
207
|
+
* run.json stores only the hashed riskCellId; the key is friendlier to type.
|
|
208
|
+
*/
|
|
209
|
+
/**
|
|
210
|
+
* Basename of an evidence artifactUrl. The executor URL-encodes the whole
|
|
211
|
+
* relative path (`<chapter>%2Fartifacts%2F<file>`), so decode before taking
|
|
212
|
+
* the last path segment.
|
|
213
|
+
*/
|
|
214
|
+
export declare function artifactFileOf(artifactUrl: string): string;
|
|
215
|
+
/** Relative artifact path under the run's chapters/ dir, from an evidence URL. */
|
|
216
|
+
export declare function artifactRelOf(artifactUrl: string): string | null;
|
|
217
|
+
export declare function cellKeyOf(cell: RiskCell): string | null;
|
|
218
|
+
export declare function formatDurationMs(ms: number | undefined): string | null;
|
|
219
|
+
export declare function formatAge(iso: string | undefined): string | null;
|
|
220
|
+
export declare function formatUntil(iso: string | null | undefined): string | null;
|
|
221
|
+
export interface TestSummary {
|
|
222
|
+
ordinal: number;
|
|
223
|
+
cell_id: string;
|
|
224
|
+
cell_key: string | null;
|
|
225
|
+
title: string;
|
|
226
|
+
/** `passed` | `failed` | a live phase | `infra_error` (could not run — our defect, not a verdict). */
|
|
227
|
+
status: string;
|
|
228
|
+
severity: string | null;
|
|
229
|
+
verdict: string | null;
|
|
230
|
+
duration: string | null;
|
|
231
|
+
live_url: string | null;
|
|
232
|
+
live_expires_at: string | null;
|
|
233
|
+
live_expires: string | null;
|
|
234
|
+
}
|
|
235
|
+
export interface RunFinding {
|
|
236
|
+
chapter_id: string;
|
|
237
|
+
ordinal: number;
|
|
238
|
+
title: string;
|
|
239
|
+
severity: string | null;
|
|
240
|
+
verdict: string | null;
|
|
241
|
+
/** First paragraph of the judge's explanation; `verify show` / verify_cell has the whole thing. */
|
|
242
|
+
summary: string | null;
|
|
243
|
+
/** Failing oracle readings as "metric: base → head". */
|
|
244
|
+
changed_readings: string[];
|
|
245
|
+
live_url: string | null;
|
|
246
|
+
}
|
|
247
|
+
export interface RunSummary {
|
|
248
|
+
schema_version: '1.0.0';
|
|
249
|
+
run_id: string;
|
|
250
|
+
status: string;
|
|
251
|
+
stale: boolean;
|
|
252
|
+
source: 'server' | 'disk';
|
|
253
|
+
created_at: string;
|
|
254
|
+
updated_at: string;
|
|
255
|
+
age: string | null;
|
|
256
|
+
counts: {
|
|
257
|
+
total: number;
|
|
258
|
+
completed: number;
|
|
259
|
+
passed: number;
|
|
260
|
+
failed: number;
|
|
261
|
+
could_not_run: number;
|
|
262
|
+
};
|
|
263
|
+
/** One entry per failing test (and per isolated bisect culprit) — the "what broke" digest. */
|
|
264
|
+
findings: RunFinding[];
|
|
265
|
+
chapters: Array<{
|
|
266
|
+
chapter_id: string;
|
|
267
|
+
execution_mode: 'risk_cells' | 'bisect';
|
|
268
|
+
analysis_mode: string | null;
|
|
269
|
+
label: string;
|
|
270
|
+
repository: string;
|
|
271
|
+
pull_request: string;
|
|
272
|
+
status: string;
|
|
273
|
+
tests: TestSummary[];
|
|
274
|
+
bisect: null | {
|
|
275
|
+
status: string;
|
|
276
|
+
signal: string;
|
|
277
|
+
total_candidates: number;
|
|
278
|
+
probes_total: number;
|
|
279
|
+
probes_by_status: Record<string, number>;
|
|
280
|
+
culprit_sha: string | null;
|
|
281
|
+
culprit_subject: string | null;
|
|
282
|
+
parent_sha: string | null;
|
|
283
|
+
};
|
|
284
|
+
}>;
|
|
285
|
+
live_environments: Array<{
|
|
286
|
+
chapter_id: string;
|
|
287
|
+
ordinal: number;
|
|
288
|
+
title: string;
|
|
289
|
+
url: string;
|
|
290
|
+
expires_at: string | null;
|
|
291
|
+
expires: string | null;
|
|
292
|
+
}>;
|
|
293
|
+
/** Anything the verifier could not run or measure — reported as our failure, never as a finding. */
|
|
294
|
+
problems: string[];
|
|
295
|
+
last_event: {
|
|
296
|
+
at: string;
|
|
297
|
+
phase: string;
|
|
298
|
+
message: string;
|
|
299
|
+
} | null;
|
|
300
|
+
results_url: string;
|
|
301
|
+
}
|
|
302
|
+
/** Frozen-oracle readings for one cell, extracted from its assertion evidence. */
|
|
303
|
+
export declare function oracleReadingsOf(cell: RiskCell): OracleReading[];
|
|
304
|
+
export declare function summarizeRun(loaded: LoadedRun, server: string): RunSummary;
|
|
305
|
+
/**
|
|
306
|
+
* Exit-code policy shared by `verify watch` and callers of the MCP wait tool:
|
|
307
|
+
* 0 — everything that ran passed (a found bisect culprit counts as success);
|
|
308
|
+
* 1 — at least one test failed;
|
|
309
|
+
* 2 — the run itself broke, went stale, or some test could not run.
|
|
310
|
+
*/
|
|
311
|
+
export declare function exitCodeFor(summary: RunSummary): number;
|
|
312
|
+
export interface CellMatch {
|
|
313
|
+
chapter: VerificationChapter;
|
|
314
|
+
cell: RiskCell;
|
|
315
|
+
}
|
|
316
|
+
/** Match a test by ordinal, riskCellId, planner cellKey, or title substring. */
|
|
317
|
+
export declare function findCell(run: VerificationRun, selector: string): CellMatch;
|
|
318
|
+
/**
|
|
319
|
+
* Resolve a cell for an operator action without using its free-text title.
|
|
320
|
+
* Ordinals are scoped by the exact run, while riskCellId and planner cellKey
|
|
321
|
+
* are stable structured identifiers carried by the run.
|
|
322
|
+
*/
|
|
323
|
+
export declare function findCellByStableSelector(run: VerificationRun, selector: string): CellMatch;
|
|
324
|
+
export interface OracleReading {
|
|
325
|
+
metric: string;
|
|
326
|
+
comparison: string;
|
|
327
|
+
base: unknown;
|
|
328
|
+
head: unknown;
|
|
329
|
+
changed: boolean;
|
|
330
|
+
result: 'pass' | 'fail' | 'unmeasurable';
|
|
331
|
+
}
|
|
332
|
+
export interface TimelineEntry {
|
|
333
|
+
sequence: number | null;
|
|
334
|
+
offset_ms: number | null;
|
|
335
|
+
method: string | null;
|
|
336
|
+
path: string;
|
|
337
|
+
status: string;
|
|
338
|
+
channel: string | null;
|
|
339
|
+
event_name: string | null;
|
|
340
|
+
step: string | null;
|
|
341
|
+
}
|
|
342
|
+
export interface UniverseDetail {
|
|
343
|
+
universe_id: string;
|
|
344
|
+
status: string;
|
|
345
|
+
sandbox_id: string | null;
|
|
346
|
+
state_provider: 'e2b-snapshot' | 'archil' | null;
|
|
347
|
+
snapshot_id: string | null;
|
|
348
|
+
fork_id: string | null;
|
|
349
|
+
execution_id: string | null;
|
|
350
|
+
state_layers: NonNullable<UniverseRef['stateLayers']>;
|
|
351
|
+
network: TimelineEntry[];
|
|
352
|
+
sdk_log: string[];
|
|
353
|
+
driver_failures: string[];
|
|
354
|
+
artifacts: Array<{
|
|
355
|
+
kind: string;
|
|
356
|
+
label: string;
|
|
357
|
+
url: string;
|
|
358
|
+
file: string;
|
|
359
|
+
}>;
|
|
360
|
+
}
|
|
361
|
+
export interface CellDetail {
|
|
362
|
+
schema_version: '1.0.0';
|
|
363
|
+
run_id: string;
|
|
364
|
+
chapter_id: string;
|
|
365
|
+
ordinal: number;
|
|
366
|
+
cell_id: string;
|
|
367
|
+
cell_key: string | null;
|
|
368
|
+
title: string;
|
|
369
|
+
status: string;
|
|
370
|
+
severity: string | null;
|
|
371
|
+
duration: string | null;
|
|
372
|
+
finding: string | null;
|
|
373
|
+
verdict: string | null;
|
|
374
|
+
explanation: string | null;
|
|
375
|
+
intent_evidence: string | null;
|
|
376
|
+
infrastructure_error: {
|
|
377
|
+
reason: string;
|
|
378
|
+
message: string;
|
|
379
|
+
} | null;
|
|
380
|
+
rationale: string | null;
|
|
381
|
+
observable_consequence: string | null;
|
|
382
|
+
oracle_readings: OracleReading[];
|
|
383
|
+
readings_changed: OracleReading[];
|
|
384
|
+
live_environment: {
|
|
385
|
+
url: string;
|
|
386
|
+
expires_at: string | null;
|
|
387
|
+
expires: string | null;
|
|
388
|
+
} | null;
|
|
389
|
+
base: UniverseDetail;
|
|
390
|
+
head: UniverseDetail;
|
|
391
|
+
}
|
|
392
|
+
export declare function cellDetail(run: VerificationRun, match: CellMatch, opts?: {
|
|
393
|
+
timeline?: boolean;
|
|
394
|
+
}): CellDetail;
|
|
395
|
+
export interface WatchCallbacks {
|
|
396
|
+
onEvent?: (event: VerificationRunEvent) => void;
|
|
397
|
+
onCellChange?: (chapterId: string, cell: RiskCell, previousStatus: string) => void;
|
|
398
|
+
onResolved?: (runId: string) => void;
|
|
399
|
+
}
|
|
400
|
+
export interface WatchResult {
|
|
401
|
+
loaded: LoadedRun;
|
|
402
|
+
timedOut: boolean;
|
|
403
|
+
}
|
|
404
|
+
/**
|
|
405
|
+
* Poll until the run reaches a terminal state. The terminal predicate is the
|
|
406
|
+
* run status alone — never a cell's: a cell legitimately sits in
|
|
407
|
+
* `provisioning` for minutes while sandbox creation retries.
|
|
408
|
+
*
|
|
409
|
+
* A stale marking (from the control plane, or from a disk read with no
|
|
410
|
+
* reachable server) is treated with suspicion rather than as terminal: a Vite
|
|
411
|
+
* restart re-creates the plugin's in-memory run registry while the old
|
|
412
|
+
* orchestrator can keep running in the same process, so "no orchestrator is
|
|
413
|
+
* driving this run" can be false. As long as run.json keeps advancing, the run
|
|
414
|
+
* is alive regardless of what the registry thinks; staleness is only accepted
|
|
415
|
+
* after a no-progress grace period.
|
|
416
|
+
*/
|
|
417
|
+
export declare function waitForTerminal(options: {
|
|
418
|
+
server: string;
|
|
419
|
+
repoRoot: string | null;
|
|
420
|
+
selector: string;
|
|
421
|
+
timeoutMs?: number;
|
|
422
|
+
intervalMs?: number;
|
|
423
|
+
callbacks?: WatchCallbacks;
|
|
424
|
+
}): Promise<WatchResult>;
|