@haystackeditor/cli 0.15.20 → 0.15.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -33,8 +33,8 @@ Most consumers of this CLI are coding agents. These invariants hold everywhere:
33
33
  - **Every JSON payload is versioned and schema'd.** Payloads carry
34
34
  `schema_version`; print the contract with `haystack schema <name>`
35
35
  (`haystack schema` lists all: `triage`, `pr`, `pr-status`, `inbox`, `ask`,
36
- `traces`, `submit`, `action`, `error`, `setup`). Schemas are JSON-Schema
37
- 2020-12 and CI-guarded against drift.
36
+ `traces`, `submit`, `action`, `cloud-verifier`, `error`, `setup`). Schemas
37
+ are JSON-Schema 2020-12 and CI-guarded against drift.
38
38
  - **One state vocabulary.** All state tokens are snake_case across every
39
39
  command: verdicts are `good_to_merge` / `needs_review` / `needs_input`;
40
40
  feed buckets are `analyzing`, `good_to_merge`, `issues`, `needs_assignment`,
@@ -65,6 +65,46 @@ haystack triage <ref> --json # poll findings later / after --no-wait
65
65
  `traces_list`, `traces_get`, `dismiss`, `mark_reviewed`, `undismiss`,
66
66
  `request_review`, `trigger_review`, and `schema`.
67
67
 
68
+ **`haystack verify mcp`** runs the cloud-verifier MCP server. Retained
69
+ application state is available through `verify_reopen`, `verify_refork`, and
70
+ `verify_cleanup`. Agents select state by an exact Archil fork ID or a stable
71
+ run/test/side coordinate; refork returns a durable child ID that can be
72
+ reopened or nested again without receiving provider credentials. The matching
73
+ CLI flow is:
74
+
75
+ ```bash
76
+ haystack verify reopen latest 3
77
+ # inspect or mutate the returned sandbox
78
+ haystack verify refork latest 3 attempt:fix-2
79
+ haystack verify reopen archil:<disk-id>:branch:<returned-child>
80
+ ```
81
+
82
+ **`haystack verify hosted`** is the authenticated production entry point for
83
+ exact GitHub commit pairs. It uses the repository-default saved account,
84
+ submits an idempotent run, and waits up to 35 minutes by default:
85
+
86
+ ```bash
87
+ haystack verify hosted start owner/repo \
88
+ --base <40-character-commit-sha> \
89
+ --head <40-character-commit-sha> \
90
+ --json
91
+ ```
92
+
93
+ Use `--no-wait` to return as soon as the workflow is queued. The result includes
94
+ an account-bound `next_command`; use it verbatim on machines with multiple
95
+ saved GitHub accounts. Repeating the same repository/base/head tuple reuses the
96
+ same production run by default. Pass a new explicit `--idempotency-key` only
97
+ when a fresh execution is intentional. A bounded wait that expires still emits
98
+ the `cloud-verifier` schema with `timed_out: true`, the last observed status,
99
+ and `next_command`, then exits 2. Read or resume a known run with:
100
+
101
+ ```bash
102
+ haystack verify hosted status cv_<48-lowercase-hex-characters> \
103
+ --account <login> \
104
+ --wait \
105
+ --json
106
+ ```
107
+
68
108
  **`haystack setup --json`** speaks NDJSON: events (`question`, `permission`,
69
109
  `progress`, `result`) on stdout, replies on stdin keyed by
70
110
  `requestID`/`permissionID`. `haystack schema setup` documents the full
@@ -0,0 +1,449 @@
1
+ export declare const DEFAULT_CONTROL_PLANE = "http://127.0.0.1:3000";
2
+ export declare const TERMINAL_RUN_STATUSES: Set<string>;
3
+ export interface UniverseEvidence {
4
+ evidenceId: string;
5
+ kind: string;
6
+ label: string;
7
+ summary: string;
8
+ value?: unknown;
9
+ artifactUrl?: string;
10
+ step?: {
11
+ ordinal: number;
12
+ name: string;
13
+ };
14
+ details?: Record<string, unknown>;
15
+ capturedAt?: string;
16
+ }
17
+ export interface UniverseRef {
18
+ universeId: string;
19
+ side: 'base' | 'head';
20
+ status: string;
21
+ snapshotId?: string;
22
+ forkId?: string;
23
+ executionId?: string;
24
+ stateProvider?: 'e2b-snapshot' | 'archil';
25
+ stateLayers?: Array<{
26
+ layerId: string;
27
+ kind: 'compute-filesystem' | 'application-filesystem';
28
+ provider: 'e2b-snapshot' | 'archil';
29
+ immutableBaseId: string;
30
+ mutableOverlayId?: string;
31
+ mountPath?: string;
32
+ consistencyPoint?: string;
33
+ }>;
34
+ sandboxId?: string;
35
+ browserUrl?: string;
36
+ evidence: UniverseEvidence[];
37
+ startedAt?: string;
38
+ completedAt?: string;
39
+ expiresAt?: string;
40
+ stateExpiresAt?: string;
41
+ }
42
+ export interface RiskCell {
43
+ riskCellId: string;
44
+ riskSourceId?: string;
45
+ ordinal: number;
46
+ title: string;
47
+ rationale?: string;
48
+ observableConsequence?: string;
49
+ activationConditions?: string[];
50
+ causalPath?: string[];
51
+ capabilityId?: string;
52
+ status: string;
53
+ severity?: string;
54
+ confidence?: number;
55
+ sourceEvidence?: Array<{
56
+ evidenceId: string;
57
+ file: string;
58
+ line?: number;
59
+ excerpt?: string;
60
+ relationship: string;
61
+ }>;
62
+ oracles?: Array<{
63
+ oracleId: string;
64
+ kind: string;
65
+ description?: string;
66
+ measurement: string;
67
+ threshold?: string;
68
+ }>;
69
+ durationMs?: number;
70
+ base: UniverseRef;
71
+ head: UniverseRef;
72
+ agentJudgment?: {
73
+ verdict: string;
74
+ explanation: string;
75
+ intentEvidence?: string;
76
+ };
77
+ infrastructureError?: {
78
+ reason: string;
79
+ message: string;
80
+ };
81
+ }
82
+ export interface BisectProbe {
83
+ commit: string;
84
+ shortSha: string;
85
+ subject: string;
86
+ position: number;
87
+ status: string;
88
+ wave?: number;
89
+ sandboxId?: string;
90
+ durationMs?: number;
91
+ error?: string;
92
+ }
93
+ export interface BisectResult {
94
+ status: string;
95
+ signal: string;
96
+ predicate: string;
97
+ totalCandidates: number;
98
+ probes: BisectProbe[];
99
+ culpritSha?: string;
100
+ culpritSubject?: string;
101
+ parentSha?: string;
102
+ incidentProbe?: BisectProbe & {
103
+ ref?: string;
104
+ label?: string;
105
+ };
106
+ }
107
+ export interface VerificationChapter {
108
+ chapterId: string;
109
+ executionMode?: 'risk_cells' | 'bisect';
110
+ analysisMode?: string;
111
+ label: string;
112
+ repository: string;
113
+ pullRequest: string;
114
+ pullRequestUrl?: string;
115
+ status: string;
116
+ riskCells: RiskCell[];
117
+ bisect?: BisectResult;
118
+ }
119
+ export interface VerificationRunEvent {
120
+ eventId: string;
121
+ at: string;
122
+ phase: string;
123
+ message: string;
124
+ chapterId?: string;
125
+ }
126
+ export interface VerificationRun {
127
+ schemaVersion: string;
128
+ runId: string;
129
+ status: string;
130
+ command: string;
131
+ createdAt: string;
132
+ updatedAt: string;
133
+ planSourceRunId?: string;
134
+ chapters: VerificationChapter[];
135
+ events: VerificationRunEvent[];
136
+ summary: {
137
+ total: number;
138
+ completed: number;
139
+ passed: number;
140
+ failed: number;
141
+ infraErrors: number;
142
+ inconclusive?: number;
143
+ };
144
+ }
145
+ export declare function resolveServer(explicit?: string): string;
146
+ export declare function setVerifyToken(token: string | undefined): void;
147
+ export declare function verifyAuthHeaders(): Record<string, string>;
148
+ export declare class VerifyUnauthorizedError extends Error {
149
+ constructor(server: string);
150
+ }
151
+ /**
152
+ * Find the repository that holds .haystack/verify — explicit flag, then
153
+ * HAYSTACK_VERIFY_REPO, then walking up from cwd. Null when nothing matches;
154
+ * server-backed commands still work without it.
155
+ */
156
+ export declare function discoverVerifyRoot(explicit?: string): string | null;
157
+ export declare function runsRoot(repoRoot: string): string;
158
+ export declare function runDir(repoRoot: string, runId: string): string;
159
+ /**
160
+ * Read when a run began from the record itself. updatedAt and mtime both move
161
+ * when cleanup or a migration rewrites old runs and therefore cannot define
162
+ * "latest".
163
+ */
164
+ export declare function runStartedAtMs(runJsonPath: string): number | null;
165
+ export declare function listRunDirs(repoRoot: string): Array<{
166
+ runId: string;
167
+ startedAt: number;
168
+ }>;
169
+ /** Same alias rules as the control plane: run id verbatim, `latest`, or a chapter id. */
170
+ export declare function resolveRunIdOnDisk(repoRoot: string, requested: string): string;
171
+ export declare function loadRunFromDisk(repoRoot: string, runId: string): VerificationRun | null;
172
+ export type ServerFetch = {
173
+ outcome: 'ok';
174
+ run: VerificationRun;
175
+ } | {
176
+ outcome: 'not_found';
177
+ } | {
178
+ outcome: 'unauthorized';
179
+ } | {
180
+ outcome: 'unreachable';
181
+ reason: string;
182
+ };
183
+ export declare function fetchRunFromServer(server: string, idOrAlias: string): Promise<ServerFetch>;
184
+ export interface LoadedRun {
185
+ run: VerificationRun;
186
+ source: 'server' | 'disk';
187
+ /**
188
+ * True when the run is non-terminal but nothing can still be driving it —
189
+ * a disk read with no reachable control plane. (The server performs the same
190
+ * marking itself by rewriting orphaned runs to `failed`.)
191
+ */
192
+ stale: boolean;
193
+ }
194
+ export declare function loadRunAuto(options: {
195
+ server: string;
196
+ repoRoot: string | null;
197
+ selector: string;
198
+ }): Promise<LoadedRun>;
199
+ export declare function isTerminal(run: VerificationRun): boolean;
200
+ export interface RunListRow {
201
+ run_id: string;
202
+ chapters: string[];
203
+ status: string;
204
+ age: string | null;
205
+ created_at?: string;
206
+ updated_at: string;
207
+ tests: number;
208
+ passed: number;
209
+ failed: number;
210
+ could_not_run: number;
211
+ culprit_sha: string | null;
212
+ }
213
+ export interface RunListResult {
214
+ rows: RunListRow[];
215
+ source: 'server' | 'disk';
216
+ total: number;
217
+ }
218
+ /**
219
+ * List runs, newest first. Prefers the control plane's GET /runs (present on
220
+ * servers with the operator API; required for a remote control plane) and
221
+ * falls back to enumerating the runs directory.
222
+ */
223
+ export declare function listRunRows(options: {
224
+ server: string;
225
+ repoRoot: string | null;
226
+ limit: number;
227
+ }): Promise<RunListResult>;
228
+ export declare function riskCellIdFor(chapterId: string, cellKey: string): string;
229
+ /**
230
+ * Recover the planner's cellKey from artifact filenames (`<cellKey>-<side>-…`).
231
+ * run.json stores only the hashed riskCellId; the key is friendlier to type.
232
+ */
233
+ /**
234
+ * Basename of an evidence artifactUrl. The executor URL-encodes the whole
235
+ * relative path (`<chapter>%2Fartifacts%2F<file>`), so decode before taking
236
+ * the last path segment.
237
+ */
238
+ export declare function artifactFileOf(artifactUrl: string): string;
239
+ /** Relative artifact path under the run's chapters/ dir, from an evidence URL. */
240
+ export declare function artifactRelOf(artifactUrl: string): string | null;
241
+ export declare function cellKeyOf(cell: RiskCell): string | null;
242
+ export declare function formatDurationMs(ms: number | undefined): string | null;
243
+ export declare function formatAge(iso: string | undefined): string | null;
244
+ export declare function formatUntil(iso: string | null | undefined): string | null;
245
+ export interface TestSummary {
246
+ ordinal: number;
247
+ cell_id: string;
248
+ cell_key: string | null;
249
+ title: string;
250
+ /** `passed` | `failed` | a live phase | `infra_error` (could not run — our defect, not a verdict). */
251
+ status: string;
252
+ severity: string | null;
253
+ verdict: string | null;
254
+ duration: string | null;
255
+ live_url: string | null;
256
+ live_expires_at: string | null;
257
+ live_expires: string | null;
258
+ }
259
+ export interface RunFinding {
260
+ chapter_id: string;
261
+ ordinal: number;
262
+ title: string;
263
+ severity: string | null;
264
+ verdict: string | null;
265
+ /** First paragraph of the judge's explanation; `verify show` / verify_cell has the whole thing. */
266
+ summary: string | null;
267
+ /** Failing oracle readings as "metric: base → head". */
268
+ changed_readings: string[];
269
+ live_url: string | null;
270
+ }
271
+ export interface RunSummary {
272
+ schema_version: '1.0.0';
273
+ run_id: string;
274
+ status: string;
275
+ stale: boolean;
276
+ source: 'server' | 'disk';
277
+ created_at: string;
278
+ updated_at: string;
279
+ age: string | null;
280
+ counts: {
281
+ total: number;
282
+ completed: number;
283
+ passed: number;
284
+ failed: number;
285
+ could_not_run: number;
286
+ };
287
+ /** One entry per failing test (and per isolated bisect culprit) — the "what broke" digest. */
288
+ findings: RunFinding[];
289
+ chapters: Array<{
290
+ chapter_id: string;
291
+ execution_mode: 'risk_cells' | 'bisect';
292
+ analysis_mode: string | null;
293
+ label: string;
294
+ repository: string;
295
+ pull_request: string;
296
+ status: string;
297
+ tests: TestSummary[];
298
+ bisect: null | {
299
+ status: string;
300
+ signal: string;
301
+ total_candidates: number;
302
+ probes_total: number;
303
+ probes_by_status: Record<string, number>;
304
+ culprit_sha: string | null;
305
+ culprit_subject: string | null;
306
+ parent_sha: string | null;
307
+ };
308
+ }>;
309
+ live_environments: Array<{
310
+ chapter_id: string;
311
+ ordinal: number;
312
+ title: string;
313
+ url: string;
314
+ expires_at: string | null;
315
+ expires: string | null;
316
+ }>;
317
+ /** Anything the verifier could not run or measure — reported as our failure, never as a finding. */
318
+ problems: string[];
319
+ last_event: {
320
+ at: string;
321
+ phase: string;
322
+ message: string;
323
+ } | null;
324
+ results_url: string;
325
+ }
326
+ /** Frozen-oracle readings for one cell, extracted from its assertion evidence. */
327
+ export declare function oracleReadingsOf(cell: RiskCell): OracleReading[];
328
+ export declare function summarizeRun(loaded: LoadedRun, server: string): RunSummary;
329
+ /**
330
+ * Exit-code policy shared by `verify watch` and callers of the MCP wait tool:
331
+ * 0 — everything that ran passed (a found bisect culprit counts as success);
332
+ * 1 — at least one test failed;
333
+ * 2 — the run itself broke, went stale, or some test could not run.
334
+ */
335
+ export declare function exitCodeFor(summary: RunSummary): number;
336
+ export interface CellMatch {
337
+ chapter: VerificationChapter;
338
+ cell: RiskCell;
339
+ }
340
+ /** Match a test by ordinal, riskCellId, planner cellKey, or title substring. */
341
+ export declare function findCell(run: VerificationRun, selector: string): CellMatch;
342
+ /**
343
+ * Resolve a cell for an operator action without using its free-text title.
344
+ * Ordinals are scoped by the exact run, while riskCellId and planner cellKey
345
+ * are stable structured identifiers carried by the run.
346
+ */
347
+ export declare function findCellByStableSelector(run: VerificationRun, selector: string): CellMatch;
348
+ export interface OracleReading {
349
+ metric: string;
350
+ comparison: string;
351
+ base: unknown;
352
+ head: unknown;
353
+ changed: boolean;
354
+ result: 'pass' | 'fail' | 'unmeasurable';
355
+ }
356
+ export interface TimelineEntry {
357
+ sequence: number | null;
358
+ offset_ms: number | null;
359
+ method: string | null;
360
+ path: string;
361
+ status: string;
362
+ channel: string | null;
363
+ event_name: string | null;
364
+ step: string | null;
365
+ }
366
+ export interface UniverseDetail {
367
+ universe_id: string;
368
+ status: string;
369
+ sandbox_id: string | null;
370
+ state_provider: 'e2b-snapshot' | 'archil' | null;
371
+ snapshot_id: string | null;
372
+ fork_id: string | null;
373
+ execution_id: string | null;
374
+ state_layers: NonNullable<UniverseRef['stateLayers']>;
375
+ network: TimelineEntry[];
376
+ sdk_log: string[];
377
+ driver_failures: string[];
378
+ artifacts: Array<{
379
+ kind: string;
380
+ label: string;
381
+ url: string;
382
+ file: string;
383
+ }>;
384
+ }
385
+ export interface CellDetail {
386
+ schema_version: '1.0.0';
387
+ run_id: string;
388
+ chapter_id: string;
389
+ ordinal: number;
390
+ cell_id: string;
391
+ cell_key: string | null;
392
+ title: string;
393
+ status: string;
394
+ severity: string | null;
395
+ duration: string | null;
396
+ finding: string | null;
397
+ verdict: string | null;
398
+ explanation: string | null;
399
+ intent_evidence: string | null;
400
+ infrastructure_error: {
401
+ reason: string;
402
+ message: string;
403
+ } | null;
404
+ rationale: string | null;
405
+ observable_consequence: string | null;
406
+ oracle_readings: OracleReading[];
407
+ readings_changed: OracleReading[];
408
+ live_environment: {
409
+ url: string;
410
+ expires_at: string | null;
411
+ expires: string | null;
412
+ } | null;
413
+ base: UniverseDetail;
414
+ head: UniverseDetail;
415
+ }
416
+ export declare function cellDetail(run: VerificationRun, match: CellMatch, opts?: {
417
+ timeline?: boolean;
418
+ }): CellDetail;
419
+ export interface WatchCallbacks {
420
+ onEvent?: (event: VerificationRunEvent) => void;
421
+ onCellChange?: (chapterId: string, cell: RiskCell, previousStatus: string) => void;
422
+ onResolved?: (runId: string) => void;
423
+ }
424
+ export interface WatchResult {
425
+ loaded: LoadedRun;
426
+ timedOut: boolean;
427
+ endReason: 'terminal' | 'stale' | 'timeout';
428
+ }
429
+ /**
430
+ * Poll until the run reaches a terminal state. The terminal predicate is the
431
+ * run status alone — never a cell's: a cell legitimately sits in
432
+ * `provisioning` for minutes while sandbox creation retries.
433
+ *
434
+ * A stale marking (from the control plane, or from a disk read with no
435
+ * reachable server) is treated with suspicion rather than as terminal: a Vite
436
+ * restart re-creates the plugin's in-memory run registry while the old
437
+ * orchestrator can keep running in the same process, so "no orchestrator is
438
+ * driving this run" can be false. As long as run.json keeps advancing, the run
439
+ * is alive regardless of what the registry thinks; staleness is only accepted
440
+ * after a no-progress grace period.
441
+ */
442
+ export declare function waitForTerminal(options: {
443
+ server: string;
444
+ repoRoot: string | null;
445
+ selector: string;
446
+ timeoutMs?: number;
447
+ intervalMs?: number;
448
+ callbacks?: WatchCallbacks;
449
+ }): Promise<WatchResult>;