@haystackeditor/cli 0.29.0 → 0.30.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -118,7 +118,9 @@ answered or finished (the bugs
118
118
  it found are in the output) or is still running under `--no-wait`; 2 when it ended
119
119
  without finishing (stopped early, or cancelled) or the machines it used could
120
120
  not be proven shut down;
121
- 1 when the command failed or no crawl could be started.
121
+ 1 when the command failed or no crawl could be started; 3 when onboarding is
122
+ blocked; 4 for the handshake (run it again with `--intent`); 5 when the app
123
+ copy is broken and the change could not be tested.
122
124
 
123
125
  **The stop hook.** `haystack init` installs a Claude Code Stop hook that runs
124
126
  `haystack verify precompute --hook` whenever an agent turn stops. It captures
@@ -15,6 +15,9 @@ export const CRAWL_WAIT_MAX_MS = 25_000;
15
15
  * (how many times the copy's calls reached the stand-in across the crawl's preparation and all its replays, not user actions). */
16
16
  export const CRAWL_MAX_STAND_IN_ROWS = 200;
17
17
  export const CRAWL_MAX_STAND_IN_LINES = 20;
18
+ /** Amendment 22: each health evidence list holds at most this many lines, and the commonest difference lines at most this many. */
19
+ export const CRAWL_HEALTH_MAX_LINES = 40;
20
+ export const CRAWL_HEALTH_MAX_DIFFERENCES = 8;
18
21
  /** Amendment 18: the brief's bounds. */
19
22
  export const CRAWL_BRIEF_MAX_FUNCTIONS = 2000;
20
23
  export const CRAWL_BRIEF_MAX_GROUPS = 2000;
@@ -11,7 +11,7 @@ export const STATUS_WORDS = {
11
11
  cancelled: 'cancelled',
12
12
  };
13
13
  export const STAGE_WORDS = {
14
- reconcile: 'cleaning up after an earlier try',
14
+ reconcile: 'checking for leftovers of an earlier try',
15
15
  source: 'rebuilding your change',
16
16
  worlds: 'building and starting the app with and without your change',
17
17
  blast: 'waiting for the analysis of what your change touches',
@@ -66,6 +66,38 @@ export function currentStep(view) {
66
66
  export function results(view) {
67
67
  return view.manifest ?? (view.answer?.status === 'answered' ? view.answer : null);
68
68
  }
69
+ /** CRAWL-V1 amendment 22: the copy of the app the crawl ran was itself broken, so it tested nothing; null when it was not judged
70
+ * broken (or the results carry no health). */
71
+ export function brokenCopy(view) {
72
+ const health = results(view)?.health;
73
+ return health !== undefined && health.status === 'app-broken' ? health : null;
74
+ }
75
+ /** What `haystack verify` says first of a broken copy. */
76
+ export const BROKEN_COPY_HEADLINE = "We couldn't test your change: the app copy is broken.";
77
+ /** What the coding agent does about a broken copy: the app's own setup says what the errors need; until a broken copy updates its
78
+ * setup (onboarding amendment 38 updates it only from a check that could not start the app), that is sent to the Haystack team
79
+ * (`haystack feedback`), and the user is told the change was not tested. */
80
+ export const BROKEN_COPY_STEPS = [
81
+ "Find what these errors point to in the repository's own setup (docker compose files, env examples, docs): the services, settings and data the app needs to run.",
82
+ 'Send what you found with `haystack feedback "<what the app needs to run>"`: a copy that starts but is broken does not update its setup yet.',
83
+ 'Tell your user this check did not test the change; do not read it as passing.',
84
+ ];
85
+ /** Text that reads as an error, for the lines a broken copy's start screen shows (every one is in --raw). */
86
+ const ERROR_LIKE = /\b(?:errors?|fail(?:ed|s|ure)?|unsupported|exception|denied|refused|unavailable|unauthori[sz]ed|forbidden|invalid|cannot|can't|could not|unable|timed out|timeout|not found|internal server)\b/iu;
87
+ /** What a broken copy's evidence says, as lines: what its start screen shows (its messages, then its texts that read as errors),
88
+ * and the requests that failed and the console errors logged with nobody clicking, each with the builds it was in. */
89
+ export function brokenCopyEvidence(health) {
90
+ const { evidence } = health;
91
+ const shows = [...new Set([...evidence.messages, ...evidence.texts.filter(text => ERROR_LIKE.test(text))])];
92
+ const idle = (kind) => {
93
+ const of = (lines) => lines.filter(line => line.startsWith(`${kind}:`)).map(line => line.slice(kind.length + 1));
94
+ const old = of(evidence.background.old);
95
+ const fresh = of(evidence.background.new);
96
+ return [...new Set([...old, ...fresh])].map(entry => `${entry} (${old.includes(entry) && fresh.includes(entry) ? 'both builds'
97
+ : old.includes(entry) ? 'without your change only' : 'with your change only'})`);
98
+ };
99
+ return { shows, failedRequests: idle('failed'), consoleErrors: idle('errors') };
100
+ }
69
101
  /** The status the command reports: a running crawl that has answered (amendment 9: its budget is up and its work is
70
102
  * done; only its shutdown remains) reads as completed. */
71
103
  export function reportedStatus(view) {
@@ -76,9 +108,21 @@ export function headline(view) {
76
108
  const found = bugs > 0 ? ` It found ${plural(bugs, 'bug')} before it stopped.` : '';
77
109
  switch (reportedStatus(view)) {
78
110
  case 'completed':
111
+ // Amendment 22: a broken copy tested nothing, whatever it found.
112
+ if (brokenCopy(view) !== null)
113
+ return BROKEN_COPY_HEADLINE;
79
114
  return bugs > 0 ? `Found ${plural(bugs, 'bug')} in your change.` : 'No bugs found in your change.';
80
115
  case 'incomplete': {
81
116
  const error = results(view)?.error ?? view.error;
117
+ // Onboarding rule 18 (amendment 38): which world did not start says whose it is, and what Haystack does about it.
118
+ if (error?.code === 'world-build-failed' && view.worlds.base === null) {
119
+ return 'The crawl could not finish: the app did not build or start at the base commit, on its setup. Haystack updates the '
120
+ + 'setup for it and runs this check again (`haystack verify onboarding` shows the update).';
121
+ }
122
+ if (error?.code === 'world-build-failed' && view.worlds.head === null) {
123
+ return 'The crawl could not finish: your change did not build or start, though the base did. Haystack checks whether the change '
124
+ + 'needs something new in the setup and, if it does, adds it and runs this check again (`haystack verify onboarding` shows it).';
125
+ }
82
126
  return `The crawl could not finish: ${error ? errorWords(error) : 'no reason was recorded'}.${found}`;
83
127
  }
84
128
  case 'prepared':
@@ -137,6 +181,9 @@ export function verifyReport(view) {
137
181
  } : null,
138
182
  notFinishedInTime: explored ? published.notFinishedInTime : 0,
139
183
  outsideServiceGaps: explored ? published.standIns?.gaps ?? [] : [],
184
+ health: !explored || published.health === undefined ? null : published.health.status === 'app-broken'
185
+ ? { status: 'app-broken', reason: published.health.reason, ...brokenCopyEvidence(published.health), whatToDo: [...BROKEN_COPY_STEPS] }
186
+ : { status: published.health.status, reason: published.health.reason, shows: [], failedRequests: [], consoleErrors: [], whatToDo: null },
140
187
  error: error ? { code: error.code, message: error.message } : null,
141
188
  machinesNotProvenShutDown: TERMINAL.has(view.status) ? view.totals.cleanupUnproven : 0,
142
189
  };
@@ -145,6 +192,12 @@ export function verifyReport(view) {
145
192
  export function reportMarkdown(report) {
146
193
  const lines = [`# ${report.headline}`, '',
147
194
  `Change: ${report.title}`, `Check ${report.runId} of ${report.repository}: base ${report.baseCommit.slice(0, 12)}, change ${report.workCommit.slice(0, 12)}.`];
195
+ // Amendment 22: a broken copy is said first, with why and what to do, whatever became of the run after it answered (a cancelled
196
+ // or failed run's published health still says it); its findings follow, marked.
197
+ const broken = report.health?.status === 'app-broken' ? report.health : null;
198
+ if (broken !== null) {
199
+ lines.splice(1, 0, '', broken.reason, ...(broken.shows.length ? ['', 'The start screen shows:', ...broken.shows.map(line => `- ${line}`)] : []), ...(broken.failedRequests.length ? ['', 'Requests that failed with nobody clicking:', ...broken.failedRequests.map(line => `- ${line}`)] : []), ...(broken.consoleErrors.length ? ['', 'Console errors with nobody clicking:', ...broken.consoleErrors.map(line => `- ${line}`)] : []), '', '## What to do', ...(broken.whatToDo ?? []).map((step, index) => `${index + 1}. ${step}`));
200
+ }
148
201
  if (report.error && report.status !== 'completed')
149
202
  lines.push(`Details: ${report.error.message}`);
150
203
  if (report.neverRan === null) {
@@ -161,7 +214,9 @@ export function reportMarkdown(report) {
161
214
  if (spot.reachedBy)
162
215
  lines.push(` - reached by: ${[spot.reachedBy.start, ...spot.reachedBy.steps].join(' > ')}${spot.reachedBy.page ? ` (page ${spot.reachedBy.page})` : ''}`);
163
216
  }
164
- lines.push('', '## What the crawl found');
217
+ lines.push('', broken === null ? '## What the crawl found' : '## What the crawl found in the broken copy');
218
+ if (broken !== null)
219
+ lines.push('The copy was broken, so these differences are the environment\'s, not evidence about the change.');
165
220
  if (report.findings.length === 0)
166
221
  lines.push('No confirmed differences between the app with and without the change.');
167
222
  report.findings.forEach((finding, index) => {
@@ -113,6 +113,23 @@ export function parseOnboardingStatus(value, expected) {
113
113
  if (ready ? !(typeof value.recordPublishedAt === 'string' && !Number.isNaN(Date.parse(value.recordPublishedAt)))
114
114
  : value.recordPublishedAt !== null)
115
115
  invalid('its record time');
116
+ // Rule 18 (amendment 38): a ready record says what it updated, and the update of it under way, if any.
117
+ const recordUpdate = value.recordUpdate;
118
+ if (!(recordUpdate === null || (ready && isRecord(recordUpdate) && typeof recordUpdate.of === 'string' && SHA256.test(recordUpdate.of)
119
+ && typeof recordUpdate.crawlRunId === 'string' && (recordUpdate.side === 'base' || recordUpdate.side === 'head')
120
+ && typeof recordUpdate.workCommit === 'string' && COMMIT.test(recordUpdate.workCommit)
121
+ && Array.isArray(recordUpdate.changes) && recordUpdate.changes.every(change => isRecord(change) && typeof change.field === 'string'
122
+ && (change.name === null || typeof change.name === 'string') && (change.kind === 'added' || change.kind === 'changed')
123
+ && Array.isArray(change.evidence) && change.evidence.every(id => typeof id === 'string')))))
124
+ invalid('its record update');
125
+ const update = value.update;
126
+ if (!(update === null || (ready && isRecord(update) && typeof update.onboardRunId === 'string' && ONBOARD_RUN_ID.test(update.onboardRunId)
127
+ && (update.side === 'base' || update.side === 'head') && typeof update.crawlRunId === 'string'
128
+ && (update.state === 'updating' || update.state === 'blocked' || update.state === 'stopped')
129
+ && (update.state === 'blocked') === (update.block !== null))))
130
+ invalid('its update');
131
+ if (isRecord(update) && update.block !== null)
132
+ parseOnboardingBlock(update.block);
116
133
  if (value.block !== null)
117
134
  parseOnboardingBlock(value.block);
118
135
  const checkpoint = value.checkpoint;
@@ -313,6 +330,9 @@ export function formatOnboarding(status) {
313
330
  lines.push('', ...stages);
314
331
  if (status.block)
315
332
  lines.push('', ...blockLines(status.block));
333
+ const updates = updateLines(status);
334
+ if (updates.length)
335
+ lines.push('', ...updates);
316
336
  if (status.dataAbsences.length)
317
337
  lines.push('', ...status.dataAbsences.map(absence => chalk.yellow(`Note: ${safe(absence)}`)));
318
338
  if (status.versionChoices.length)
@@ -325,6 +345,37 @@ export function formatOnboarding(status) {
325
345
  lines.push('', ...unreviewed.map(note => chalk.yellow(`Note: ${safe(note)}`)));
326
346
  return lines.join('\n');
327
347
  }
348
+ /** Rule 18 (amendment 38): what the ready setup's last update was, and the update of it under way or ended without one: one
349
+ * line each, and a blocked update's block (what it needs, and what to tell Haystack). */
350
+ export function updateLines(status) {
351
+ const lines = [];
352
+ const failed = (side) => side === 'base' ? 'the app did not start at its base commit' : 'the change did not start';
353
+ if (status.recordUpdate) {
354
+ const changes = status.recordUpdate.changes.map(change => `${change.kind} ${change.field === 'env' ? 'variable' : change.field}`
355
+ + `${change.name === null ? '' : ` ${change.name}`}${change.evidence.length ? ` (from ${change.evidence.join(', ')})` : ''}`);
356
+ lines.push(`Its setup was updated in place after check ${status.recordUpdate.crawlRunId}, where ${failed(status.recordUpdate.side)}`
357
+ + `${changes.length ? `: ${changes.join('; ')}` : ''}.`);
358
+ }
359
+ const update = status.update;
360
+ if (update === null)
361
+ return lines;
362
+ if (update.state === 'updating') {
363
+ lines.push(`Updating the setup (${update.onboardRunId}): check ${update.crawlRunId} found ${failed(update.side)}. The check runs `
364
+ + 'again on the updated setup when it is ready.');
365
+ }
366
+ else if (update.state === 'stopped') {
367
+ lines.push(`The update of the setup for check ${update.crawlRunId} (${update.onboardRunId}) stopped before finishing; the next check `
368
+ + 'that fails the same way starts it again.');
369
+ }
370
+ else {
371
+ lines.push(update.side === 'head'
372
+ ? chalk.bold(`Finding: check ${update.crawlRunId}'s change does not start, and no change to the setup started it.`)
373
+ : chalk.bold(`Haystack could not update the setup for check ${update.crawlRunId}, where the app did not start at its base commit.`));
374
+ if (update.block)
375
+ lines.push(...blockLines(update.block));
376
+ }
377
+ return lines;
378
+ }
328
379
  /** Rule 2 (amendment 34): the repository's one setup, which base it was made at and how long ago, in one sentence. */
329
380
  export function readyLine(status, now = Date.now()) {
330
381
  if (status.recordBaseCommit === null || status.recordPublishedAt === null)
@@ -339,12 +390,16 @@ function age(ms) {
339
390
  : [Math.floor(minutes / 1_440), 'day'];
340
391
  return count === 0 ? 'just now' : `${count} ${unit}${count === 1 ? '' : 's'} ago`;
341
392
  }
342
- /** Rule 17 (amendment 32): an onboarding the independent review skipped, as the sentence every result shows; none otherwise. */
393
+ /** Rule 17 (amendments 32 and 39): an onboarding the independent review skipped, as the sentence every result shows, saying why
394
+ * for each reason the contract names (a reason this CLI does not know yet is still shown, as the reason the server gave); none
395
+ * otherwise. */
343
396
  export function reviewNotes(review) {
344
397
  if (review?.verdict !== 'skipped')
345
398
  return [];
346
- return [`This onboarding was NOT reviewed: the independent review that checks the setup did not cheat is turned off for this `
347
- + `team (${review.reason}).`];
399
+ const why = review.reason === 'internal-test-tenant' ? 'is turned off for this team'
400
+ : review.reason === 'internal-test-requester' ? 'is turned off for onboardings Haystack\'s own people or automation start'
401
+ : 'did not run';
402
+ return [`This onboarding was NOT reviewed: the independent review that checks the setup did not cheat ${why} (${review.reason}).`];
348
403
  }
349
404
  /** Rule 15(g), amendment 3: a stand-in gap as the sentence every result shows. */
350
405
  export function standInGapNotes(gaps) {
@@ -29,9 +29,9 @@ import { findGitRoot } from '../utils/hooks.js';
29
29
  import { resolveAuthContext } from '../utils/auth.js';
30
30
  import { classifyHttpError, readWithRetries, SERVICE_SILENCE_LIMIT_MS, ServiceSilentError, transientReadFailure, repoApiPath, } from '../utils/haystack-api.js';
31
31
  import { GATEWAY_TIMEOUT_MS, gatewayFetch } from './case-batch.js';
32
- import { CRAWL_BRIEF_MAX_FUNCTIONS, CRAWL_BRIEF_MAX_FLOW_NAME_CHARS, CRAWL_BRIEF_MAX_GROUPS, CRAWL_BRIEF_MAX_PRODUCTION_ROWS, CRAWL_BUDGET_MAX_MS, CRAWL_BUDGET_MIN_MS, CRAWL_MAX_FINDINGS, CRAWL_MAX_FINDING_STEPS, CRAWL_MAX_IDEAS, CRAWL_MAX_STAND_IN_LINES, CRAWL_MAX_STAND_IN_ROWS, CRAWL_MAX_TEXT_CHARS, CRAWL_WAIT_MAX_MS, CRAWL_POOLS, } from './crawl-contract.js';
32
+ import { CRAWL_BRIEF_MAX_FUNCTIONS, CRAWL_BRIEF_MAX_FLOW_NAME_CHARS, CRAWL_BRIEF_MAX_GROUPS, CRAWL_BRIEF_MAX_PRODUCTION_ROWS, CRAWL_BUDGET_MAX_MS, CRAWL_BUDGET_MIN_MS, CRAWL_HEALTH_MAX_DIFFERENCES, CRAWL_HEALTH_MAX_LINES, CRAWL_MAX_FINDINGS, CRAWL_MAX_FINDING_STEPS, CRAWL_MAX_IDEAS, CRAWL_MAX_STAND_IN_LINES, CRAWL_MAX_STAND_IN_ROWS, CRAWL_MAX_TEXT_CHARS, CRAWL_WAIT_MAX_MS, CRAWL_POOLS, } from './crawl-contract.js';
33
33
  import { captureCheckout, crawlUnavailableText, EXPLICIT_WALL_MS, normalModeFailure, postPrecomputeCapture, } from './verify-precompute.js';
34
- import { SPOT_WORDS, TERMINAL, VERDICT_WORDS, VERDICTS, currentStep, errorWords, headline, plural, reportedStatus, results, verifyReport, } from './crawl-report.js';
34
+ import { BROKEN_COPY_STEPS, SPOT_WORDS, TERMINAL, VERDICT_WORDS, VERDICTS, brokenCopy, brokenCopyEvidence, currentStep, errorWords, headline, plural, reportedStatus, results, verifyReport, } from './crawl-report.js';
35
35
  import { formatOnboarding, onboardingExitCode, readOnboardingStatus, readyLine, reportOnboardingState, reviewNotes, standInGapNotes, waitForOnboarding, } from './verify-onboarding.js';
36
36
  import { formatCapture, preVerifyCapture, readCaptureWindows } from './capture-brief.js';
37
37
  import { flowDetail, flowSummaries } from './verify-flows.js';
@@ -183,9 +183,30 @@ function checkResults(value, view, what) {
183
183
  invalid('its ideas');
184
184
  if (value.brief !== undefined)
185
185
  checkBrief(value.brief);
186
+ if (value.health !== undefined)
187
+ checkHealth(value.health);
186
188
  if (value.error !== null)
187
189
  checkError(value.error, 'its error');
188
190
  }
191
+ /** Amendment 22: the health of the app copy, as the worker bounds it: `not-judged` with its reason, or a judgment. */
192
+ function checkHealth(value) {
193
+ const lines = (item) => isStrings(item, CRAWL_HEALTH_MAX_LINES);
194
+ if (!isRecord(value) || typeof value.reason !== 'string' || value.reason.length === 0)
195
+ invalid('its app health');
196
+ if (value.status === 'not-judged')
197
+ return;
198
+ const evidence = value.evidence;
199
+ const images = value.images;
200
+ if ((value.status !== 'ok' && value.status !== 'app-broken') || typeof value.working !== 'number' || !(value.working >= 0 && value.working <= 1)
201
+ || !isMember(value.tested, ['tested', 'app_broken', 'not_reached', 'unclear'])
202
+ || !isRecord(evidence) || !lines(evidence.texts) || !lines(evidence.messages) || !lines(evidence.overlays)
203
+ || !isRecord(evidence.background) || !lines(evidence.background.old) || !lines(evidence.background.new)
204
+ || !Array.isArray(evidence.commonDifferences) || evidence.commonDifferences.length > CRAWL_HEALTH_MAX_DIFFERENCES
205
+ || !evidence.commonDifferences.every(row => isRecord(row) && typeof row.line === 'string' && isCount(row.count))
206
+ || !isRecord(images) || (images.old !== null && typeof images.old !== 'string')
207
+ || (images.new !== null && typeof images.new !== 'string'))
208
+ invalid('its app health');
209
+ }
189
210
  /** Amendment 18: a brief's shape and bounds, as the worker checks them. */
190
211
  function checkBrief(value) {
191
212
  const text = (item, empty = false) => typeof item === 'string' && (empty || item.length > 0) && [...item].length <= CRAWL_MAX_TEXT_CHARS;
@@ -388,6 +409,15 @@ function findingLines(finding, position) {
388
409
  export function formatCrawl(view) {
389
410
  const status = reportedStatus(view);
390
411
  const lines = [chalk.bold(headline(view))];
412
+ // Amendment 22: a broken copy tested nothing: said first (the headline), then why, what both builds showed, and what to do. Its
413
+ // findings still follow, marked as the broken copy's.
414
+ // Amendment 22: whatever became of the run after it answered; the exit code keeps the run's own status (crawlExitCode).
415
+ const broken = brokenCopy(view);
416
+ if (broken !== null) {
417
+ const { shows, failedRequests, consoleErrors } = brokenCopyEvidence(broken);
418
+ const listed = (title, items) => (items.length ? [` ${title}`, ...items.map(item => ` ${safe(item)}`)] : []);
419
+ lines.push(` ${safe(broken.reason)}`, ...listed('The start screen shows:', shows), ...listed('Requests that failed with nobody clicking:', failedRequests), ...listed('Console errors with nobody clicking:', consoleErrors), chalk.bold(' What to do:'), ...BROKEN_COPY_STEPS.map((step, index) => ` ${index + 1}. ${step}`), '');
420
+ }
391
421
  const published = results(view);
392
422
  const error = published?.error ?? view.error;
393
423
  if (error && status !== 'completed')
@@ -411,7 +441,9 @@ export function formatCrawl(view) {
411
441
  lines.push(' No changed spots were recorded.');
412
442
  for (const spot of published.reach.spots)
413
443
  lines.push(...spotLines(spot));
414
- lines.push('', chalk.bold('What the crawl found'));
444
+ lines.push('', chalk.bold(broken === null ? 'What the crawl found' : 'What the crawl found in the broken copy'));
445
+ if (broken !== null)
446
+ lines.push(chalk.yellow(' The copy was broken, so these differences are the environment\'s, not evidence about your change.'));
415
447
  // Bugs first, then the undecided, then what the judge called intended.
416
448
  const findings = VERDICTS.flatMap(verdict => published.findings.filter(finding => finding.verdict === verdict));
417
449
  if (findings.length === 0)
@@ -475,12 +507,16 @@ export function formatCrawl(view) {
475
507
  * is done and only its shutdown remains), or still running with --no-wait;
476
508
  * 2 ended without finishing (incomplete, cancelled) or finished with cleanup
477
509
  * unproven (an answered crawl has not finished its cleanup, so it is not judged); bugs
478
- * found never change it. The command's own failures exit 1 elsewhere. */
510
+ * found never change it. 5 (amendment 22): it finished, but the copy of the app was
511
+ * broken, so the change was not tested; it wins over unproven cleanup, since what the
512
+ * agent does next is fix how the app runs. The command's own failures exit 1 elsewhere. */
479
513
  export { verifyReport };
480
514
  export function crawlExitCode(view) {
481
515
  const status = reportedStatus(view);
482
516
  if (!TERMINAL.has(status))
483
517
  return 0;
518
+ if (status === 'completed' && brokenCopy(view) !== null)
519
+ return 5;
484
520
  if (status !== 'completed' || (TERMINAL.has(view.status) && view.totals.cleanupUnproven > 0))
485
521
  return 2;
486
522
  return 0;
@@ -651,12 +687,13 @@ function staleCrawl(found, identity) {
651
687
  return found.mode === 'prepare' || found.status === 'prepared' ? 'prepare-only' : null;
652
688
  }
653
689
  function staleWords(found, stale) {
690
+ const kind = found.mode === 'prepare' ? 'preparation' : 'crawl';
654
691
  switch (stale) {
655
- case 'retitled': return `The last crawl of this code ran under another title, "${safe(found.changeTitle)}" (${found.runId})`;
656
- case 'other-time': return `The last crawl of this code was asked for ${timeWords(found.budgetMs)} (${found.runId})`;
657
- case 'cancelled': return `This change's last crawl was cancelled (${found.runId})`;
658
- case 'incomplete': return `This change's last crawl stopped before finishing (${found.runId})`;
659
- case 'cancelling': return `This change's crawl is being cancelled (${found.runId})`;
692
+ case 'retitled': return `The last ${kind} of this code ran under another title, "${safe(found.changeTitle)}" (${found.runId})`;
693
+ case 'other-time': return `The last ${kind} of this code was asked for ${timeWords(found.budgetMs)} (${found.runId})`;
694
+ case 'cancelled': return `This change's last ${kind} was cancelled (${found.runId})`;
695
+ case 'incomplete': return `This change's last ${kind} stopped before finishing (${found.runId})`;
696
+ case 'cancelling': return `This change's ${kind} is being cancelled (${found.runId})`;
660
697
  case 'prepare-only': return found.status === 'prepared'
661
698
  ? `Your change was built and frozen ahead of time (${found.runId})`
662
699
  : `Your change is being built and frozen ahead of time (${found.runId})`;
@@ -671,6 +708,8 @@ async function submitCapture(capture, token, found) {
671
708
  // sender does: admission is idempotent for one capture, so a repeat can only find what the first created.
672
709
  const acknowledgment = await readWithRetries(() => postPrecomputeCapture(capture, token, Date.now() + EXPLICIT_WALL_MS)).catch((error) => { throw normalModeFailure(error); });
673
710
  const crawl = acknowledgment.crawl;
711
+ // Amendment 12: the handshake submits mode 'prepare', which builds and prepares the change and never starts a crawl.
712
+ const preparing = capture.derivation.request.crawl?.mode === 'prepare';
674
713
  const before = found === null ? 'No crawl existed for this change' : staleWords(found.view, found.stale);
675
714
  if (crawl === undefined) {
676
715
  return { runId: null, onboarding: null,
@@ -682,22 +721,22 @@ async function submitCapture(capture, token, found) {
682
721
  }
683
722
  if (crawl.status === 'onboarding') {
684
723
  return { runId: null, onboarding: 'onboarding', line: `${before}. The app is not onboarded yet: `
685
- + `the service is preparing it (${crawl.onboardRunId}) and runs the crawl when it is ready.` };
724
+ + `the service is preparing it (${crawl.onboardRunId}) and continues with this change when it is ready.` };
686
725
  }
687
726
  if (crawl.status === 'onboarding-blocked') {
688
727
  return { runId: null, onboarding: 'onboarding-blocked', line: `${before}, and the app's onboarding is blocked.` };
689
728
  }
690
729
  if (found === null) {
691
730
  return { runId: crawl.runId, onboarding: null, line: crawl.status === 'queued'
692
- ? `No crawl existed for this change yet; started one (${crawl.runId}).`
693
- : `The service already had a crawl for this change (${crawl.runId}).` };
731
+ ? `No crawl existed for this change yet; ${preparing ? 'started preparing it' : 'started one'} (${crawl.runId}).`
732
+ : `The service already had a check for this change (${crawl.runId}).` };
694
733
  }
695
734
  if (!anotherCrawl(found.stale) && crawl.runId === found.view.runId) {
696
735
  return { runId: crawl.runId, onboarding: null, line: found.stale === 'prepare-only'
697
736
  ? `${before}; checking it now, reusing what was built.` : `${before}; asked the service to run it again.` };
698
737
  }
699
738
  return { runId: crawl.runId, onboarding: null, line: crawl.status === 'queued'
700
- ? `${before}; started one for this change (${crawl.runId}).`
739
+ ? `${before}; ${preparing ? 'started preparing the change again' : 'started one for this change'} (${crawl.runId}).`
701
740
  : `${before}; the service already had one for this change (${crawl.runId}).` };
702
741
  }
703
742
  function checkedChange(baseCommit, workCommit) {
@@ -980,16 +1019,24 @@ export function formatBrief(brief) {
980
1019
  }
981
1020
  return lines.join('\n');
982
1021
  }
983
- /** The handshake's question, asked after everything it shows. */
984
1022
  /** The command to steer with; it keeps the options the handshake was given that choose the crawl (where, and for how long). */
985
1023
  export function steerCommand(options) {
986
1024
  const kept = [options.repo ? ` --repo ${options.repo}` : '', options.account ? ` --account ${options.account}` : '',
987
1025
  options.minutes ? ` --minutes ${options.minutes}` : ''].join('');
988
1026
  return `haystack verify${kept} --intent "<what you were asked to do>" --idea "<something to try>" [--idea ...]`;
989
1027
  }
990
- const steerLines = (command) => ['',
991
- 'Next: pick the ideas worth trying, add your own, and say what you were asked to do (the task in the user\'s words, not what your code does):',
992
- ` ${command}`, 'Your ideas are explored first; the ideas above are explored in every crawl too.'];
1028
+ /** The handshake's question, asked after everything it shows; it speaks of "the ideas above" only when it showed some. */
1029
+ export function steerLines(command, shownIdeas) {
1030
+ return shownIdeas
1031
+ ? ['', 'Next: pick the ideas worth trying, add your own, and say what you were asked to do (the task in the user\'s words, not what your code does):',
1032
+ ` ${command}`, 'Your ideas are explored first; the ideas above are explored in every crawl too.']
1033
+ : ['', 'Next: say what you were asked to do (the task in the user\'s words, not what your code does) and what to try:',
1034
+ ` ${command}`, 'Your ideas are explored first.'];
1035
+ }
1036
+ /** Why a run stopped, as its report says it: the published error, else the run's own. */
1037
+ function stoppedBecause(view) {
1038
+ return results(view)?.error ?? view.error ?? null;
1039
+ }
993
1040
  /** The handshake (Akshay, 10/7): a plain `haystack verify` on a change no crawl has started for shows what the change touches,
994
1041
  * how production and real users run it, and Haystack's own ideas, then asks the coding agent to run `haystack verify` again
995
1042
  * with --intent (and any --idea). It submits the capture in mode 'prepare' when no run of it is preparing, prepared or
@@ -999,8 +1046,10 @@ async function handshake(capture, identity, change, found, stale, token, options
999
1046
  let view = found;
1000
1047
  let onboarding = null;
1001
1048
  const finish = (brief, captured, flows) => {
1049
+ const error = view === null ? null : stoppedBecause(view);
1002
1050
  printVerifyJson(options, change, onboarding, null, {
1003
- run: view === null ? null : { runId: view.runId, status: view.status }, brief, capture: captured, flows, next: steerCommand(options),
1051
+ run: view === null ? null : { runId: view.runId, status: view.status, error: error ? { code: error.code, message: error.message } : null },
1052
+ brief, capture: captured, flows, next: steerCommand(options),
1004
1053
  });
1005
1054
  };
1006
1055
  if (view === null || (stale !== null && stale !== 'prepare-only')) {
@@ -1055,12 +1104,18 @@ async function handshake(capture, identity, change, found, stale, token, options
1055
1104
  console.log(`What your change touches is not worked out yet (${currentStep(view)}); run \`haystack verify\` again to see it, or steer now.`);
1056
1105
  }
1057
1106
  else {
1058
- console.log(`Run ${view.runId} ended (${view.status}) without a brief${view.error ? `: ${errorWords(view.error)}` : ''}; steer the crawl without one.`);
1107
+ // A failed prepare leaves no brief (amendment 18); the agent sees why, and that a steered crawl runs those steps again.
1108
+ const error = stoppedBecause(view);
1109
+ console.log(`Run ${view.runId} ended (${view.status}) without a brief${error ? `: ${errorWords(error)}` : ''}.`);
1110
+ if (error) {
1111
+ console.log(` Details: ${safe(error.message)}`);
1112
+ console.log('A steered crawl runs these steps again, so it may stop the same way; if it does, tell us with `haystack feedback`.');
1113
+ }
1059
1114
  }
1060
1115
  if (captured !== null)
1061
1116
  for (const section of captured)
1062
1117
  console.log(`\n${formatCapture(section)}`);
1063
- console.log(steerLines(steerCommand(options)).join('\n'));
1118
+ console.log(steerLines(steerCommand(options), (brief?.planning?.ideas.length ?? 0) > 0).join('\n'));
1064
1119
  }
1065
1120
  /** verify's one JSON document: the report (`--json`) or the service's whole record (`--raw`). */
1066
1121
  function printVerifyJson(options, change, onboarding, view, handshake) {
package/dist/index.js CHANGED
@@ -120,7 +120,8 @@ an hour). The file is JSON with any of three notes, each plain text:
120
120
  Leave out what you do not know. Onboarding still proves everything it takes
121
121
  from the notes. The app is onboarded once: notes written before that feed its
122
122
  onboarding (an onboarding that stopped on a question starts again with them);
123
- after it, new notes start no onboarding.
123
+ after it, the next check that cannot start the app uses them to update the
124
+ setup in place.
124
125
 
125
126
  Every run also plans telemetry for the app: --app names it when the repository
126
127
  has several (init never picks one), --origin its production origins and
@@ -209,6 +210,16 @@ changed spot, how far the crawl got with it, and the steps that reached it),
209
210
  what the crawl found (bugs first, with the steps to see each), and the changed
210
211
  code it never ran.
211
212
 
213
+ When the copy of the app was itself broken (error popups or messages from its
214
+ own setup, its own requests failing in both builds), the crawl could not test
215
+ your change, and it says so first: "We couldn't test your change: the app copy
216
+ is broken.", why, what the start screen showed, the requests that failed and
217
+ the console errors, and what to do: find what those errors need in the
218
+ repository's own setup (docker compose files, env examples, docs), send that
219
+ with \`haystack feedback\` (a copy that starts but is broken does not update its
220
+ setup yet), and tell your user the change was not tested. What the crawl
221
+ found in the broken copy is listed after it, marked as such.
222
+
212
223
  The first verify on a base the service has not onboarded yet prepares the app
213
224
  (reads the repository, plans how to run it, builds its runtime, starts it,
214
225
  prepares its data and test accounts, proves a workflow); the command shows each
@@ -225,6 +236,8 @@ Exit codes:
225
236
  stopped before finishing
226
237
  3 onboarding is blocked: the output says what would unblock it
227
238
  4 the handshake: run it again with --intent (and any --idea)
239
+ 5 the crawl finished, but the copy of the app was broken, so your change
240
+ was not tested: the output says what to do
228
241
 
229
242
  The hosted fleet run of pushed commits is \`haystack verify hosted start\`.
230
243
 
package/dist/schema.js CHANGED
@@ -12,9 +12,9 @@
12
12
  export const SCHEMA_VERSIONS = {
13
13
  'cloud-verifier': '1.0.0',
14
14
  'case-batch': '1.0.1',
15
- verify: '2.1.1',
15
+ verify: '2.3.0',
16
16
  'verify-flow': '1.0.0',
17
- 'verify-raw': '1.1.0',
17
+ 'verify-raw': '1.3.0',
18
18
  'verify-answer': '1.0.1',
19
19
  'verify-onboarding': '1.0.0',
20
20
  init: '1.0.2',
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@haystackeditor/cli",
3
- "version": "0.29.0",
3
+ "version": "0.30.2",
4
4
  "description": "haystack verify: run your app with and without a change, and see what the change broke",
5
5
  "type": "module",
6
6
  "bin": {
@@ -21,7 +21,7 @@
21
21
  ]
22
22
  },
23
23
  "schema_version": {
24
- "const": "1.1.0"
24
+ "const": "1.3.0"
25
25
  },
26
26
  "change": {
27
27
  "description": "Schema 1.0.2: what this verify checked: the capture's base commit (where HEAD meets origin's default branch), its work commit (committed and uncommitted changes), and the files that differ between them.",
@@ -65,6 +65,7 @@
65
65
  "run",
66
66
  "brief",
67
67
  "capture",
68
+ "flows",
68
69
  "next"
69
70
  ],
70
71
  "properties": {
@@ -77,7 +78,8 @@
77
78
  "type": "object",
78
79
  "required": [
79
80
  "runId",
80
- "status"
81
+ "status",
82
+ "error"
81
83
  ],
82
84
  "properties": {
83
85
  "runId": {
@@ -95,6 +97,29 @@
95
97
  "incomplete",
96
98
  "cancelled"
97
99
  ]
100
+ },
101
+ "error": {
102
+ "oneOf": [
103
+ {
104
+ "type": "null"
105
+ },
106
+ {
107
+ "type": "object",
108
+ "additionalProperties": false,
109
+ "required": [
110
+ "code",
111
+ "message"
112
+ ],
113
+ "properties": {
114
+ "code": {
115
+ "type": "string"
116
+ },
117
+ "message": {
118
+ "type": "string"
119
+ }
120
+ }
121
+ }
122
+ ]
98
123
  }
99
124
  }
100
125
  }
@@ -127,6 +152,84 @@
127
152
  },
128
153
  "next": {
129
154
  "type": "string"
155
+ },
156
+ "flows": {
157
+ "description": "Amendment 21 (2.1.1): the brief's flows in its order, each with its number (from 1, as `haystack verify flow <n>` takes it), files, function counts and, per app with capture set up, its routes' share of captured sessions (null without capture). Empty when the brief has none.",
158
+ "type": "array",
159
+ "items": {
160
+ "type": "object",
161
+ "required": [
162
+ "number",
163
+ "id",
164
+ "name",
165
+ "files",
166
+ "functions",
167
+ "changedFunctions",
168
+ "capture"
169
+ ],
170
+ "properties": {
171
+ "number": {
172
+ "type": "integer",
173
+ "minimum": 1
174
+ },
175
+ "id": {
176
+ "type": "string"
177
+ },
178
+ "name": {
179
+ "type": "string"
180
+ },
181
+ "files": {
182
+ "type": "array",
183
+ "items": {
184
+ "type": "string"
185
+ }
186
+ },
187
+ "functions": {
188
+ "type": "integer",
189
+ "minimum": 0
190
+ },
191
+ "changedFunctions": {
192
+ "type": "integer",
193
+ "minimum": 0
194
+ },
195
+ "capture": {
196
+ "oneOf": [
197
+ {
198
+ "type": "null"
199
+ },
200
+ {
201
+ "type": "array",
202
+ "items": {
203
+ "type": "object",
204
+ "required": [
205
+ "app",
206
+ "applicationId",
207
+ "state"
208
+ ],
209
+ "properties": {
210
+ "app": {
211
+ "type": "string"
212
+ },
213
+ "state": {
214
+ "enum": [
215
+ "ready",
216
+ "no-data",
217
+ "unavailable"
218
+ ]
219
+ },
220
+ "routes": {
221
+ "$ref": "#/$defs/captureWindow"
222
+ },
223
+ "reason": {
224
+ "type": "string"
225
+ }
226
+ }
227
+ }
228
+ }
229
+ ]
230
+ }
231
+ }
232
+ }
130
233
  }
131
234
  }
132
235
  }
@@ -931,6 +1034,56 @@
931
1034
  "planning"
932
1035
  ]
933
1036
  },
1037
+ "health": {
1038
+ "description": "Schema 1.2.0, CRAWL-V1 amendment 22, in the final answer and the sealed manifest only: whether the copy of the app worked, so the crawl could test the change. not-judged carries only its reason. Otherwise status (app-broken: Clef answered app_broken and scored it working under 0.3, so the findings are the environment's; ok), working (0 to 1), tested (tested, app_broken, not_reached or unclear), reason (one sentence), evidence (the start screen's texts, messages and overlays and what each build reported with nobody clicking, at most 40 lines each, and at most 8 commonest difference lines with their counts) and images (health/old.png and health/new.png in the bundle, null in answers).",
1039
+ "type": "object",
1040
+ "required": [
1041
+ "status",
1042
+ "reason"
1043
+ ],
1044
+ "properties": {
1045
+ "status": {
1046
+ "enum": [
1047
+ "ok",
1048
+ "app-broken",
1049
+ "not-judged"
1050
+ ]
1051
+ },
1052
+ "reason": {
1053
+ "type": "string"
1054
+ },
1055
+ "working": {
1056
+ "type": "number",
1057
+ "minimum": 0,
1058
+ "maximum": 1
1059
+ },
1060
+ "tested": {
1061
+ "enum": [
1062
+ "tested",
1063
+ "app_broken",
1064
+ "not_reached",
1065
+ "unclear"
1066
+ ]
1067
+ },
1068
+ "evidence": {
1069
+ "type": "object",
1070
+ "required": [
1071
+ "texts",
1072
+ "messages",
1073
+ "overlays",
1074
+ "background",
1075
+ "commonDifferences"
1076
+ ]
1077
+ },
1078
+ "images": {
1079
+ "type": "object",
1080
+ "required": [
1081
+ "old",
1082
+ "new"
1083
+ ]
1084
+ }
1085
+ }
1086
+ },
934
1087
  "notFinishedInTime": {
935
1088
  "$ref": "#/$defs/count"
936
1089
  },
@@ -11,7 +11,7 @@
11
11
  ],
12
12
  "properties": {
13
13
  "schema_version": {
14
- "const": "2.1.1"
14
+ "const": "2.3.0"
15
15
  },
16
16
  "change": {
17
17
  "description": "Schema 1.0.2: what this verify checked: the capture's base commit (where HEAD meets origin's default branch), its work commit (committed and uncommitted changes), and the files that differ between them.",
@@ -70,6 +70,7 @@
70
70
  "neverRan",
71
71
  "notFinishedInTime",
72
72
  "outsideServiceGaps",
73
+ "health",
73
74
  "error",
74
75
  "machinesNotProvenShutDown"
75
76
  ],
@@ -342,6 +343,70 @@
342
343
  },
343
344
  "description": "Calls the copy's stand-ins could not serve: missing coverage, never bugs."
344
345
  },
346
+ "health": {
347
+ "description": "Schema 2.2.0 (CRAWL-V1 amendment 22): whether the copy of the app worked, so the crawl could test the change; null when the results say nothing of it (a crawl that did not explore, or a service from before it). app-broken: the copy was itself broken, so the change was not tested and the findings are the environment's; the command says so first and exits 5 (a run cancelled or failed after it answered keeps its own exit code). For app-broken, shows (what the start screen shows that reads as an error), failedRequests and consoleErrors (logged with nobody clicking, each with the builds it was in) and whatToDo; empty and null otherwise.",
348
+ "oneOf": [
349
+ {
350
+ "type": "null"
351
+ },
352
+ {
353
+ "type": "object",
354
+ "additionalProperties": false,
355
+ "required": [
356
+ "status",
357
+ "reason",
358
+ "shows",
359
+ "failedRequests",
360
+ "consoleErrors",
361
+ "whatToDo"
362
+ ],
363
+ "properties": {
364
+ "status": {
365
+ "enum": [
366
+ "ok",
367
+ "app-broken",
368
+ "not-judged"
369
+ ]
370
+ },
371
+ "reason": {
372
+ "type": "string",
373
+ "description": "One sentence in plain words."
374
+ },
375
+ "shows": {
376
+ "type": "array",
377
+ "items": {
378
+ "type": "string"
379
+ }
380
+ },
381
+ "failedRequests": {
382
+ "type": "array",
383
+ "items": {
384
+ "type": "string"
385
+ }
386
+ },
387
+ "consoleErrors": {
388
+ "type": "array",
389
+ "items": {
390
+ "type": "string"
391
+ }
392
+ },
393
+ "whatToDo": {
394
+ "oneOf": [
395
+ {
396
+ "type": "null"
397
+ },
398
+ {
399
+ "type": "array",
400
+ "items": {
401
+ "type": "string"
402
+ }
403
+ }
404
+ ]
405
+ }
406
+ }
407
+ }
408
+ ]
409
+ },
345
410
  "error": {
346
411
  "oneOf": [
347
412
  {
@@ -393,7 +458,8 @@
393
458
  "type": "object",
394
459
  "required": [
395
460
  "runId",
396
- "status"
461
+ "status",
462
+ "error"
397
463
  ],
398
464
  "properties": {
399
465
  "runId": {
@@ -411,6 +477,29 @@
411
477
  "incomplete",
412
478
  "cancelled"
413
479
  ]
480
+ },
481
+ "error": {
482
+ "oneOf": [
483
+ {
484
+ "type": "null"
485
+ },
486
+ {
487
+ "type": "object",
488
+ "additionalProperties": false,
489
+ "required": [
490
+ "code",
491
+ "message"
492
+ ],
493
+ "properties": {
494
+ "code": {
495
+ "type": "string"
496
+ },
497
+ "message": {
498
+ "type": "string"
499
+ }
500
+ }
501
+ }
502
+ ]
414
503
  }
415
504
  }
416
505
  }