@haystackeditor/cli 0.15.25 → 0.15.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +39 -3
- package/dist/assets/hooks/package-lock.json +598 -0
- package/dist/assets/skills/map-cloud-verifier-universe/SKILL.md +2051 -0
- package/dist/assets/skills/map-cloud-verifier-universe/agents/openai.yaml +4 -0
- package/dist/assets/skills/map-cloud-verifier-universe/references/output-contract.md +3411 -0
- package/dist/assets/telemetry/browser-runtime.js +861 -0
- package/dist/assets/telemetry/runtime.cjs +1509 -0
- package/dist/commands/cloud-verifier-behaviors.d.ts +27 -0
- package/dist/commands/cloud-verifier-behaviors.js +218 -0
- package/dist/commands/cloud-verifier-data-store-census.d.ts +47 -0
- package/dist/commands/cloud-verifier-data-store-census.js +539 -0
- package/dist/commands/cloud-verifier-data-store-drift.d.ts +42 -0
- package/dist/commands/cloud-verifier-data-store-drift.js +158 -0
- package/dist/commands/cloud-verifier-identity-census.d.ts +88 -0
- package/dist/commands/cloud-verifier-identity-census.js +4057 -0
- package/dist/commands/cloud-verifier-materialization.d.ts +16 -0
- package/dist/commands/cloud-verifier-materialization.js +704 -0
- package/dist/commands/cloud-verifier-pascal-selector-census.d.ts +29 -0
- package/dist/commands/cloud-verifier-pascal-selector-census.js +1382 -0
- package/dist/commands/cloud-verifier-python-manifest-selector-census.d.ts +27 -0
- package/dist/commands/cloud-verifier-python-manifest-selector-census.js +2015 -0
- package/dist/commands/cloud-verifier-specialized-operational-census.d.ts +51 -0
- package/dist/commands/cloud-verifier-specialized-operational-census.js +11432 -0
- package/dist/commands/cloud-verifier-universe.d.ts +31 -0
- package/dist/commands/cloud-verifier-universe.js +10178 -0
- package/dist/commands/design-verify.d.ts +23 -0
- package/dist/commands/design-verify.js +142 -0
- package/dist/commands/hooks.js +2 -2
- package/dist/commands/prepare-universe-review.d.ts +115 -0
- package/dist/commands/prepare-universe-review.js +1092 -0
- package/dist/commands/production-source-deny-policy.d.ts +15 -0
- package/dist/commands/production-source-deny-policy.js +100 -0
- package/dist/commands/scaffold-provisional-universe.d.ts +468 -0
- package/dist/commands/scaffold-provisional-universe.js +808 -0
- package/dist/commands/schema-cmd.js +32 -0
- package/dist/commands/skills.d.ts +2 -2
- package/dist/commands/skills.js +415 -47
- package/dist/commands/telemetry.d.ts +53 -0
- package/dist/commands/telemetry.js +832 -0
- package/dist/commands/verify-hosted-reproducibility.d.ts +6 -2
- package/dist/commands/verify-hosted-reproducibility.js +65 -20
- package/dist/index.js +142 -13
- package/dist/triage/runner.js +7 -6
- package/dist/utils/design-verifier-api.d.ts +129 -0
- package/dist/utils/design-verifier-api.js +175 -0
- package/dist/utils/design-verifier-result.d.ts +52 -0
- package/dist/utils/design-verifier-result.js +299 -0
- package/dist/utils/haystack-api.d.ts +2 -0
- package/dist/utils/haystack-api.js +10 -3
- package/dist/utils/hooks.js +2 -1
- package/package.json +7 -2
|
@@ -4,11 +4,13 @@ export interface HostedReproducibilityOptions {
|
|
|
4
4
|
head: string;
|
|
5
5
|
runs?: string;
|
|
6
6
|
seriesKey?: string;
|
|
7
|
+
freshPlans?: boolean;
|
|
7
8
|
account?: string;
|
|
8
9
|
interval?: string;
|
|
9
10
|
timeout?: string;
|
|
10
11
|
json?: boolean;
|
|
11
12
|
}
|
|
13
|
+
type HostedReproducibilityMode = 'exact_plan_replay' | 'fresh_plans';
|
|
12
14
|
interface HostedReproducibilityRun {
|
|
13
15
|
state: HostedRunState;
|
|
14
16
|
timedOut: boolean;
|
|
@@ -19,6 +21,7 @@ interface HostedReproducibilityIdentity {
|
|
|
19
21
|
baseCommit: string;
|
|
20
22
|
headCommit: string;
|
|
21
23
|
requestedRuns?: number;
|
|
24
|
+
mode?: HostedReproducibilityMode;
|
|
22
25
|
}
|
|
23
26
|
interface ParsedAssessment {
|
|
24
27
|
schemaVersion: 8 | 9;
|
|
@@ -60,6 +63,7 @@ export interface HostedReproducibilityReport {
|
|
|
60
63
|
repository: string;
|
|
61
64
|
base_commit: string;
|
|
62
65
|
head_commit: string;
|
|
66
|
+
reproducibility_mode: HostedReproducibilityMode;
|
|
63
67
|
requested_runs: number;
|
|
64
68
|
launched_runs: number;
|
|
65
69
|
unstarted_runs: number;
|
|
@@ -67,7 +71,7 @@ export interface HostedReproducibilityReport {
|
|
|
67
71
|
run_ids: string[];
|
|
68
72
|
replay_source_run_id: string | null;
|
|
69
73
|
plan_replay_verified: boolean;
|
|
70
|
-
planner_reproducibility_tested:
|
|
74
|
+
planner_reproducibility_tested: boolean;
|
|
71
75
|
planner_reproducible: boolean;
|
|
72
76
|
assessment_reproducible: boolean;
|
|
73
77
|
reproducible: boolean;
|
|
@@ -78,7 +82,7 @@ export interface HostedReproducibilityReport {
|
|
|
78
82
|
inconclusive_cell_ids: string[];
|
|
79
83
|
unrunnable_runs: RunFailure[];
|
|
80
84
|
}
|
|
81
|
-
export declare function hostedReproducibilityRunKey(seriesKey: string, repository: string, baseCommit: string, headCommit: string, ordinal: number): string;
|
|
85
|
+
export declare function hostedReproducibilityRunKey(seriesKey: string, repository: string, baseCommit: string, headCommit: string, ordinal: number, mode?: HostedReproducibilityMode): string;
|
|
82
86
|
export declare function analyzeHostedReproducibility(runs: HostedReproducibilityRun[], identity: HostedReproducibilityIdentity): HostedReproducibilityReport;
|
|
83
87
|
export declare function isHostedReproducibilityReplaySource(run: HostedReproducibilityRun, identity: HostedReproducibilityIdentity): boolean;
|
|
84
88
|
export declare function verifyHostedReproducibilityCommand(repositoryArg: string, options: HostedReproducibilityOptions): Promise<void>;
|
|
@@ -69,15 +69,24 @@ function requireSeriesKey(value) {
|
|
|
69
69
|
}
|
|
70
70
|
return key;
|
|
71
71
|
}
|
|
72
|
-
export function hostedReproducibilityRunKey(seriesKey, repository, baseCommit, headCommit, ordinal) {
|
|
72
|
+
export function hostedReproducibilityRunKey(seriesKey, repository, baseCommit, headCommit, ordinal, mode = 'exact_plan_replay') {
|
|
73
73
|
const digest = createHash('sha256')
|
|
74
|
-
.update(JSON.stringify(
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
74
|
+
.update(JSON.stringify(mode === 'exact_plan_replay'
|
|
75
|
+
? [
|
|
76
|
+
seriesKey,
|
|
77
|
+
repository,
|
|
78
|
+
baseCommit,
|
|
79
|
+
headCommit,
|
|
80
|
+
ordinal,
|
|
81
|
+
]
|
|
82
|
+
: [
|
|
83
|
+
mode,
|
|
84
|
+
seriesKey,
|
|
85
|
+
repository,
|
|
86
|
+
baseCommit,
|
|
87
|
+
headCommit,
|
|
88
|
+
ordinal,
|
|
89
|
+
]))
|
|
81
90
|
.digest('hex');
|
|
82
91
|
return `repro:${digest}`;
|
|
83
92
|
}
|
|
@@ -371,10 +380,17 @@ export function analyzeHostedReproducibility(runs, identity) {
|
|
|
371
380
|
});
|
|
372
381
|
}
|
|
373
382
|
}
|
|
383
|
+
const requestedRuns = identity.requestedRuns ?? runs.length;
|
|
384
|
+
const allRequestedRunsLaunched = runs.length === requestedRuns;
|
|
374
385
|
const planVariants = groupedPlanVariants(parsed);
|
|
375
386
|
const assessmentVariants = groupedOutcomeVariants(parsed);
|
|
376
|
-
const
|
|
377
|
-
const
|
|
387
|
+
const mode = identity.mode ?? 'exact_plan_replay';
|
|
388
|
+
const replaySourceRunId = mode === 'exact_plan_replay'
|
|
389
|
+
? parsed[0]?.runId ?? null
|
|
390
|
+
: null;
|
|
391
|
+
const planReplayVerified = mode === 'exact_plan_replay'
|
|
392
|
+
&& allRequestedRunsLaunched
|
|
393
|
+
&& parsed.length === runs.length
|
|
378
394
|
&& parsed.length > 1
|
|
379
395
|
&& parsed[0]?.assessment.schemaVersion === 8
|
|
380
396
|
&& parsed[0]?.assessment.replaySourceRunId === null
|
|
@@ -384,7 +400,17 @@ export function analyzeHostedReproducibility(runs, identity) {
|
|
|
384
400
|
=== parsed[0].assessment.schemaVersion
|
|
385
401
|
&& item.assessment.riskPlanSha256
|
|
386
402
|
=== parsed[0].assessment.riskPlanSha256);
|
|
387
|
-
const
|
|
403
|
+
const plannerReproducibilityTested = mode === 'fresh_plans' && parsed.length > 1;
|
|
404
|
+
const plannerReproducible = plannerReproducibilityTested
|
|
405
|
+
&& allRequestedRunsLaunched
|
|
406
|
+
&& parsed.length === runs.length
|
|
407
|
+
&& parsed.every(item => item.assessment.schemaVersion === 8
|
|
408
|
+
&& item.assessment.replaySourceRunId === null)
|
|
409
|
+
&& planVariants.length === 1;
|
|
410
|
+
const assessmentReproducible = (mode === 'exact_plan_replay'
|
|
411
|
+
? planReplayVerified
|
|
412
|
+
: plannerReproducible)
|
|
413
|
+
&& assessmentVariants.length === 1;
|
|
388
414
|
const planDriftCellIds = driftIds(parsed, assessment => new Map(assessment.planIdentities.map(plan => [
|
|
389
415
|
plan.cellId,
|
|
390
416
|
{ capabilityId: plan.capabilityId },
|
|
@@ -401,12 +427,12 @@ export function analyzeHostedReproducibility(runs, identity) {
|
|
|
401
427
|
const inconclusiveCellIds = [...new Set(parsed.flatMap(item => item.assessment.outcomes
|
|
402
428
|
.filter(outcome => outcome.status === 'inconclusive')
|
|
403
429
|
.map(outcome => outcome.cellId)))].sort();
|
|
404
|
-
const requestedRuns = identity.requestedRuns ?? runs.length;
|
|
405
430
|
return {
|
|
406
431
|
series_key: identity.seriesKey,
|
|
407
432
|
repository: identity.repository,
|
|
408
433
|
base_commit: identity.baseCommit,
|
|
409
434
|
head_commit: identity.headCommit,
|
|
435
|
+
reproducibility_mode: mode,
|
|
410
436
|
requested_runs: requestedRuns,
|
|
411
437
|
launched_runs: runs.length,
|
|
412
438
|
unstarted_runs: Math.max(0, requestedRuns - runs.length),
|
|
@@ -414,10 +440,12 @@ export function analyzeHostedReproducibility(runs, identity) {
|
|
|
414
440
|
run_ids: runs.map(run => run.state.runId),
|
|
415
441
|
replay_source_run_id: replaySourceRunId,
|
|
416
442
|
plan_replay_verified: planReplayVerified,
|
|
417
|
-
planner_reproducibility_tested:
|
|
418
|
-
planner_reproducible:
|
|
443
|
+
planner_reproducibility_tested: plannerReproducibilityTested,
|
|
444
|
+
planner_reproducible: plannerReproducible,
|
|
419
445
|
assessment_reproducible: assessmentReproducible,
|
|
420
|
-
reproducible:
|
|
446
|
+
reproducible: (mode === 'exact_plan_replay'
|
|
447
|
+
? planReplayVerified
|
|
448
|
+
: plannerReproducible)
|
|
421
449
|
&& assessmentReproducible
|
|
422
450
|
&& inconclusiveCellIds.length === 0,
|
|
423
451
|
plan_variants: planVariants,
|
|
@@ -444,9 +472,20 @@ function renderHostedReproducibility(report) {
|
|
|
444
472
|
const icon = report.reproducible ? chalk.green('✓') : chalk.yellow('!');
|
|
445
473
|
console.log(`${icon} Hosted reproducibility: ${report.reproducible ? 'reproducible' : 'not established'}`);
|
|
446
474
|
console.log(` ${chalk.dim('Runs:')} ${report.analyzable_runs}/${report.requested_runs} analyzable; ${report.launched_runs} launched`);
|
|
475
|
+
console.log(` ${chalk.dim('Mode:')} ${report.reproducibility_mode === 'fresh_plans'
|
|
476
|
+
? 'independent fresh plans'
|
|
477
|
+
: 'exact plan replay'}`);
|
|
447
478
|
console.log(` ${chalk.dim('Plan variants:')} ${report.plan_variants.length}`);
|
|
448
|
-
console.log(` ${chalk.dim('Exact plan replay:')} ${report.
|
|
449
|
-
|
|
479
|
+
console.log(` ${chalk.dim('Exact plan replay:')} ${report.reproducibility_mode === 'fresh_plans'
|
|
480
|
+
? 'not requested'
|
|
481
|
+
: report.plan_replay_verified
|
|
482
|
+
? 'verified'
|
|
483
|
+
: 'not verified'}`);
|
|
484
|
+
console.log(` ${chalk.dim('Independent planner reproducibility:')} ${report.planner_reproducibility_tested
|
|
485
|
+
? report.planner_reproducible
|
|
486
|
+
? 'verified'
|
|
487
|
+
: 'not reproducible'
|
|
488
|
+
: 'not tested'}`);
|
|
450
489
|
console.log(` ${chalk.dim('Assessment variants:')} ${report.assessment_variants.length}`);
|
|
451
490
|
if (report.plan_drift_cell_ids.length > 0) {
|
|
452
491
|
console.log(` ${chalk.dim('Plan drift:')} ${report.plan_drift_cell_ids.length} stable cell id(s)`);
|
|
@@ -465,6 +504,9 @@ export async function verifyHostedReproducibilityCommand(repositoryArg, options)
|
|
|
465
504
|
const repository = parseHostedRepository(repositoryArg);
|
|
466
505
|
const requestedRuns = boundedInteger(options.runs, DEFAULT_RUNS, '--runs', 2, MAX_RUNS);
|
|
467
506
|
const seriesKey = requireSeriesKey(options.seriesKey);
|
|
507
|
+
const mode = options.freshPlans
|
|
508
|
+
? 'fresh_plans'
|
|
509
|
+
: 'exact_plan_replay';
|
|
468
510
|
const auth = await resolveAuthContext({
|
|
469
511
|
preferredLogin: options.account,
|
|
470
512
|
owner: repository.owner,
|
|
@@ -479,8 +521,8 @@ export async function verifyHostedReproducibilityCommand(repositoryArg, options)
|
|
|
479
521
|
const started = await startHostedVerification(repositoryArg, {
|
|
480
522
|
base: options.base,
|
|
481
523
|
head: options.head,
|
|
482
|
-
idempotencyKey: hostedReproducibilityRunKey(seriesKey, repository.fullName, options.base, options.head, index + 1),
|
|
483
|
-
replayPlanFrom: index === 0
|
|
524
|
+
idempotencyKey: hostedReproducibilityRunKey(seriesKey, repository.fullName, options.base, options.head, index + 1, mode),
|
|
525
|
+
replayPlanFrom: options.freshPlans || index === 0
|
|
484
526
|
? undefined
|
|
485
527
|
: runs[0]?.state.runId,
|
|
486
528
|
wait: false,
|
|
@@ -497,7 +539,9 @@ export async function verifyHostedReproducibilityCommand(repositoryArg, options)
|
|
|
497
539
|
timedOut = waited.timedOut;
|
|
498
540
|
}
|
|
499
541
|
runs.push({ state, timedOut });
|
|
500
|
-
if (
|
|
542
|
+
if (!options.freshPlans
|
|
543
|
+
&&
|
|
544
|
+
index === 0
|
|
501
545
|
&& !isHostedReproducibilityReplaySource(runs[0], {
|
|
502
546
|
seriesKey,
|
|
503
547
|
repository: repository.fullName,
|
|
@@ -516,6 +560,7 @@ export async function verifyHostedReproducibilityCommand(repositoryArg, options)
|
|
|
516
560
|
baseCommit: options.base,
|
|
517
561
|
headCommit: options.head,
|
|
518
562
|
requestedRuns,
|
|
563
|
+
mode,
|
|
519
564
|
});
|
|
520
565
|
if (options.json) {
|
|
521
566
|
process.stdout.write(`${JSON.stringify(withSchema('cloud-verifier', {
|
package/dist/index.js
CHANGED
|
@@ -30,11 +30,18 @@ import { initCommand } from './commands/init.js';
|
|
|
30
30
|
import { authListCommand, authUseCommand, loginCommand, logoutCommand, whoamiCommand, } from './commands/login.js';
|
|
31
31
|
import { handleAgenticTool, handleAutoMerge, handleAutoFix, handleWaitForReviewers, isAutoMergeEnabled, isAutoFixEnabled } from './commands/config.js';
|
|
32
32
|
import { installSkills, listSkills } from './commands/skills.js';
|
|
33
|
+
import { cloudVerifierUniverseValidateCommand } from './commands/cloud-verifier-universe.js';
|
|
34
|
+
import { cloudVerifierIdentityCensusCommand } from './commands/cloud-verifier-identity-census.js';
|
|
35
|
+
import { cloudVerifierDataStoreDriftCommand } from './commands/cloud-verifier-data-store-drift.js';
|
|
36
|
+
import { cloudVerifierBehaviorsValidateCommand } from './commands/cloud-verifier-behaviors.js';
|
|
37
|
+
import { prepareUniverseReviewCommand } from './commands/prepare-universe-review.js';
|
|
38
|
+
import { scaffoldProvisionalUniverseCommand } from './commands/scaffold-provisional-universe.js';
|
|
33
39
|
import { hooksInstall, hooksStatus, hooksUpdate } from './commands/hooks.js';
|
|
34
40
|
import { submitCommand } from './commands/submit.js';
|
|
35
41
|
import { installSessionHooks, sessionHooksStatus } from './commands/install-session-hooks.js';
|
|
36
42
|
import { listPolicies, addPolicy, removePolicy, initPolicies, addInstruction } from './commands/policy.js';
|
|
37
43
|
import { triageCommand } from './commands/triage.js';
|
|
44
|
+
import { designVerifyCommand } from './commands/design-verify.js';
|
|
38
45
|
import { dismissCommand, markReviewedCommand, undismissCommand } from './commands/dismiss.js';
|
|
39
46
|
import { requestReviewCommand } from './commands/request-review.js';
|
|
40
47
|
import { reviewCommand } from './commands/review.js';
|
|
@@ -43,6 +50,7 @@ import { prReadCommand } from './commands/pr.js';
|
|
|
43
50
|
import { inboxListCommand } from './commands/inbox.js';
|
|
44
51
|
import { askHaystackCommand } from './commands/ask.js';
|
|
45
52
|
import { tracesGetCommand, tracesListCommand } from './commands/traces.js';
|
|
53
|
+
import { telemetryInstrumentCommand } from './commands/telemetry.js';
|
|
46
54
|
import { setupCommand } from './commands/setup.js';
|
|
47
55
|
import { schemaCommand, listSchemas } from './commands/schema-cmd.js';
|
|
48
56
|
import { registerWebhook, listWebhooks, rotateWebhookSecret, setWebhookEnabled, listDeliveries, replayDelivery, } from './commands/webhooks.js';
|
|
@@ -262,12 +270,13 @@ Example:
|
|
|
262
270
|
});
|
|
263
271
|
hostedVerify
|
|
264
272
|
.command('reproducibility')
|
|
265
|
-
.description('Replay one exact hosted risk plan
|
|
273
|
+
.description('Replay one exact hosted risk plan or run fresh plans to report drift')
|
|
266
274
|
.argument('<repository>', 'Exact GitHub owner/repository name')
|
|
267
275
|
.option('--base <sha>', 'Required: exact 40-character lowercase base commit SHA')
|
|
268
276
|
.option('--head <sha>', 'Required: exact 40-character lowercase head commit SHA')
|
|
269
277
|
.option('--runs <n>', 'Number of sequential production runs (default 3; min 2, max 10)')
|
|
270
278
|
.option('--series-key <key>', 'Stable series key for resumable idempotent retries')
|
|
279
|
+
.option('--fresh-plans', 'Run the planner independently for every run instead of replaying the first plan')
|
|
271
280
|
.option('--account <login>', 'Use a specific saved Haystack account')
|
|
272
281
|
.option('--interval <seconds>', 'Polling interval for each run (default 5)')
|
|
273
282
|
.option('--timeout <minutes>', 'Maximum wait per run (default 35)')
|
|
@@ -278,8 +287,14 @@ validated plan against the same repository and commit pair with a distinct,
|
|
|
278
287
|
series-derived idempotency key. A reproducible verdict requires schema-9 replay
|
|
279
288
|
provenance, stable assessment outcomes, and no inconclusive cells.
|
|
280
289
|
|
|
290
|
+
With --fresh-plans, every run plans independently. A reproducible verdict then
|
|
291
|
+
requires one stable-id plan variant, one assessment variant, no replay
|
|
292
|
+
provenance, and no inconclusive cells. Use this mode to detect planner drift;
|
|
293
|
+
use the default mode to isolate executor or assessment drift.
|
|
294
|
+
|
|
281
295
|
Example:
|
|
282
296
|
haystack verify hosted reproducibility owner/repo --base <sha> --head <sha> --runs 3
|
|
297
|
+
haystack verify hosted reproducibility owner/repo --base <sha> --head <sha> --runs 3 --fresh-plans
|
|
283
298
|
`)
|
|
284
299
|
.action(async (repository, _options, cmd) => {
|
|
285
300
|
const options = { ...cmd.optsWithGlobals(), ...cmd.opts() };
|
|
@@ -797,6 +812,32 @@ traces
|
|
|
797
812
|
.option('--limit <count>', 'Chunks per page', '20')
|
|
798
813
|
.option('--json', 'Output as JSON')
|
|
799
814
|
.action((ref, checkpoint, options) => runPublicCommand(() => tracesGetCommand(ref, checkpoint, options), options.json));
|
|
815
|
+
const telemetry = program.command('telemetry').description('Add privacy-safe production telemetry without an application SDK');
|
|
816
|
+
telemetry
|
|
817
|
+
.command('instrument <dist>')
|
|
818
|
+
.description('Instrument compiled Node.js output without changing application source')
|
|
819
|
+
.requiredOption('--entry <file>', 'Entry file relative to the compiled output directory')
|
|
820
|
+
.option('--source-root <dir>', 'Source directory whose files correspond exactly to compiled output (default: sibling src/)')
|
|
821
|
+
.option('--include-symbols <file>', 'Experimental JSON allow-list of exact sourcePath + qualifiedName symbols')
|
|
822
|
+
.option('--json', 'Machine-readable instrumentation manifest')
|
|
823
|
+
.addHelpText('after', `
|
|
824
|
+
Run this once after the normal application build. It rewrites only compiled
|
|
825
|
+
JavaScript and copies a dependency-free runtime beside it; application source
|
|
826
|
+
does not import Haystack.
|
|
827
|
+
|
|
828
|
+
At runtime, telemetry stays off unless all three variables are set:
|
|
829
|
+
HAYSTACK_TELEMETRY=1
|
|
830
|
+
HAYSTACK_TELEMETRY_ENDPOINT=https://.../v3/telemetry
|
|
831
|
+
HAYSTACK_TELEMETRY_TOKEN=...
|
|
832
|
+
|
|
833
|
+
Example:
|
|
834
|
+
npm run build && haystack telemetry instrument dist --entry index.js
|
|
835
|
+
`)
|
|
836
|
+
.action((dist, options) => runPublicCommand(() => telemetryInstrumentCommand(dist, {
|
|
837
|
+
...options,
|
|
838
|
+
includeSymbolsFile: options.includeSymbols,
|
|
839
|
+
includeSymbols: undefined,
|
|
840
|
+
}), options.json));
|
|
800
841
|
program
|
|
801
842
|
.command('dismiss <pr>')
|
|
802
843
|
.description('Dismiss analysis findings for a PR')
|
|
@@ -819,6 +860,31 @@ Examples:
|
|
|
819
860
|
haystack dismiss acme/widgets#99 # Dismiss for specific repo
|
|
820
861
|
`)
|
|
821
862
|
.action(dismissCommand);
|
|
863
|
+
program
|
|
864
|
+
.command('design-verify <pr>')
|
|
865
|
+
// "verification" is a banned word in top-level help (public-contract test
|
|
866
|
+
// keeps retired auto-fix/verification wording out of the CLI surface).
|
|
867
|
+
.description('Run a blind design-by-execution review of a PR')
|
|
868
|
+
.option('--json', 'Machine-readable output')
|
|
869
|
+
.option('--no-wait', 'Return after the run is admitted instead of waiting for its verdict')
|
|
870
|
+
.option('--run <run-id>', 'Resume and wait for an existing exact run id')
|
|
871
|
+
.option('--poll-interval <seconds>', 'Status polling interval', '3')
|
|
872
|
+
.option('--timeout <seconds>', 'Maximum time to wait; the server run continues after timeout', '7200')
|
|
873
|
+
.addHelpText('after', `
|
|
874
|
+
Executes the change instead of reading it: infers intent from the diff
|
|
875
|
+
alone, designs pre-registered behavioral test cases, runs them against
|
|
876
|
+
both sides of the PR in isolated sandboxes, and reports a mechanical
|
|
877
|
+
verdict of unexpected behavior changes.
|
|
878
|
+
|
|
879
|
+
The command waits for a behavioral verdict by default. Use --no-wait to
|
|
880
|
+
return the run id immediately, or --run <run-id> to resume waiting.
|
|
881
|
+
|
|
882
|
+
PR identifier formats:
|
|
883
|
+
123 PR number (uses current repo)
|
|
884
|
+
owner/repo#123 Fully qualified
|
|
885
|
+
https://github.com/owner/repo/pull/123 GitHub URL
|
|
886
|
+
`)
|
|
887
|
+
.action(designVerifyCommand);
|
|
822
888
|
program
|
|
823
889
|
.command('mark-reviewed <pr>')
|
|
824
890
|
.description('Mark human review as not needed for a PR')
|
|
@@ -1028,32 +1094,95 @@ Examples:
|
|
|
1028
1094
|
// Skills subcommands
|
|
1029
1095
|
const skills = program
|
|
1030
1096
|
.command('skills')
|
|
1031
|
-
.description('Manage AI skills for
|
|
1097
|
+
.description('Manage AI skills for coding agents');
|
|
1032
1098
|
skills
|
|
1033
1099
|
.command('install')
|
|
1034
|
-
.description('Install Haystack skills
|
|
1100
|
+
.description('Install portable Haystack skills and optional coding-CLI shims')
|
|
1035
1101
|
.option('--cli <name>', 'Target CLI: claude, codex, cursor, or manual')
|
|
1036
1102
|
.addHelpText('after', `
|
|
1037
|
-
This
|
|
1038
|
-
/setup-haystack - AI-assisted project setup
|
|
1103
|
+
This installs portable skills in the Git repository's .agents/skills directory:
|
|
1039
1104
|
/submit - Submit a PR via Haystack
|
|
1105
|
+
/map-your-system - Map how Haystack QA can run this system
|
|
1106
|
+
/map-cloud-verifier-universe - Map a production-derived hermetic universe
|
|
1040
1107
|
Supported CLIs:
|
|
1041
|
-
claude Claude Code
|
|
1042
|
-
codex
|
|
1043
|
-
cursor
|
|
1044
|
-
manual
|
|
1108
|
+
claude Also install Claude Code command shims
|
|
1109
|
+
codex Use portable .agents/skills discovery only
|
|
1110
|
+
cursor Use portable .agents/skills discovery only
|
|
1111
|
+
manual Install portable skills and show their location
|
|
1045
1112
|
|
|
1046
1113
|
Examples:
|
|
1047
|
-
haystack skills install #
|
|
1114
|
+
haystack skills install # Install portable skills only
|
|
1048
1115
|
haystack skills install --cli codex # Install for Codex only
|
|
1049
|
-
haystack skills install --cli
|
|
1116
|
+
haystack skills install --cli claude # Also install Claude command shims
|
|
1050
1117
|
`)
|
|
1051
|
-
.action(
|
|
1118
|
+
.action(async (opts) => {
|
|
1119
|
+
try {
|
|
1120
|
+
await installSkills(opts);
|
|
1121
|
+
}
|
|
1122
|
+
catch (err) {
|
|
1123
|
+
console.error(chalk.red('skills install failed:'), err instanceof Error ? err.message : err);
|
|
1124
|
+
process.exit(1);
|
|
1125
|
+
}
|
|
1126
|
+
});
|
|
1052
1127
|
skills
|
|
1053
1128
|
.command('list')
|
|
1054
1129
|
.description('List available Haystack skills')
|
|
1055
1130
|
.action(listSkills);
|
|
1056
|
-
|
|
1131
|
+
skills
|
|
1132
|
+
.command('census-universe')
|
|
1133
|
+
.description('Deterministically census Cloud Verifier source identities before mapping')
|
|
1134
|
+
.option('--json', 'Machine-readable result')
|
|
1135
|
+
.option('--root <path>', 'Repository root (default: current Git worktree)')
|
|
1136
|
+
.action((opts) => runPublicCommand(async () => cloudVerifierIdentityCensusCommand(opts), opts.json));
|
|
1137
|
+
skills
|
|
1138
|
+
.command('data-store-drift')
|
|
1139
|
+
.description('Check whether the detected data stores still match the checked-in map (per-PR drift gate)')
|
|
1140
|
+
.option('--json', 'Machine-readable result')
|
|
1141
|
+
.option('--root <path>', 'Repository root (default: current Git worktree)')
|
|
1142
|
+
.option('--baseline <path>', 'Baseline map path (default: .haystack/cloud-verifier/data-stores.json)')
|
|
1143
|
+
.option('--base-ref <ref>', 'Only run when the diff against this ref touches a store-relevant file')
|
|
1144
|
+
.option('--update', 'Write the current detected stores as the new baseline')
|
|
1145
|
+
.action((opts) => runPublicCommand(async () => cloudVerifierDataStoreDriftCommand(opts), opts.json));
|
|
1146
|
+
skills
|
|
1147
|
+
.command('validate-universe')
|
|
1148
|
+
.description('Validate Cloud Verifier universe artifacts and stable-ID references')
|
|
1149
|
+
.option('--json', 'Machine-readable result')
|
|
1150
|
+
.action(async (opts) => {
|
|
1151
|
+
try {
|
|
1152
|
+
await cloudVerifierUniverseValidateCommand(opts);
|
|
1153
|
+
}
|
|
1154
|
+
catch (err) {
|
|
1155
|
+
console.error(chalk.red('validate-universe failed:'), err instanceof Error ? err.message : err);
|
|
1156
|
+
process.exit(1);
|
|
1157
|
+
}
|
|
1158
|
+
});
|
|
1159
|
+
skills
|
|
1160
|
+
.command('validate-behaviors')
|
|
1161
|
+
.description('Validate customer actions and their exact runtime identities')
|
|
1162
|
+
.option('--json', 'Machine-readable result')
|
|
1163
|
+
.action(async (opts) => {
|
|
1164
|
+
try {
|
|
1165
|
+
await cloudVerifierBehaviorsValidateCommand(opts);
|
|
1166
|
+
}
|
|
1167
|
+
catch (err) {
|
|
1168
|
+
console.error(chalk.red('validate-behaviors failed:'), err instanceof Error ? err.message : err);
|
|
1169
|
+
process.exit(1);
|
|
1170
|
+
}
|
|
1171
|
+
});
|
|
1172
|
+
skills
|
|
1173
|
+
.command('scaffold-provisional-universe')
|
|
1174
|
+
.description('Create a compact missing-authority receipt from strict JSON facts')
|
|
1175
|
+
.requiredOption('--input <path>', 'Strict provisional-universe JSON input file')
|
|
1176
|
+
.action((opts) => runPublicCommand(async () => scaffoldProvisionalUniverseCommand(opts), true));
|
|
1177
|
+
skills
|
|
1178
|
+
.command('prepare-universe-review')
|
|
1179
|
+
.description('Create a tracked-source snapshot with conventional test paths removed')
|
|
1180
|
+
.option('--json', 'Machine-readable isolation result and snapshot manifest')
|
|
1181
|
+
.option('--exclude-inactive-submodule <path...>', 'Omit exact tracked gitlink paths explicitly known to be inactive')
|
|
1182
|
+
.option('--omit-unscannable-files', 'Omit files whose test content cannot be ruled out, instead of blocking the review. '
|
|
1183
|
+
+ 'The reviewer still never sees a test; the omissions are recorded as uncovered scope')
|
|
1184
|
+
.option('--full-receipt', 'Include complete non-blocking path receipts in JSON output')
|
|
1185
|
+
.action((opts) => runPublicCommand(() => prepareUniverseReviewCommand(opts), opts.json));
|
|
1057
1186
|
const hooks = program
|
|
1058
1187
|
.command('hooks')
|
|
1059
1188
|
.description('Manage git hooks for AI agent quality checks');
|
package/dist/triage/runner.js
CHANGED
|
@@ -101,7 +101,13 @@ export function buildCommand(cli, prompt, maxTurns) {
|
|
|
101
101
|
case 'codex':
|
|
102
102
|
return {
|
|
103
103
|
command: 'codex',
|
|
104
|
-
args: [
|
|
104
|
+
args: [
|
|
105
|
+
'--sandbox', 'workspace-write',
|
|
106
|
+
'--ask-for-approval', 'never',
|
|
107
|
+
'exec',
|
|
108
|
+
'--ephemeral',
|
|
109
|
+
prompt,
|
|
110
|
+
],
|
|
105
111
|
};
|
|
106
112
|
case 'gemini':
|
|
107
113
|
return {
|
|
@@ -119,17 +125,12 @@ function spawnChecker(cli, config, cwd, timeoutMs) {
|
|
|
119
125
|
const { command, args } = buildCommand(cli, config.prompt, config.maxTurns);
|
|
120
126
|
let stdout = '';
|
|
121
127
|
let stderr = '';
|
|
122
|
-
const codexHome = join(cwd, '.haystack', 'triage', 'codex-home');
|
|
123
|
-
if (cli === 'codex') {
|
|
124
|
-
mkdirSync(codexHome, { recursive: true });
|
|
125
|
-
}
|
|
126
128
|
const proc = spawn(command, args, {
|
|
127
129
|
cwd,
|
|
128
130
|
stdio: ['ignore', 'pipe', 'pipe'],
|
|
129
131
|
env: {
|
|
130
132
|
...process.env,
|
|
131
133
|
CLAUDECODE: undefined,
|
|
132
|
-
CODEX_HOME: cli === 'codex' ? codexHome : process.env.CODEX_HOME,
|
|
133
134
|
},
|
|
134
135
|
});
|
|
135
136
|
proc.stdout?.on('data', (data) => {
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Client for the design-verifier trigger route on the auth worker.
|
|
3
|
+
*
|
|
4
|
+
* The worker owns everything sensitive: it authenticates the CLI token,
|
|
5
|
+
* enforces hsk_live repo scope, resolves the PR's base/head SHAs via the
|
|
6
|
+
* GitHub App install token, and forwards to the AWS design-verifier API with
|
|
7
|
+
* the server-side shared secret. This client just POSTs and interprets the
|
|
8
|
+
* response.
|
|
9
|
+
*/
|
|
10
|
+
export type DesignVerifierTriggerResult = {
|
|
11
|
+
status: 'started' | 'queued' | 'duplicate';
|
|
12
|
+
runId: string;
|
|
13
|
+
} | {
|
|
14
|
+
status: 'denied';
|
|
15
|
+
message: string;
|
|
16
|
+
} | {
|
|
17
|
+
status: 'rate_limited';
|
|
18
|
+
message: string;
|
|
19
|
+
} | {
|
|
20
|
+
status: 'unavailable';
|
|
21
|
+
message: string;
|
|
22
|
+
} | {
|
|
23
|
+
status: 'error';
|
|
24
|
+
message: string;
|
|
25
|
+
};
|
|
26
|
+
export interface DesignVerifierCommandSpec {
|
|
27
|
+
cwd: string;
|
|
28
|
+
argv: string[];
|
|
29
|
+
env: Record<string, string>;
|
|
30
|
+
timeoutMs?: number;
|
|
31
|
+
}
|
|
32
|
+
export interface DesignVerifierExecutionDiagnostic {
|
|
33
|
+
diagnosticId: string;
|
|
34
|
+
kind: string;
|
|
35
|
+
message: string;
|
|
36
|
+
type?: string;
|
|
37
|
+
code?: string;
|
|
38
|
+
}
|
|
39
|
+
export interface DesignVerifierExecutionSummary {
|
|
40
|
+
side: 'base1' | 'base2' | 'head';
|
|
41
|
+
status: 'completed' | 'failed' | 'missing';
|
|
42
|
+
commandExitCode: number | null;
|
|
43
|
+
durationMs: number | null;
|
|
44
|
+
scenarioError: string | null;
|
|
45
|
+
diagnostics: DesignVerifierExecutionDiagnostic[];
|
|
46
|
+
diagnosticsOmitted: number;
|
|
47
|
+
outputExcerpt: Array<{
|
|
48
|
+
stream: 'stdout' | 'stderr' | 'system' | 'output';
|
|
49
|
+
text: string;
|
|
50
|
+
}>;
|
|
51
|
+
outputLinesOmitted: number;
|
|
52
|
+
traceArtifact: string;
|
|
53
|
+
outputArtifact: string;
|
|
54
|
+
}
|
|
55
|
+
export interface DesignVerifierStageTiming {
|
|
56
|
+
stage: 'prepare' | 'bake' | 'blast-radius' | 'prefetch' | 'case-design' | 'verdict';
|
|
57
|
+
status: string;
|
|
58
|
+
durationMs: number | null;
|
|
59
|
+
completedAt: string | null;
|
|
60
|
+
source: 'phase-artifact' | 'phase-artifact-clock' | 'cache-hit' | 'missing';
|
|
61
|
+
breakdownMs: Record<string, number>;
|
|
62
|
+
pollCount?: number;
|
|
63
|
+
}
|
|
64
|
+
export interface DesignVerifierTimingSummary {
|
|
65
|
+
admissionMs: number | null;
|
|
66
|
+
pipelineDurationMs: number | null;
|
|
67
|
+
recordedStageMs: number;
|
|
68
|
+
unattributedMs: number | null;
|
|
69
|
+
stages: DesignVerifierStageTiming[];
|
|
70
|
+
}
|
|
71
|
+
export interface DesignVerifierStatusResult {
|
|
72
|
+
version: 'dv-cli-status-v1';
|
|
73
|
+
runId: string;
|
|
74
|
+
owner: string;
|
|
75
|
+
repo: string;
|
|
76
|
+
pr: number;
|
|
77
|
+
baseSha: string | null;
|
|
78
|
+
headSha: string | null;
|
|
79
|
+
status: string;
|
|
80
|
+
stage: string | null;
|
|
81
|
+
createdAt: string | null;
|
|
82
|
+
startedAt: string | null;
|
|
83
|
+
finishedAt: string | null;
|
|
84
|
+
durationMs: number | null;
|
|
85
|
+
error: string | null;
|
|
86
|
+
result?: {
|
|
87
|
+
verdict: Record<string, any>;
|
|
88
|
+
cases: Array<Record<string, any>>;
|
|
89
|
+
scenarios: Array<Record<string, any> & {
|
|
90
|
+
executions?: DesignVerifierExecutionSummary[];
|
|
91
|
+
}>;
|
|
92
|
+
changedFiles: string[];
|
|
93
|
+
timings?: DesignVerifierTimingSummary;
|
|
94
|
+
probes: Array<{
|
|
95
|
+
caseId: string;
|
|
96
|
+
title: string;
|
|
97
|
+
files: string[];
|
|
98
|
+
validate: DesignVerifierCommandSpec | null;
|
|
99
|
+
run: DesignVerifierCommandSpec | null;
|
|
100
|
+
}>;
|
|
101
|
+
replay: {
|
|
102
|
+
artifactPrefix: string;
|
|
103
|
+
keys: Record<string, string>;
|
|
104
|
+
};
|
|
105
|
+
};
|
|
106
|
+
}
|
|
107
|
+
export type DesignVerifierStatusFetchResult = {
|
|
108
|
+
status: 'ok';
|
|
109
|
+
run: DesignVerifierStatusResult;
|
|
110
|
+
} | {
|
|
111
|
+
status: 'not_found' | 'denied' | 'unavailable' | 'error';
|
|
112
|
+
message: string;
|
|
113
|
+
};
|
|
114
|
+
export declare function triggerDesignVerifier(owner: string, repo: string, prNumber: number, token: string): Promise<DesignVerifierTriggerResult>;
|
|
115
|
+
export declare function getDesignVerifierStatus(owner: string, repo: string, prNumber: number, runId: string, token: string): Promise<DesignVerifierStatusFetchResult>;
|
|
116
|
+
export declare function waitForDesignVerifier(owner: string, repo: string, prNumber: number, runId: string, token: string, options?: {
|
|
117
|
+
intervalMs?: number;
|
|
118
|
+
timeoutMs?: number;
|
|
119
|
+
onProgress?: (run: DesignVerifierStatusResult) => void;
|
|
120
|
+
sleep?: (ms: number) => Promise<void>;
|
|
121
|
+
}): Promise<{
|
|
122
|
+
status: 'terminal';
|
|
123
|
+
run: DesignVerifierStatusResult;
|
|
124
|
+
} | {
|
|
125
|
+
status: 'timeout';
|
|
126
|
+
} | {
|
|
127
|
+
status: 'error';
|
|
128
|
+
message: string;
|
|
129
|
+
}>;
|