@haystackeditor/cli 0.15.23 → 0.15.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +34 -0
- package/dist/commands/verify-hosted-mcp.js +6 -0
- package/dist/commands/verify-hosted-reproducibility.d.ts +85 -0
- package/dist/commands/verify-hosted-reproducibility.js +531 -0
- package/dist/commands/verify-hosted.d.ts +5 -3
- package/dist/commands/verify-hosted.js +16 -7
- package/dist/index.js +30 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -105,6 +105,21 @@ haystack verify hosted status cv_<48-lowercase-hex-characters> \
|
|
|
105
105
|
--json
|
|
106
106
|
```
|
|
107
107
|
|
|
108
|
+
To execute a completed run's exact validated risk plan again, keep the same
|
|
109
|
+
repository and commit pair and pass:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
haystack verify hosted start owner/repo \
|
|
113
|
+
--base <40-character-commit-sha> \
|
|
114
|
+
--head <40-character-commit-sha> \
|
|
115
|
+
--replay-plan-from cv_<48-lowercase-hex-characters> \
|
|
116
|
+
--json
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
A successful replay emits a schema-9 terminal result whose
|
|
120
|
+
`riskPlanReplayReceipt` binds the source run and canonical risk-plan hash. The
|
|
121
|
+
replay does not mint or deliver a planning-model credential.
|
|
122
|
+
|
|
108
123
|
Agents can use the same production lifecycle over stdio MCP:
|
|
109
124
|
|
|
110
125
|
```bash
|
|
@@ -116,6 +131,25 @@ The hosted server exposes only `verify_start`, `verify_status`, and
|
|
|
116
131
|
credentials, repository source, provider metadata, and sandbox mutation are
|
|
117
132
|
not exposed.
|
|
118
133
|
|
|
134
|
+
Freeze one production plan and repeat its exact execution to detect executor or
|
|
135
|
+
assessment drift:
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
haystack verify hosted reproducibility owner/repo \
|
|
139
|
+
--base <40-character-commit-sha> \
|
|
140
|
+
--head <40-character-commit-sha> \
|
|
141
|
+
--runs 3 \
|
|
142
|
+
--json
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
The first run creates the source plan; every later run names it through
|
|
146
|
+
`replayPlanFromRunId`. Each run gets a distinct idempotency key derived from one
|
|
147
|
+
printed series key. Pass `--series-key <key>` to resume the same series without
|
|
148
|
+
creating duplicate runs. The report verifies schema-9 replay provenance and
|
|
149
|
+
compares stable risk-cell, capability, and comparison IDs. Missing replay
|
|
150
|
+
provenance, assessment drift, timeouts, unsupported receipts, and inconclusive
|
|
151
|
+
cells all prevent a reproducible verdict and exit 2.
|
|
152
|
+
|
|
119
153
|
**`haystack setup --json`** speaks NDJSON: events (`question`, `permission`,
|
|
120
154
|
`progress`, `result`) on stdout, replies on stdin keyed by
|
|
121
155
|
`requestID`/`permissionID`. `haystack schema setup` documents the full
|
|
@@ -53,6 +53,10 @@ export const HOSTED_VERIFY_TOOLS = [
|
|
|
53
53
|
pattern: '^[A-Za-z0-9][A-Za-z0-9._:-]{7,127}$',
|
|
54
54
|
description: 'Optional stable retry key. Omit to derive it from the exact repository and commits.',
|
|
55
55
|
},
|
|
56
|
+
replay_plan_from_run_id: {
|
|
57
|
+
...runIdInput,
|
|
58
|
+
description: 'Optional completed owned run whose exact validated risk plan must be replayed.',
|
|
59
|
+
},
|
|
56
60
|
account: accountInput,
|
|
57
61
|
},
|
|
58
62
|
required: ['repository', 'base_commit', 'head_commit'],
|
|
@@ -120,6 +124,7 @@ function assertToolInputFields(name, input) {
|
|
|
120
124
|
'base_commit',
|
|
121
125
|
'head_commit',
|
|
122
126
|
'idempotency_key',
|
|
127
|
+
'replay_plan_from_run_id',
|
|
123
128
|
'repository',
|
|
124
129
|
]);
|
|
125
130
|
return;
|
|
@@ -147,6 +152,7 @@ export async function executeHostedVerifyTool(name, input, context) {
|
|
|
147
152
|
base: requireInputString(input, 'base_commit'),
|
|
148
153
|
head: requireInputString(input, 'head_commit'),
|
|
149
154
|
idempotencyKey: optionalInputString(input, 'idempotency_key'),
|
|
155
|
+
replayPlanFrom: optionalInputString(input, 'replay_plan_from_run_id'),
|
|
150
156
|
wait: false,
|
|
151
157
|
}, context.token);
|
|
152
158
|
return buildHostedJson('start', started.state, {
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import { type HostedRunState } from './verify-hosted.js';
|
|
2
|
+
export interface HostedReproducibilityOptions {
|
|
3
|
+
base: string;
|
|
4
|
+
head: string;
|
|
5
|
+
runs?: string;
|
|
6
|
+
seriesKey?: string;
|
|
7
|
+
account?: string;
|
|
8
|
+
interval?: string;
|
|
9
|
+
timeout?: string;
|
|
10
|
+
json?: boolean;
|
|
11
|
+
}
|
|
12
|
+
interface HostedReproducibilityRun {
|
|
13
|
+
state: HostedRunState;
|
|
14
|
+
timedOut: boolean;
|
|
15
|
+
}
|
|
16
|
+
interface HostedReproducibilityIdentity {
|
|
17
|
+
seriesKey: string;
|
|
18
|
+
repository: string;
|
|
19
|
+
baseCommit: string;
|
|
20
|
+
headCommit: string;
|
|
21
|
+
requestedRuns?: number;
|
|
22
|
+
}
|
|
23
|
+
interface ParsedAssessment {
|
|
24
|
+
schemaVersion: 8 | 9;
|
|
25
|
+
riskPlanSha256: string;
|
|
26
|
+
replaySourceRunId: string | null;
|
|
27
|
+
replaySourceResultSchemaVersion: 8 | 9 | null;
|
|
28
|
+
planIdentities: Array<{
|
|
29
|
+
cellId: string;
|
|
30
|
+
capabilityId: string | null;
|
|
31
|
+
}>;
|
|
32
|
+
cellIds: string[];
|
|
33
|
+
status: string;
|
|
34
|
+
outcomes: Array<{
|
|
35
|
+
cellId: string;
|
|
36
|
+
capabilityId: string | null;
|
|
37
|
+
comparisonId: string | null;
|
|
38
|
+
status: string;
|
|
39
|
+
reasonCode: string;
|
|
40
|
+
}>;
|
|
41
|
+
}
|
|
42
|
+
interface RunFailure {
|
|
43
|
+
run_id: string;
|
|
44
|
+
status: string;
|
|
45
|
+
reason: string;
|
|
46
|
+
}
|
|
47
|
+
interface Variant {
|
|
48
|
+
fingerprint: string;
|
|
49
|
+
run_ids: string[];
|
|
50
|
+
cell_ids: string[];
|
|
51
|
+
}
|
|
52
|
+
interface OutcomeVariant {
|
|
53
|
+
fingerprint: string;
|
|
54
|
+
run_ids: string[];
|
|
55
|
+
assessment_status: string;
|
|
56
|
+
outcomes: ParsedAssessment['outcomes'];
|
|
57
|
+
}
|
|
58
|
+
export interface HostedReproducibilityReport {
|
|
59
|
+
series_key: string;
|
|
60
|
+
repository: string;
|
|
61
|
+
base_commit: string;
|
|
62
|
+
head_commit: string;
|
|
63
|
+
requested_runs: number;
|
|
64
|
+
launched_runs: number;
|
|
65
|
+
unstarted_runs: number;
|
|
66
|
+
analyzable_runs: number;
|
|
67
|
+
run_ids: string[];
|
|
68
|
+
replay_source_run_id: string | null;
|
|
69
|
+
plan_replay_verified: boolean;
|
|
70
|
+
planner_reproducibility_tested: false;
|
|
71
|
+
planner_reproducible: boolean;
|
|
72
|
+
assessment_reproducible: boolean;
|
|
73
|
+
reproducible: boolean;
|
|
74
|
+
plan_variants: Variant[];
|
|
75
|
+
assessment_variants: OutcomeVariant[];
|
|
76
|
+
plan_drift_cell_ids: string[];
|
|
77
|
+
assessment_drift_cell_ids: string[];
|
|
78
|
+
inconclusive_cell_ids: string[];
|
|
79
|
+
unrunnable_runs: RunFailure[];
|
|
80
|
+
}
|
|
81
|
+
export declare function hostedReproducibilityRunKey(seriesKey: string, repository: string, baseCommit: string, headCommit: string, ordinal: number): string;
|
|
82
|
+
export declare function analyzeHostedReproducibility(runs: HostedReproducibilityRun[], identity: HostedReproducibilityIdentity): HostedReproducibilityReport;
|
|
83
|
+
export declare function isHostedReproducibilityReplaySource(run: HostedReproducibilityRun, identity: HostedReproducibilityIdentity): boolean;
|
|
84
|
+
export declare function verifyHostedReproducibilityCommand(repositoryArg: string, options: HostedReproducibilityOptions): Promise<void>;
|
|
85
|
+
export {};
|
|
@@ -0,0 +1,531 @@
|
|
|
1
|
+
import { createHash, randomBytes } from 'node:crypto';
|
|
2
|
+
import chalk from 'chalk';
|
|
3
|
+
import { withSchema } from '../schema.js';
|
|
4
|
+
import { resolveAuthContext } from '../utils/auth.js';
|
|
5
|
+
import { hostedPollingOptions, parseHostedRepository, startHostedVerification, waitForHostedVerification, } from './verify-hosted.js';
|
|
6
|
+
const CELL_ID = /^cell_[0-9a-f]{32}$/;
|
|
7
|
+
const CAPABILITY_ID = /^repository_driver_[0-9a-f]{64}$/;
|
|
8
|
+
const COMPARISON_ID = /^driver_comparison_[0-9a-f]{64}$/;
|
|
9
|
+
const RUN_ID = /^cv_[0-9a-f]{48}$/;
|
|
10
|
+
const SHA256 = /^[0-9a-f]{64}$/;
|
|
11
|
+
const SERIES_KEY = /^[A-Za-z0-9][A-Za-z0-9._:-]{7,63}$/;
|
|
12
|
+
const TERMINAL_STATUSES = new Set([
|
|
13
|
+
'complete',
|
|
14
|
+
'errored',
|
|
15
|
+
'expired',
|
|
16
|
+
'terminated',
|
|
17
|
+
]);
|
|
18
|
+
const ASSESSMENT_STATUSES = new Set([
|
|
19
|
+
'improvement',
|
|
20
|
+
'inconclusive',
|
|
21
|
+
'regression',
|
|
22
|
+
'unchanged',
|
|
23
|
+
]);
|
|
24
|
+
const ASSESSMENT_REASON_CODES = new Set([
|
|
25
|
+
'case_universe_changed',
|
|
26
|
+
'coverage_increased',
|
|
27
|
+
'coverage_reduced',
|
|
28
|
+
'inconclusive',
|
|
29
|
+
'mixed_outcomes',
|
|
30
|
+
'new_failures',
|
|
31
|
+
'no_bound_driver',
|
|
32
|
+
'resolved_failures',
|
|
33
|
+
'unchanged',
|
|
34
|
+
]);
|
|
35
|
+
const OVERALL_ASSESSMENT_STATUSES = new Set([
|
|
36
|
+
'inconclusive',
|
|
37
|
+
'no_regressions_detected',
|
|
38
|
+
'regressions_detected',
|
|
39
|
+
]);
|
|
40
|
+
const DEFAULT_RUNS = 3;
|
|
41
|
+
const MAX_RUNS = 10;
|
|
42
|
+
function record(value, label) {
|
|
43
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) {
|
|
44
|
+
throw new Error(`${label} is malformed`);
|
|
45
|
+
}
|
|
46
|
+
return value;
|
|
47
|
+
}
|
|
48
|
+
function array(value, label) {
|
|
49
|
+
if (!Array.isArray(value))
|
|
50
|
+
throw new Error(`${label} is malformed`);
|
|
51
|
+
return value;
|
|
52
|
+
}
|
|
53
|
+
function boundedInteger(value, fallback, label, minimum, maximum) {
|
|
54
|
+
if (value === undefined)
|
|
55
|
+
return fallback;
|
|
56
|
+
const parsed = Number(value);
|
|
57
|
+
if (!Number.isSafeInteger(parsed)
|
|
58
|
+
|| parsed < minimum
|
|
59
|
+
|| parsed > maximum) {
|
|
60
|
+
throw new Error(`${label} must be an integer from ${minimum} through ${maximum}.`);
|
|
61
|
+
}
|
|
62
|
+
return parsed;
|
|
63
|
+
}
|
|
64
|
+
function requireSeriesKey(value) {
|
|
65
|
+
const key = value
|
|
66
|
+
?? `series:${new Date().toISOString().replace(/[^0-9]/g, '').slice(0, 14)}:${randomBytes(6).toString('hex')}`;
|
|
67
|
+
if (!SERIES_KEY.test(key)) {
|
|
68
|
+
throw new Error('--series-key must be 8-64 characters and use only letters, numbers, ".", "_", ":", or "-".');
|
|
69
|
+
}
|
|
70
|
+
return key;
|
|
71
|
+
}
|
|
72
|
+
export function hostedReproducibilityRunKey(seriesKey, repository, baseCommit, headCommit, ordinal) {
|
|
73
|
+
const digest = createHash('sha256')
|
|
74
|
+
.update(JSON.stringify([
|
|
75
|
+
seriesKey,
|
|
76
|
+
repository,
|
|
77
|
+
baseCommit,
|
|
78
|
+
headCommit,
|
|
79
|
+
ordinal,
|
|
80
|
+
]))
|
|
81
|
+
.digest('hex');
|
|
82
|
+
return `repro:${digest}`;
|
|
83
|
+
}
|
|
84
|
+
function stable(value) {
|
|
85
|
+
if (Array.isArray(value))
|
|
86
|
+
return value.map(stable);
|
|
87
|
+
if (value && typeof value === 'object') {
|
|
88
|
+
const object = value;
|
|
89
|
+
return Object.fromEntries(Object.keys(object)
|
|
90
|
+
.sort()
|
|
91
|
+
.map(key => [key, stable(object[key])]));
|
|
92
|
+
}
|
|
93
|
+
return value;
|
|
94
|
+
}
|
|
95
|
+
function fingerprint(value) {
|
|
96
|
+
return createHash('sha256')
|
|
97
|
+
.update(JSON.stringify(stable(value)))
|
|
98
|
+
.digest('hex');
|
|
99
|
+
}
|
|
100
|
+
function expectedCellStatus(reasonCode) {
|
|
101
|
+
if (reasonCode === 'new_failures'
|
|
102
|
+
|| reasonCode === 'coverage_reduced'
|
|
103
|
+
|| reasonCode === 'mixed_outcomes') {
|
|
104
|
+
return 'regression';
|
|
105
|
+
}
|
|
106
|
+
if (reasonCode === 'resolved_failures'
|
|
107
|
+
|| reasonCode === 'coverage_increased') {
|
|
108
|
+
return 'improvement';
|
|
109
|
+
}
|
|
110
|
+
if (reasonCode === 'unchanged')
|
|
111
|
+
return 'unchanged';
|
|
112
|
+
return 'inconclusive';
|
|
113
|
+
}
|
|
114
|
+
function expectedOverallStatus(outcomes) {
|
|
115
|
+
if (outcomes.some(outcome => outcome.status === 'regression')) {
|
|
116
|
+
return 'regressions_detected';
|
|
117
|
+
}
|
|
118
|
+
if (outcomes.length === 0
|
|
119
|
+
|| outcomes.some(outcome => outcome.status === 'inconclusive')) {
|
|
120
|
+
return 'inconclusive';
|
|
121
|
+
}
|
|
122
|
+
return 'no_regressions_detected';
|
|
123
|
+
}
|
|
124
|
+
function requireCount(receipt, field, expected) {
|
|
125
|
+
if (receipt[field] !== expected) {
|
|
126
|
+
throw new Error(`driver assessment ${field} is inconsistent`);
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
function parseAssessment(state, identity) {
|
|
130
|
+
if (state.status !== 'complete' || !state.output) {
|
|
131
|
+
throw new Error(`run ended with status ${state.status}`);
|
|
132
|
+
}
|
|
133
|
+
if (state.repository !== identity.repository
|
|
134
|
+
|| state.baseCommit !== identity.baseCommit
|
|
135
|
+
|| state.headCommit !== identity.headCommit) {
|
|
136
|
+
throw new Error('run identity does not match the requested exact launch');
|
|
137
|
+
}
|
|
138
|
+
const output = record(state.output, 'terminal output');
|
|
139
|
+
if ((output.schemaVersion !== 8 && output.schemaVersion !== 9)
|
|
140
|
+
|| output.status !== 'driver_assessment_recorded') {
|
|
141
|
+
throw new Error('run did not produce a schema-8 or schema-9 driver assessment');
|
|
142
|
+
}
|
|
143
|
+
const riskPlan = record(output.riskPlanReceipt, 'risk plan receipt');
|
|
144
|
+
const riskPlanSha256 = fingerprint(riskPlan);
|
|
145
|
+
let replaySourceRunId = null;
|
|
146
|
+
let replaySourceResultSchemaVersion = null;
|
|
147
|
+
if (output.schemaVersion === 9) {
|
|
148
|
+
if (output.planningCredentialDelivery !== 'not-required-plan-replay') {
|
|
149
|
+
throw new Error('schema-9 replay used an invalid planning credential mode');
|
|
150
|
+
}
|
|
151
|
+
const replay = record(output.riskPlanReplayReceipt, 'risk plan replay receipt');
|
|
152
|
+
const fields = Object.keys(replay);
|
|
153
|
+
if (fields.length !== 6
|
|
154
|
+
|| fields.some(field => ![
|
|
155
|
+
'schemaVersion',
|
|
156
|
+
'sourceContentPersisted',
|
|
157
|
+
'sourceResultSchemaVersion',
|
|
158
|
+
'sourceRiskPlanReceiptSha256',
|
|
159
|
+
'sourceRunId',
|
|
160
|
+
'status',
|
|
161
|
+
].includes(field))
|
|
162
|
+
|| replay.schemaVersion !== 1
|
|
163
|
+
|| replay.status !== 'risk_plan_replay_verified'
|
|
164
|
+
|| typeof replay.sourceRunId !== 'string'
|
|
165
|
+
|| !RUN_ID.test(replay.sourceRunId)
|
|
166
|
+
|| (replay.sourceResultSchemaVersion !== 8
|
|
167
|
+
&& replay.sourceResultSchemaVersion !== 9)
|
|
168
|
+
|| replay.sourceRiskPlanReceiptSha256 !== riskPlanSha256
|
|
169
|
+
|| typeof replay.sourceRiskPlanReceiptSha256 !== 'string'
|
|
170
|
+
|| !SHA256.test(replay.sourceRiskPlanReceiptSha256)
|
|
171
|
+
|| replay.sourceContentPersisted !== false) {
|
|
172
|
+
throw new Error('risk plan replay receipt is malformed');
|
|
173
|
+
}
|
|
174
|
+
replaySourceRunId = replay.sourceRunId;
|
|
175
|
+
replaySourceResultSchemaVersion = replay.sourceResultSchemaVersion;
|
|
176
|
+
}
|
|
177
|
+
else if (output.riskPlanReplayReceipt !== undefined) {
|
|
178
|
+
throw new Error('schema-8 result contains unversioned replay provenance');
|
|
179
|
+
}
|
|
180
|
+
else if (output.planningCredentialDelivery
|
|
181
|
+
!== 'per-command-env-gateway-placeholder') {
|
|
182
|
+
throw new Error('schema-8 result used an invalid planning credential mode');
|
|
183
|
+
}
|
|
184
|
+
const assessment = record(output.driverAssessmentReceipt, 'driver assessment receipt');
|
|
185
|
+
const planCells = array(riskPlan.cells, 'risk plan cells');
|
|
186
|
+
const assessmentCells = array(assessment.cells, 'assessment cells');
|
|
187
|
+
if (riskPlan.schemaVersion !== 2
|
|
188
|
+
|| riskPlan.status !== 'risk_plan_ready'
|
|
189
|
+
|| riskPlan.baseCommit !== identity.baseCommit
|
|
190
|
+
|| riskPlan.headCommit !== identity.headCommit
|
|
191
|
+
|| riskPlan.sourceContentPersisted !== false) {
|
|
192
|
+
throw new Error('risk plan receipt identity is malformed');
|
|
193
|
+
}
|
|
194
|
+
const planIdentities = planCells.map((raw) => {
|
|
195
|
+
const cell = record(raw, 'risk plan cell');
|
|
196
|
+
if (typeof cell.cellId !== 'string'
|
|
197
|
+
|| !CELL_ID.test(cell.cellId)
|
|
198
|
+
|| (cell.capabilityId !== null
|
|
199
|
+
&& (typeof cell.capabilityId !== 'string'
|
|
200
|
+
|| !CAPABILITY_ID.test(cell.capabilityId)))) {
|
|
201
|
+
throw new Error('risk plan cell identity is malformed');
|
|
202
|
+
}
|
|
203
|
+
return {
|
|
204
|
+
cellId: cell.cellId,
|
|
205
|
+
capabilityId: cell.capabilityId,
|
|
206
|
+
};
|
|
207
|
+
}).sort((left, right) => left.cellId.localeCompare(right.cellId));
|
|
208
|
+
const cellIds = planIdentities.map(cell => cell.cellId);
|
|
209
|
+
if (cellIds.length === 0
|
|
210
|
+
|| new Set(cellIds).size !== cellIds.length
|
|
211
|
+
|| assessmentCells.length !== cellIds.length) {
|
|
212
|
+
throw new Error('risk plan and assessment cell sets are inconsistent');
|
|
213
|
+
}
|
|
214
|
+
if (assessment.schemaVersion !== 1
|
|
215
|
+
|| typeof assessment.status !== 'string'
|
|
216
|
+
|| !OVERALL_ASSESSMENT_STATUSES.has(assessment.status)
|
|
217
|
+
|| assessment.baseCommit !== identity.baseCommit
|
|
218
|
+
|| assessment.headCommit !== identity.headCommit
|
|
219
|
+
|| typeof assessment.driverComparisonReceiptSha256 !== 'string'
|
|
220
|
+
|| !SHA256.test(assessment.driverComparisonReceiptSha256)
|
|
221
|
+
|| assessment.sourceContentPersisted !== false) {
|
|
222
|
+
throw new Error('driver assessment receipt identity is malformed');
|
|
223
|
+
}
|
|
224
|
+
const outcomes = assessmentCells.map((raw) => {
|
|
225
|
+
const cell = record(raw, 'assessment cell');
|
|
226
|
+
if (typeof cell.cellId !== 'string'
|
|
227
|
+
|| !CELL_ID.test(cell.cellId)
|
|
228
|
+
|| (cell.capabilityId !== null
|
|
229
|
+
&& (typeof cell.capabilityId !== 'string'
|
|
230
|
+
|| !CAPABILITY_ID.test(cell.capabilityId)))
|
|
231
|
+
|| (cell.comparisonId !== null
|
|
232
|
+
&& (typeof cell.comparisonId !== 'string'
|
|
233
|
+
|| !COMPARISON_ID.test(cell.comparisonId)))
|
|
234
|
+
|| typeof cell.status !== 'string'
|
|
235
|
+
|| !ASSESSMENT_STATUSES.has(cell.status)
|
|
236
|
+
|| typeof cell.reasonCode !== 'string'
|
|
237
|
+
|| !ASSESSMENT_REASON_CODES.has(cell.reasonCode)
|
|
238
|
+
|| cell.status !== expectedCellStatus(cell.reasonCode)
|
|
239
|
+
|| (cell.capabilityId === null
|
|
240
|
+
? (cell.comparisonId !== null
|
|
241
|
+
|| cell.reasonCode !== 'no_bound_driver')
|
|
242
|
+
: (cell.comparisonId === null
|
|
243
|
+
|| cell.reasonCode === 'no_bound_driver'))) {
|
|
244
|
+
throw new Error('assessment cell outcome is malformed');
|
|
245
|
+
}
|
|
246
|
+
return {
|
|
247
|
+
cellId: cell.cellId,
|
|
248
|
+
capabilityId: cell.capabilityId,
|
|
249
|
+
comparisonId: cell.comparisonId,
|
|
250
|
+
status: cell.status,
|
|
251
|
+
reasonCode: cell.reasonCode,
|
|
252
|
+
};
|
|
253
|
+
}).sort((left, right) => left.cellId.localeCompare(right.cellId));
|
|
254
|
+
if (new Set(outcomes.map(outcome => outcome.cellId)).size !== outcomes.length
|
|
255
|
+
|| outcomes.some((outcome, index) => outcome.cellId !== cellIds[index]
|
|
256
|
+
|| outcome.capabilityId !== planIdentities[index].capabilityId)) {
|
|
257
|
+
throw new Error('risk plan and assessment cell identities do not match');
|
|
258
|
+
}
|
|
259
|
+
const counts = {
|
|
260
|
+
comparedCellCount: outcomes.filter(outcome => outcome.status !== 'inconclusive').length,
|
|
261
|
+
regressionCellCount: outcomes.filter(outcome => outcome.status === 'regression').length,
|
|
262
|
+
improvementCellCount: outcomes.filter(outcome => outcome.status === 'improvement').length,
|
|
263
|
+
unchangedCellCount: outcomes.filter(outcome => outcome.status === 'unchanged').length,
|
|
264
|
+
inconclusiveCellCount: outcomes.filter(outcome => outcome.status === 'inconclusive').length,
|
|
265
|
+
};
|
|
266
|
+
requireCount(assessment, 'plannedCellCount', outcomes.length);
|
|
267
|
+
for (const [field, count] of Object.entries(counts)) {
|
|
268
|
+
requireCount(assessment, field, count);
|
|
269
|
+
}
|
|
270
|
+
if (assessment.status !== expectedOverallStatus(outcomes)) {
|
|
271
|
+
throw new Error('driver assessment overall status is inconsistent');
|
|
272
|
+
}
|
|
273
|
+
return {
|
|
274
|
+
schemaVersion: output.schemaVersion,
|
|
275
|
+
riskPlanSha256,
|
|
276
|
+
replaySourceRunId,
|
|
277
|
+
replaySourceResultSchemaVersion,
|
|
278
|
+
planIdentities,
|
|
279
|
+
cellIds,
|
|
280
|
+
status: assessment.status,
|
|
281
|
+
outcomes,
|
|
282
|
+
};
|
|
283
|
+
}
|
|
284
|
+
function groupedPlanVariants(parsed) {
|
|
285
|
+
const variants = new Map();
|
|
286
|
+
for (const item of parsed) {
|
|
287
|
+
const digest = fingerprint(item.assessment.planIdentities);
|
|
288
|
+
const existing = variants.get(digest);
|
|
289
|
+
if (existing) {
|
|
290
|
+
existing.run_ids.push(item.runId);
|
|
291
|
+
}
|
|
292
|
+
else {
|
|
293
|
+
variants.set(digest, {
|
|
294
|
+
fingerprint: digest,
|
|
295
|
+
run_ids: [item.runId],
|
|
296
|
+
cell_ids: item.assessment.cellIds,
|
|
297
|
+
});
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
return [...variants.values()].sort((left, right) => left.fingerprint.localeCompare(right.fingerprint));
|
|
301
|
+
}
|
|
302
|
+
function groupedOutcomeVariants(parsed) {
|
|
303
|
+
const variants = new Map();
|
|
304
|
+
for (const item of parsed) {
|
|
305
|
+
const digest = fingerprint({
|
|
306
|
+
status: item.assessment.status,
|
|
307
|
+
outcomes: item.assessment.outcomes,
|
|
308
|
+
});
|
|
309
|
+
const existing = variants.get(digest);
|
|
310
|
+
if (existing) {
|
|
311
|
+
existing.run_ids.push(item.runId);
|
|
312
|
+
}
|
|
313
|
+
else {
|
|
314
|
+
variants.set(digest, {
|
|
315
|
+
fingerprint: digest,
|
|
316
|
+
run_ids: [item.runId],
|
|
317
|
+
assessment_status: item.assessment.status,
|
|
318
|
+
outcomes: item.assessment.outcomes,
|
|
319
|
+
});
|
|
320
|
+
}
|
|
321
|
+
}
|
|
322
|
+
return [...variants.values()].sort((left, right) => left.fingerprint.localeCompare(right.fingerprint));
|
|
323
|
+
}
|
|
324
|
+
function driftIds(parsed, values) {
|
|
325
|
+
const byCell = new Map();
|
|
326
|
+
for (const [runIndex, item] of parsed.entries()) {
|
|
327
|
+
const present = values(item.assessment);
|
|
328
|
+
for (const [cellId, observed] of byCell) {
|
|
329
|
+
observed.add(present.has(cellId)
|
|
330
|
+
? JSON.stringify(present.get(cellId))
|
|
331
|
+
: '__missing__');
|
|
332
|
+
}
|
|
333
|
+
for (const [cellId, value] of present) {
|
|
334
|
+
if (byCell.has(cellId))
|
|
335
|
+
continue;
|
|
336
|
+
const observed = new Set();
|
|
337
|
+
if (runIndex > 0)
|
|
338
|
+
observed.add('__missing__');
|
|
339
|
+
observed.add(JSON.stringify(value));
|
|
340
|
+
byCell.set(cellId, observed);
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
return [...byCell.entries()]
|
|
344
|
+
.filter(([, observed]) => observed.size > 1)
|
|
345
|
+
.map(([cellId]) => cellId)
|
|
346
|
+
.sort();
|
|
347
|
+
}
|
|
348
|
+
export function analyzeHostedReproducibility(runs, identity) {
|
|
349
|
+
const parsed = [];
|
|
350
|
+
const unrunnableRuns = [];
|
|
351
|
+
for (const run of runs) {
|
|
352
|
+
if (run.timedOut) {
|
|
353
|
+
unrunnableRuns.push({
|
|
354
|
+
run_id: run.state.runId,
|
|
355
|
+
status: run.state.status,
|
|
356
|
+
reason: 'wait_timed_out',
|
|
357
|
+
});
|
|
358
|
+
continue;
|
|
359
|
+
}
|
|
360
|
+
try {
|
|
361
|
+
parsed.push({
|
|
362
|
+
runId: run.state.runId,
|
|
363
|
+
assessment: parseAssessment(run.state, identity),
|
|
364
|
+
});
|
|
365
|
+
}
|
|
366
|
+
catch (error) {
|
|
367
|
+
unrunnableRuns.push({
|
|
368
|
+
run_id: run.state.runId,
|
|
369
|
+
status: run.state.status,
|
|
370
|
+
reason: error instanceof Error ? error.message : String(error),
|
|
371
|
+
});
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
const planVariants = groupedPlanVariants(parsed);
|
|
375
|
+
const assessmentVariants = groupedOutcomeVariants(parsed);
|
|
376
|
+
const replaySourceRunId = parsed[0]?.runId ?? null;
|
|
377
|
+
const planReplayVerified = parsed.length === runs.length
|
|
378
|
+
&& parsed.length > 1
|
|
379
|
+
&& parsed[0]?.assessment.schemaVersion === 8
|
|
380
|
+
&& parsed[0]?.assessment.replaySourceRunId === null
|
|
381
|
+
&& parsed.slice(1).every(item => item.assessment.schemaVersion === 9
|
|
382
|
+
&& item.assessment.replaySourceRunId === replaySourceRunId
|
|
383
|
+
&& item.assessment.replaySourceResultSchemaVersion
|
|
384
|
+
=== parsed[0].assessment.schemaVersion
|
|
385
|
+
&& item.assessment.riskPlanSha256
|
|
386
|
+
=== parsed[0].assessment.riskPlanSha256);
|
|
387
|
+
const assessmentReproducible = planReplayVerified && assessmentVariants.length === 1;
|
|
388
|
+
const planDriftCellIds = driftIds(parsed, assessment => new Map(assessment.planIdentities.map(plan => [
|
|
389
|
+
plan.cellId,
|
|
390
|
+
{ capabilityId: plan.capabilityId },
|
|
391
|
+
])));
|
|
392
|
+
const assessmentDriftCellIds = driftIds(parsed, assessment => new Map(assessment.outcomes.map(outcome => [
|
|
393
|
+
outcome.cellId,
|
|
394
|
+
{
|
|
395
|
+
capabilityId: outcome.capabilityId,
|
|
396
|
+
comparisonId: outcome.comparisonId,
|
|
397
|
+
status: outcome.status,
|
|
398
|
+
reasonCode: outcome.reasonCode,
|
|
399
|
+
},
|
|
400
|
+
])));
|
|
401
|
+
const inconclusiveCellIds = [...new Set(parsed.flatMap(item => item.assessment.outcomes
|
|
402
|
+
.filter(outcome => outcome.status === 'inconclusive')
|
|
403
|
+
.map(outcome => outcome.cellId)))].sort();
|
|
404
|
+
const requestedRuns = identity.requestedRuns ?? runs.length;
|
|
405
|
+
return {
|
|
406
|
+
series_key: identity.seriesKey,
|
|
407
|
+
repository: identity.repository,
|
|
408
|
+
base_commit: identity.baseCommit,
|
|
409
|
+
head_commit: identity.headCommit,
|
|
410
|
+
requested_runs: requestedRuns,
|
|
411
|
+
launched_runs: runs.length,
|
|
412
|
+
unstarted_runs: Math.max(0, requestedRuns - runs.length),
|
|
413
|
+
analyzable_runs: parsed.length,
|
|
414
|
+
run_ids: runs.map(run => run.state.runId),
|
|
415
|
+
replay_source_run_id: replaySourceRunId,
|
|
416
|
+
plan_replay_verified: planReplayVerified,
|
|
417
|
+
planner_reproducibility_tested: false,
|
|
418
|
+
planner_reproducible: false,
|
|
419
|
+
assessment_reproducible: assessmentReproducible,
|
|
420
|
+
reproducible: planReplayVerified
|
|
421
|
+
&& assessmentReproducible
|
|
422
|
+
&& inconclusiveCellIds.length === 0,
|
|
423
|
+
plan_variants: planVariants,
|
|
424
|
+
assessment_variants: assessmentVariants,
|
|
425
|
+
plan_drift_cell_ids: planDriftCellIds,
|
|
426
|
+
assessment_drift_cell_ids: assessmentDriftCellIds,
|
|
427
|
+
inconclusive_cell_ids: inconclusiveCellIds,
|
|
428
|
+
unrunnable_runs: unrunnableRuns,
|
|
429
|
+
};
|
|
430
|
+
}
|
|
431
|
+
export function isHostedReproducibilityReplaySource(run, identity) {
|
|
432
|
+
if (run.timedOut || run.state.status !== 'complete')
|
|
433
|
+
return false;
|
|
434
|
+
try {
|
|
435
|
+
const assessment = parseAssessment(run.state, identity);
|
|
436
|
+
return assessment.schemaVersion === 8
|
|
437
|
+
&& assessment.replaySourceRunId === null;
|
|
438
|
+
}
|
|
439
|
+
catch {
|
|
440
|
+
return false;
|
|
441
|
+
}
|
|
442
|
+
}
|
|
443
|
+
function renderHostedReproducibility(report) {
|
|
444
|
+
const icon = report.reproducible ? chalk.green('✓') : chalk.yellow('!');
|
|
445
|
+
console.log(`${icon} Hosted reproducibility: ${report.reproducible ? 'reproducible' : 'not established'}`);
|
|
446
|
+
console.log(` ${chalk.dim('Runs:')} ${report.analyzable_runs}/${report.requested_runs} analyzable; ${report.launched_runs} launched`);
|
|
447
|
+
console.log(` ${chalk.dim('Plan variants:')} ${report.plan_variants.length}`);
|
|
448
|
+
console.log(` ${chalk.dim('Exact plan replay:')} ${report.plan_replay_verified ? 'verified' : 'not verified'}`);
|
|
449
|
+
console.log(` ${chalk.dim('Independent planner reproducibility:')} not tested`);
|
|
450
|
+
console.log(` ${chalk.dim('Assessment variants:')} ${report.assessment_variants.length}`);
|
|
451
|
+
if (report.plan_drift_cell_ids.length > 0) {
|
|
452
|
+
console.log(` ${chalk.dim('Plan drift:')} ${report.plan_drift_cell_ids.length} stable cell id(s)`);
|
|
453
|
+
}
|
|
454
|
+
if (report.assessment_drift_cell_ids.length > 0) {
|
|
455
|
+
console.log(` ${chalk.dim('Assessment drift:')} ${report.assessment_drift_cell_ids.length} cell(s)`);
|
|
456
|
+
}
|
|
457
|
+
if (report.inconclusive_cell_ids.length > 0) {
|
|
458
|
+
console.log(` ${chalk.dim('Inconclusive:')} ${report.inconclusive_cell_ids.length} cell(s)`);
|
|
459
|
+
}
|
|
460
|
+
if (report.unrunnable_runs.length > 0) {
|
|
461
|
+
console.log(` ${chalk.dim('Unrunnable:')} ${report.unrunnable_runs.length} run(s)`);
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
export async function verifyHostedReproducibilityCommand(repositoryArg, options) {
|
|
465
|
+
const repository = parseHostedRepository(repositoryArg);
|
|
466
|
+
const requestedRuns = boundedInteger(options.runs, DEFAULT_RUNS, '--runs', 2, MAX_RUNS);
|
|
467
|
+
const seriesKey = requireSeriesKey(options.seriesKey);
|
|
468
|
+
const auth = await resolveAuthContext({
|
|
469
|
+
preferredLogin: options.account,
|
|
470
|
+
owner: repository.owner,
|
|
471
|
+
repo: repository.repository,
|
|
472
|
+
});
|
|
473
|
+
console.error(chalk.dim(`Hosted reproducibility series: ${seriesKey}`));
|
|
474
|
+
const runs = [];
|
|
475
|
+
for (let index = 0; index < requestedRuns; index += 1) {
|
|
476
|
+
if (!options.json) {
|
|
477
|
+
console.error(chalk.dim(`Starting hosted reproducibility run ${index + 1}/${requestedRuns}…`));
|
|
478
|
+
}
|
|
479
|
+
const started = await startHostedVerification(repositoryArg, {
|
|
480
|
+
base: options.base,
|
|
481
|
+
head: options.head,
|
|
482
|
+
idempotencyKey: hostedReproducibilityRunKey(seriesKey, repository.fullName, options.base, options.head, index + 1),
|
|
483
|
+
replayPlanFrom: index === 0
|
|
484
|
+
? undefined
|
|
485
|
+
: runs[0]?.state.runId,
|
|
486
|
+
wait: false,
|
|
487
|
+
}, auth.token);
|
|
488
|
+
let state = started.state;
|
|
489
|
+
let timedOut = false;
|
|
490
|
+
if (!TERMINAL_STATUSES.has(state.status)) {
|
|
491
|
+
const waited = await waitForHostedVerification(state.runId, auth.token, hostedPollingOptions(options.interval, options.timeout, current => {
|
|
492
|
+
if (!options.json) {
|
|
493
|
+
console.error(chalk.dim(`Run ${index + 1}/${requestedRuns}: ${current.status}…`));
|
|
494
|
+
}
|
|
495
|
+
}));
|
|
496
|
+
state = waited.state;
|
|
497
|
+
timedOut = waited.timedOut;
|
|
498
|
+
}
|
|
499
|
+
runs.push({ state, timedOut });
|
|
500
|
+
if (index === 0
|
|
501
|
+
&& !isHostedReproducibilityReplaySource(runs[0], {
|
|
502
|
+
seriesKey,
|
|
503
|
+
repository: repository.fullName,
|
|
504
|
+
baseCommit: options.base,
|
|
505
|
+
headCommit: options.head,
|
|
506
|
+
})) {
|
|
507
|
+
if (!options.json) {
|
|
508
|
+
console.error(chalk.yellow('Source run is not replayable; later runs were not launched.'));
|
|
509
|
+
}
|
|
510
|
+
break;
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
const report = analyzeHostedReproducibility(runs, {
|
|
514
|
+
seriesKey,
|
|
515
|
+
repository: repository.fullName,
|
|
516
|
+
baseCommit: options.base,
|
|
517
|
+
headCommit: options.head,
|
|
518
|
+
requestedRuns,
|
|
519
|
+
});
|
|
520
|
+
if (options.json) {
|
|
521
|
+
process.stdout.write(`${JSON.stringify(withSchema('cloud-verifier', {
|
|
522
|
+
operation: 'reproducibility',
|
|
523
|
+
...report,
|
|
524
|
+
}), null, 2)}\n`);
|
|
525
|
+
}
|
|
526
|
+
else {
|
|
527
|
+
renderHostedReproducibility(report);
|
|
528
|
+
}
|
|
529
|
+
if (!report.reproducible)
|
|
530
|
+
process.exitCode = 2;
|
|
531
|
+
}
|
|
@@ -2,6 +2,7 @@ export interface HostedStartOptions {
|
|
|
2
2
|
base: string;
|
|
3
3
|
head: string;
|
|
4
4
|
idempotencyKey?: string;
|
|
5
|
+
replayPlanFrom?: string;
|
|
5
6
|
account?: string;
|
|
6
7
|
wait?: boolean;
|
|
7
8
|
interval?: string;
|
|
@@ -15,7 +16,7 @@ export interface HostedStatusOptions {
|
|
|
15
16
|
timeout?: string;
|
|
16
17
|
json?: boolean;
|
|
17
18
|
}
|
|
18
|
-
interface HostedRunState {
|
|
19
|
+
export interface HostedRunState {
|
|
19
20
|
runId: string;
|
|
20
21
|
status: string;
|
|
21
22
|
repository?: string;
|
|
@@ -26,7 +27,7 @@ interface HostedRunState {
|
|
|
26
27
|
output?: Record<string, unknown>;
|
|
27
28
|
error?: string;
|
|
28
29
|
}
|
|
29
|
-
interface HostedPollOptions {
|
|
30
|
+
export interface HostedPollOptions {
|
|
30
31
|
intervalMs: number;
|
|
31
32
|
timeoutMs: number;
|
|
32
33
|
onStatus?: (state: HostedRunState) => void;
|
|
@@ -41,7 +42,8 @@ export declare function parseHostedRepository(value: string): {
|
|
|
41
42
|
repository: string;
|
|
42
43
|
fullName: string;
|
|
43
44
|
};
|
|
44
|
-
export declare function defaultHostedIdempotencyKey(repository: string, baseCommit: string, headCommit: string): string;
|
|
45
|
+
export declare function defaultHostedIdempotencyKey(repository: string, baseCommit: string, headCommit: string, replayPlanFrom?: string): string;
|
|
46
|
+
export declare function hostedPollingOptions(interval: string | undefined, timeout: string | undefined, onStatus?: (state: HostedRunState) => void): HostedPollOptions;
|
|
45
47
|
export declare function startHostedVerification(repositoryArg: string, options: HostedStartOptions, token: string): Promise<{
|
|
46
48
|
state: HostedRunState;
|
|
47
49
|
idempotencyKey: string;
|
|
@@ -121,9 +121,11 @@ function requireCommit(value, flag) {
|
|
|
121
121
|
}
|
|
122
122
|
return value;
|
|
123
123
|
}
|
|
124
|
-
export function defaultHostedIdempotencyKey(repository, baseCommit, headCommit) {
|
|
124
|
+
export function defaultHostedIdempotencyKey(repository, baseCommit, headCommit, replayPlanFrom) {
|
|
125
125
|
const digest = createHash('sha256')
|
|
126
|
-
.update(JSON.stringify(
|
|
126
|
+
.update(JSON.stringify(replayPlanFrom
|
|
127
|
+
? [repository, baseCommit, headCommit, replayPlanFrom]
|
|
128
|
+
: [repository, baseCommit, headCommit]))
|
|
127
129
|
.digest('hex');
|
|
128
130
|
return `cli:${digest}`;
|
|
129
131
|
}
|
|
@@ -142,7 +144,7 @@ function parsePositiveNumber(value, fallback, flag, maximum) {
|
|
|
142
144
|
}
|
|
143
145
|
return parsed;
|
|
144
146
|
}
|
|
145
|
-
function
|
|
147
|
+
export function hostedPollingOptions(interval, timeout, onStatus) {
|
|
146
148
|
return {
|
|
147
149
|
intervalMs: parsePositiveNumber(interval, DEFAULT_INTERVAL_SECONDS, '--interval', 300) * 1000,
|
|
148
150
|
timeoutMs: parsePositiveNumber(timeout, DEFAULT_TIMEOUT_MINUTES, '--timeout', 1_440) * 60_000,
|
|
@@ -153,8 +155,12 @@ export async function startHostedVerification(repositoryArg, options, token) {
|
|
|
153
155
|
const repository = parseHostedRepository(repositoryArg);
|
|
154
156
|
const baseCommit = requireCommit(options.base, '--base');
|
|
155
157
|
const headCommit = requireCommit(options.head, '--head');
|
|
158
|
+
const replayPlanFrom = options.replayPlanFrom;
|
|
159
|
+
if (replayPlanFrom !== undefined && !RUN_ID.test(replayPlanFrom)) {
|
|
160
|
+
throw new Error('--replay-plan-from must match cv_ followed by 48 lowercase hex characters.');
|
|
161
|
+
}
|
|
156
162
|
const idempotencyKey = requireIdempotencyKey(options.idempotencyKey
|
|
157
|
-
?? defaultHostedIdempotencyKey(repository.fullName, baseCommit, headCommit));
|
|
163
|
+
?? defaultHostedIdempotencyKey(repository.fullName, baseCommit, headCommit, replayPlanFrom));
|
|
158
164
|
const response = await haystackJson('/api/agent/cloud-verifier/runs', token, {
|
|
159
165
|
method: 'POST',
|
|
160
166
|
headers: { 'Content-Type': 'application/json' },
|
|
@@ -164,6 +170,9 @@ export async function startHostedVerification(repositoryArg, options, token) {
|
|
|
164
170
|
baseCommit,
|
|
165
171
|
headCommit,
|
|
166
172
|
idempotencyKey,
|
|
173
|
+
...(replayPlanFrom
|
|
174
|
+
? { replayPlanFromRunId: replayPlanFrom }
|
|
175
|
+
: {}),
|
|
167
176
|
}),
|
|
168
177
|
});
|
|
169
178
|
const state = parseHostedRunState(response);
|
|
@@ -200,7 +209,7 @@ function delay(milliseconds) {
|
|
|
200
209
|
export async function waitForHostedVerification(runId, token, options) {
|
|
201
210
|
const startedAt = Date.now();
|
|
202
211
|
let lastStatus;
|
|
203
|
-
|
|
212
|
+
for (;;) {
|
|
204
213
|
const state = await getHostedVerification(runId, token);
|
|
205
214
|
if (state.status !== lastStatus) {
|
|
206
215
|
options.onStatus?.(state);
|
|
@@ -295,7 +304,7 @@ export async function verifyHostedStartCommand(repositoryArg, options) {
|
|
|
295
304
|
let state = started.state;
|
|
296
305
|
let timedOut;
|
|
297
306
|
if (options.wait !== false && !TERMINAL_STATUSES.has(state.status)) {
|
|
298
|
-
const waited = await waitForHostedVerification(state.runId, auth.token,
|
|
307
|
+
const waited = await waitForHostedVerification(state.runId, auth.token, hostedPollingOptions(options.interval, options.timeout, current => {
|
|
299
308
|
if (!options.json) {
|
|
300
309
|
console.error(chalk.dim(`Hosted verifier: ${current.status}…`));
|
|
301
310
|
}
|
|
@@ -322,7 +331,7 @@ export async function verifyHostedStatusCommand(runId, options) {
|
|
|
322
331
|
let state = await getHostedVerification(runId, auth.token);
|
|
323
332
|
let timedOut;
|
|
324
333
|
if (options.wait && !TERMINAL_STATUSES.has(state.status)) {
|
|
325
|
-
const waited = await waitForHostedVerification(runId, auth.token,
|
|
334
|
+
const waited = await waitForHostedVerification(runId, auth.token, hostedPollingOptions(options.interval, options.timeout, current => {
|
|
326
335
|
if (!options.json) {
|
|
327
336
|
console.error(chalk.dim(`Hosted verifier: ${current.status}…`));
|
|
328
337
|
}
|
package/dist/index.js
CHANGED
|
@@ -219,6 +219,7 @@ hostedVerify
|
|
|
219
219
|
.option('--base <sha>', 'Required: exact 40-character lowercase base commit SHA')
|
|
220
220
|
.option('--head <sha>', 'Required: exact 40-character lowercase head commit SHA')
|
|
221
221
|
.option('--idempotency-key <key>', 'Stable retry key (default: deterministic from repository and commits)')
|
|
222
|
+
.option('--replay-plan-from <run-id>', 'Reuse the exact validated risk plan from a completed owned run')
|
|
222
223
|
.option('--account <login>', 'Use a specific saved Haystack account')
|
|
223
224
|
.option('--no-wait', 'Return after the production workflow is queued')
|
|
224
225
|
.option('--interval <seconds>', 'Polling interval while waiting (default 5)')
|
|
@@ -228,9 +229,12 @@ hostedVerify
|
|
|
228
229
|
The default idempotency key makes an exact retry return the original run. Pass
|
|
229
230
|
--idempotency-key with a new stable key only when you intentionally want a
|
|
230
231
|
fresh production execution of the same commit pair.
|
|
232
|
+
--replay-plan-from requires the same repository and exact commit pair and emits
|
|
233
|
+
a schema-9 terminal receipt with replay provenance.
|
|
231
234
|
|
|
232
235
|
Examples:
|
|
233
236
|
haystack verify hosted start owner/repo --base <sha> --head <sha>
|
|
237
|
+
haystack verify hosted start owner/repo --base <sha> --head <sha> --replay-plan-from cv_<48 lowercase hex characters>
|
|
234
238
|
haystack verify hosted start owner/repo --base <sha> --head <sha> --no-wait --json
|
|
235
239
|
`)
|
|
236
240
|
.action(async (repository, _options, cmd) => {
|
|
@@ -256,6 +260,32 @@ Example:
|
|
|
256
260
|
const { verifyHostedStatusCommand } = await import('./commands/verify-hosted.js');
|
|
257
261
|
return runPublicCommand(() => verifyHostedStatusCommand(runId, options), options.json);
|
|
258
262
|
});
|
|
263
|
+
hostedVerify
|
|
264
|
+
.command('reproducibility')
|
|
265
|
+
.description('Replay one exact hosted risk plan and report executor or assessment drift')
|
|
266
|
+
.argument('<repository>', 'Exact GitHub owner/repository name')
|
|
267
|
+
.option('--base <sha>', 'Required: exact 40-character lowercase base commit SHA')
|
|
268
|
+
.option('--head <sha>', 'Required: exact 40-character lowercase head commit SHA')
|
|
269
|
+
.option('--runs <n>', 'Number of sequential production runs (default 3; min 2, max 10)')
|
|
270
|
+
.option('--series-key <key>', 'Stable series key for resumable idempotent retries')
|
|
271
|
+
.option('--account <login>', 'Use a specific saved Haystack account')
|
|
272
|
+
.option('--interval <seconds>', 'Polling interval for each run (default 5)')
|
|
273
|
+
.option('--timeout <minutes>', 'Maximum wait per run (default 35)')
|
|
274
|
+
.option('--json', 'Versioned machine-readable reproducibility report')
|
|
275
|
+
.addHelpText('after', `
|
|
276
|
+
The first run creates a fresh risk plan. Every later run replays that exact
|
|
277
|
+
validated plan against the same repository and commit pair with a distinct,
|
|
278
|
+
series-derived idempotency key. A reproducible verdict requires schema-9 replay
|
|
279
|
+
provenance, stable assessment outcomes, and no inconclusive cells.
|
|
280
|
+
|
|
281
|
+
Example:
|
|
282
|
+
haystack verify hosted reproducibility owner/repo --base <sha> --head <sha> --runs 3
|
|
283
|
+
`)
|
|
284
|
+
.action(async (repository, _options, cmd) => {
|
|
285
|
+
const options = { ...cmd.optsWithGlobals(), ...cmd.opts() };
|
|
286
|
+
const { verifyHostedReproducibilityCommand } = await import('./commands/verify-hosted-reproducibility.js');
|
|
287
|
+
return runPublicCommand(() => verifyHostedReproducibilityCommand(repository, options), options.json);
|
|
288
|
+
});
|
|
259
289
|
hostedVerify
|
|
260
290
|
.command('mcp')
|
|
261
291
|
.description('Run a stdio MCP server for the production verifier lifecycle')
|