harnery 0.37.0 → 0.38.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. package/dist/commander.d.ts +9 -0
  2. package/dist/commander.d.ts.map +1 -1
  3. package/dist/commander.js +7 -2
  4. package/dist/commands/admission.d.ts +21 -0
  5. package/dist/commands/admission.d.ts.map +1 -0
  6. package/dist/commands/admission.js +565 -0
  7. package/dist/commands/agents.d.ts +7 -0
  8. package/dist/commands/agents.d.ts.map +1 -1
  9. package/dist/commands/agents.js +61 -1
  10. package/dist/commands/artifacts.d.ts.map +1 -1
  11. package/dist/commands/artifacts.js +48 -2
  12. package/dist/commands/browse-ai.d.ts +2 -2
  13. package/dist/commands/browse-ai.d.ts.map +1 -1
  14. package/dist/commands/browse-ai.js +6 -4
  15. package/dist/commands/browse.d.ts.map +1 -1
  16. package/dist/commands/browse.js +349 -21
  17. package/dist/commands/fetch.js +1 -0
  18. package/dist/commands/qa-record.d.ts +144 -0
  19. package/dist/commands/qa-record.d.ts.map +1 -0
  20. package/dist/commands/qa-record.js +0 -0
  21. package/dist/commands/qa-run.d.ts +7 -3
  22. package/dist/commands/qa-run.d.ts.map +1 -1
  23. package/dist/commands/qa-run.js +268 -13
  24. package/dist/commands/qa-status.d.ts +71 -0
  25. package/dist/commands/qa-status.d.ts.map +1 -0
  26. package/dist/commands/qa-status.js +490 -0
  27. package/dist/commands/qa-verify.d.ts +40 -0
  28. package/dist/commands/qa-verify.d.ts.map +1 -0
  29. package/dist/commands/qa-verify.js +180 -0
  30. package/dist/commands/review-pack.d.ts +4 -0
  31. package/dist/commands/review-pack.d.ts.map +1 -0
  32. package/dist/commands/review-pack.js +1001 -0
  33. package/dist/core/agents/qa-signal.d.ts +111 -0
  34. package/dist/core/agents/qa-signal.d.ts.map +1 -0
  35. package/dist/core/agents/qa-signal.js +231 -0
  36. package/dist/core/agents/session-name-display.d.ts +20 -5
  37. package/dist/core/agents/session-name-display.d.ts.map +1 -1
  38. package/dist/core/agents/session-name-display.js +67 -7
  39. package/dist/core/agents/state/heartbeat-reader.d.ts +7 -0
  40. package/dist/core/agents/state/heartbeat-reader.d.ts.map +1 -1
  41. package/dist/core/agents/state/heartbeat-writer.d.ts +11 -0
  42. package/dist/core/agents/state/heartbeat-writer.d.ts.map +1 -1
  43. package/dist/core/agents/state/heartbeat-writer.js +17 -0
  44. package/dist/core/agents/state/live-coordination-view.d.ts.map +1 -1
  45. package/dist/core/agents/state/live-coordination-view.js +1 -0
  46. package/dist/core/agents/state/live-coordination-writer.js +5 -0
  47. package/dist/core/artifacts/constants.d.ts +1 -1
  48. package/dist/core/artifacts/constants.js +1 -1
  49. package/dist/core/artifacts/index.d.ts +57 -6
  50. package/dist/core/artifacts/index.d.ts.map +1 -1
  51. package/dist/core/artifacts/index.js +265 -10
  52. package/dist/core/config.d.ts +8 -0
  53. package/dist/core/config.d.ts.map +1 -1
  54. package/dist/core/config.js +16 -0
  55. package/dist/core/diagnostics/bundle.d.ts +16 -0
  56. package/dist/core/diagnostics/bundle.d.ts.map +1 -1
  57. package/dist/core/diagnostics/bundle.js +101 -7
  58. package/dist/core/events/v3/bootstrap.d.ts.map +1 -1
  59. package/dist/core/events/v3/bootstrap.js +10 -0
  60. package/dist/core/events/v3/coordination-view.d.ts +3 -0
  61. package/dist/core/events/v3/coordination-view.d.ts.map +1 -1
  62. package/dist/core/events/v3/coordination-view.js +69 -8
  63. package/dist/core/events/v3/producers/intake.d.ts.map +1 -1
  64. package/dist/core/events/v3/producers/intake.js +9 -2
  65. package/dist/core/events/v3/producers/recorder.d.ts +23 -0
  66. package/dist/core/events/v3/producers/recorder.d.ts.map +1 -1
  67. package/dist/core/events/v3/producers/recorder.js +264 -18
  68. package/dist/core/hooks/cli.js +120 -27
  69. package/dist/core/hooks/resolve/transcript.d.ts.map +1 -1
  70. package/dist/core/hooks/resolve/transcript.js +10 -3
  71. package/dist/core/hooks/session-name-presence.d.ts.map +1 -1
  72. package/dist/core/hooks/session-name-presence.js +4 -1
  73. package/dist/core/qa-artifacts.d.ts +20 -0
  74. package/dist/core/qa-artifacts.d.ts.map +1 -0
  75. package/dist/core/qa-artifacts.js +110 -0
  76. package/dist/core/resources/contract.d.ts +6 -0
  77. package/dist/core/resources/contract.d.ts.map +1 -1
  78. package/dist/core/resources/sampler.d.ts +7 -0
  79. package/dist/core/resources/sampler.d.ts.map +1 -1
  80. package/dist/core/resources/sampler.js +106 -5
  81. package/dist/lib/admission.d.ts +71 -0
  82. package/dist/lib/admission.d.ts.map +1 -0
  83. package/dist/lib/admission.js +264 -0
  84. package/dist/lib/agent-browser/client.d.ts +1 -1
  85. package/dist/lib/agent-browser/client.d.ts.map +1 -1
  86. package/dist/lib/agent-browser/client.js +1 -5
  87. package/dist/lib/browser/capture-fidelity.d.ts +39 -0
  88. package/dist/lib/browser/capture-fidelity.d.ts.map +1 -0
  89. package/dist/lib/browser/capture-fidelity.js +84 -0
  90. package/dist/lib/browser/client.d.ts +41 -1
  91. package/dist/lib/browser/client.d.ts.map +1 -1
  92. package/dist/lib/browser/client.js +167 -9
  93. package/dist/lib/browser/critique.d.ts +38 -1
  94. package/dist/lib/browser/critique.d.ts.map +1 -1
  95. package/dist/lib/browser/critique.js +34 -6
  96. package/dist/lib/browser/index.d.ts +4 -2
  97. package/dist/lib/browser/index.d.ts.map +1 -1
  98. package/dist/lib/browser/index.js +2 -0
  99. package/dist/lib/browser/page-review-judge.d.ts +64 -0
  100. package/dist/lib/browser/page-review-judge.d.ts.map +1 -0
  101. package/dist/lib/browser/page-review-judge.js +270 -0
  102. package/dist/lib/browser/page-review-pack.d.ts +613 -0
  103. package/dist/lib/browser/page-review-pack.d.ts.map +1 -0
  104. package/dist/lib/browser/page-review-pack.js +1751 -0
  105. package/dist/lib/browser/qa-run-contracts.d.ts +214 -10
  106. package/dist/lib/browser/qa-run-contracts.d.ts.map +1 -1
  107. package/dist/lib/browser/qa-run-contracts.js +136 -1
  108. package/dist/lib/browser/qa-run.d.ts +100 -10
  109. package/dist/lib/browser/qa-run.d.ts.map +1 -1
  110. package/dist/lib/browser/qa-run.js +768 -169
  111. package/dist/lib/browser/request-diagnostics.d.ts +13 -0
  112. package/dist/lib/browser/request-diagnostics.d.ts.map +1 -0
  113. package/dist/lib/browser/request-diagnostics.js +18 -0
  114. package/dist/lib/browser/tiling.d.ts +19 -0
  115. package/dist/lib/browser/tiling.d.ts.map +1 -1
  116. package/dist/lib/browser/tiling.js +28 -0
  117. package/dist/lib/cookies/client.d.ts +9 -0
  118. package/dist/lib/cookies/client.d.ts.map +1 -1
  119. package/dist/lib/cookies/client.js +197 -44
  120. package/dist/lib/cookies/extra.d.ts +18 -0
  121. package/dist/lib/cookies/extra.d.ts.map +1 -0
  122. package/dist/lib/cookies/extra.js +14 -0
  123. package/dist/lib/cookies/index.d.ts +2 -1
  124. package/dist/lib/cookies/index.d.ts.map +1 -1
  125. package/dist/lib/cookies/index.js +2 -1
  126. package/dist/lib/durable-job.d.ts +124 -0
  127. package/dist/lib/durable-job.d.ts.map +1 -0
  128. package/dist/lib/durable-job.js +296 -0
  129. package/dist/lib/http/client.d.ts +7 -1
  130. package/dist/lib/http/client.d.ts.map +1 -1
  131. package/dist/lib/http/client.js +2 -0
  132. package/dist/lib/instructions/templates.d.ts.map +1 -1
  133. package/dist/lib/instructions/templates.js +5 -2
  134. package/package.json +8 -2
  135. package/src/commander.ts +50 -2
  136. package/src/commands/admission.ts +699 -0
  137. package/src/commands/agents.ts +87 -1
  138. package/src/commands/artifacts.ts +97 -21
  139. package/src/commands/browse-ai.ts +10 -5
  140. package/src/commands/browse.ts +481 -20
  141. package/src/commands/fetch.ts +1 -0
  142. package/src/commands/qa-record.ts +682 -0
  143. package/src/commands/qa-run.ts +335 -16
  144. package/src/commands/qa-status.ts +608 -0
  145. package/src/commands/qa-verify.ts +238 -0
  146. package/src/commands/review-pack.ts +1281 -0
  147. package/src/core/agents/qa-signal.ts +261 -0
  148. package/src/core/agents/session-name-display.ts +78 -7
  149. package/src/core/agents/state/heartbeat-reader.ts +7 -0
  150. package/src/core/agents/state/heartbeat-writer.ts +23 -0
  151. package/src/core/agents/state/live-coordination-view.ts +1 -0
  152. package/src/core/agents/state/live-coordination-writer.ts +5 -0
  153. package/src/core/artifacts/constants.ts +1 -1
  154. package/src/core/artifacts/index.ts +370 -21
  155. package/src/core/config.ts +23 -0
  156. package/src/core/diagnostics/bundle.ts +119 -11
  157. package/src/core/events/v3/bootstrap.ts +10 -0
  158. package/src/core/events/v3/coordination-view.ts +100 -11
  159. package/src/core/events/v3/producers/intake.ts +9 -2
  160. package/src/core/events/v3/producers/recorder.ts +312 -18
  161. package/src/core/hooks/cli.ts +140 -32
  162. package/src/core/hooks/resolve/transcript.ts +10 -3
  163. package/src/core/hooks/session-name-presence.ts +6 -1
  164. package/src/core/qa-artifacts.ts +126 -0
  165. package/src/core/resources/contract.ts +7 -0
  166. package/src/core/resources/sampler.ts +138 -6
  167. package/src/lib/admission.ts +347 -0
  168. package/src/lib/agent-browser/client.ts +2 -10
  169. package/src/lib/browser/capture-fidelity.ts +98 -0
  170. package/src/lib/browser/client.ts +206 -10
  171. package/src/lib/browser/critique.ts +62 -7
  172. package/src/lib/browser/index.ts +36 -0
  173. package/src/lib/browser/page-review-judge.ts +360 -0
  174. package/src/lib/browser/page-review-pack.ts +2384 -0
  175. package/src/lib/browser/qa-run-contracts.ts +366 -3
  176. package/src/lib/browser/qa-run.ts +868 -190
  177. package/src/lib/browser/request-diagnostics.ts +27 -0
  178. package/src/lib/browser/tiling.ts +32 -0
  179. package/src/lib/cookies/client.ts +228 -42
  180. package/src/lib/cookies/extra.ts +28 -0
  181. package/src/lib/cookies/index.ts +2 -0
  182. package/src/lib/durable-job.ts +407 -0
  183. package/src/lib/http/client.ts +13 -1
  184. package/src/lib/instructions/templates.ts +5 -2
@@ -17,10 +17,18 @@
17
17
  //
18
18
  // Toolkit tier: this module must not import src/core (layering check).
19
19
 
20
+ import { createHash } from "node:crypto";
20
21
  import type { QaContext, QaManifest } from "./qa-plan.js";
21
22
 
22
23
  export const QA_RUN_JOB_SCHEMA_VERSION = 1 as const;
23
- export const QA_RUN_RESULT_SCHEMA_VERSION = 1 as const;
24
+ export const QA_RUN_RESULT_SCHEMA_VERSION = 4 as const;
25
+
26
+ /** How a result's evidence was produced. `runner` means the qa-run matrix
27
+ * executed the checks itself. `manual` means an operator or agent performed
28
+ * the checks by hand and recorded them (qa-record); such a result can report
29
+ * a defect but can never claim a pass, because nothing re-executable proved
30
+ * the absence of one. */
31
+ export type QaRunEvidenceSource = "runner" | "manual";
24
32
 
25
33
  /** One rendering context the runner will capture and check. */
26
34
  export interface QaRunContext {
@@ -70,6 +78,26 @@ export interface QaRunPolicy {
70
78
  command_concurrency?: number;
71
79
  /** Per-command timeout in milliseconds (default 120000). */
72
80
  command_timeout_ms?: number;
81
+ /** Overall runner deadline in milliseconds (default 900000). When the run
82
+ * exceeds it, remaining commands are skipped, a `deadline` blocker is
83
+ * recorded, and the result finalizes as incomplete — releasing the
84
+ * admission slot instead of holding it open-endedly. */
85
+ run_deadline_ms?: number;
86
+ /** Full-page critique bands per context (browse `--check-critique-max-tiles`,
87
+ * default 24). Raise it when a tall page must be reviewed end to end and the
88
+ * per-tile cost is accepted; each critique row's `coverage` records what the
89
+ * run actually saw, which is why this knob can stay out of the job digest. */
90
+ critique_max_tiles?: number;
91
+ /** Vision calls in flight during the judge stage, across every context of
92
+ * the run (1 to 16). Default: the host provider's own concurrency. The
93
+ * capture stage closes every browser before judging starts, so this knob
94
+ * costs model-call parallelism, never browser memory. */
95
+ critique_pool?: number;
96
+ /** Minutes the run's page review pack lives after the judge finishes before
97
+ * the whole pack directory is deleted (default 90; 1 to 43200). The result
98
+ * document keeps every finding inline, so a run stays reportable after its
99
+ * pack is gone. */
100
+ review_pack_retention_minutes?: number;
73
101
  }
74
102
 
75
103
  export interface QaRunJob {
@@ -106,6 +134,28 @@ export interface QaRunCommandOutcome {
106
134
  wall_time_ms: number;
107
135
  }
108
136
 
137
+ /** Vision-call latency one critique backend reported over the judge pool,
138
+ * in milliseconds over the tile calls it served. The percentiles are the
139
+ * provider's own sample across every context (one pool, one sample). */
140
+ export interface QaRunCritiqueLatency {
141
+ count: number;
142
+ p50: number;
143
+ p95: number;
144
+ }
145
+
146
+ /** What share of the page the critique tiles covered, lifted from the browse
147
+ * envelope. `capped` means the tiler dropped bands past its per-context
148
+ * maximum; in signoff mode that is a blocker, in review mode a flag. Across
149
+ * several scope commands the heights take the worst case and the band counts
150
+ * add. */
151
+ export interface QaRunCritiqueCoverage {
152
+ page_height_px: number;
153
+ reviewed_height_px: number;
154
+ bands_total: number;
155
+ bands_reviewed: number;
156
+ capped: boolean;
157
+ }
158
+
109
159
  export interface QaRunCritiqueOutcome {
110
160
  context_id: string;
111
161
  provider: string;
@@ -113,19 +163,152 @@ export interface QaRunCritiqueOutcome {
113
163
  tiles_reviewed: number;
114
164
  tiles_reused: number;
115
165
  outcome: "passed" | "failed" | "unknown";
116
- findings: Array<{ severity: string; summary: string; selector?: string }>;
166
+ findings: Array<{ severity: string; summary: string; selector?: string; tile?: string }>;
167
+ /** Tile coverage of the page. Absent when the capture died or reported none. */
168
+ coverage?: QaRunCritiqueCoverage;
169
+ }
170
+
171
+ /** The judge stage as one unit: every context's tiles through one bounded
172
+ * pool of vision calls, with no browser open. `latency_ms` is the backends'
173
+ * own per-call sample over the whole pool, keyed by backend name. */
174
+ export interface QaRunCritiquePool {
175
+ concurrency: number;
176
+ tiles_total: number;
177
+ tiles_reviewed: number;
178
+ tiles_reused: number;
179
+ wall_time_ms: number;
180
+ provider: string;
181
+ latency_ms?: Record<string, QaRunCritiqueLatency>;
182
+ }
183
+
184
+ /** Where the run's page review pack lives: the on-disk evidence an agent can
185
+ * review without a browser (tiles, DOM, coverage, `review.md`, and the
186
+ * delegated-review `findings.json`). */
187
+ export interface QaRunReviewPack {
188
+ schema: string;
189
+ dir: string;
190
+ review: string;
191
+ findings: string;
192
+ /** When the pack directory is deleted (ISO). Absent when the run never
193
+ * reached the point of knowing (capture failed before any context landed). */
194
+ expires_at?: string;
195
+ /** Bytes on disk at finalize time. */
196
+ size_bytes?: number;
117
197
  }
118
198
 
119
199
  export interface QaRunBlocker {
120
- stage: "validate" | "plan" | "gates" | "interactions" | "critique" | "snapshot" | "result";
200
+ stage:
201
+ | "validate"
202
+ | "admission"
203
+ | "plan"
204
+ | "gates"
205
+ | "interactions"
206
+ | "capture"
207
+ | "critique"
208
+ | "snapshot"
209
+ | "deadline"
210
+ | "result";
121
211
  context_id?: string;
122
212
  reason: string;
123
213
  }
124
214
 
125
215
  export type QaRunVerdict = "passed" | "failed" | "incomplete";
126
216
 
217
+ /** Runner stages in execution order. `last_completed_stage` names the last
218
+ * one that finished without contributing a blocker. */
219
+ export const QA_RUN_STAGES = [
220
+ "plan",
221
+ "gates",
222
+ "interactions",
223
+ "capture",
224
+ "critique",
225
+ "snapshot",
226
+ ] as const;
227
+ export type QaRunStage = (typeof QA_RUN_STAGES)[number];
228
+
229
+ /** Identity of one runner invocation. This block is what makes a result
230
+ * verifiable evidence rather than a loose file: a consumer matches it against
231
+ * the invocation being reported (see assessQaRunEvidence) instead of trusting
232
+ * whatever sits in a reused directory. */
233
+ export interface QaRunIdentity {
234
+ /** Minted per invocation (crypto.randomUUID). */
235
+ run_id: string;
236
+ /** ISO-8601 UTC bounds of the invocation. */
237
+ started_at: string;
238
+ completed_at: string;
239
+ /** Git SHA or content identifier of what was tested, when resolvable. */
240
+ tested_revision?: string;
241
+ /** Where tested_revision came from: the job document, a git probe of the
242
+ * working directory, or nowhere (`unknown`, tested_revision absent). */
243
+ revision_source: "job" | "git" | "unknown";
244
+ /** `git status --porcelain` was non-empty when the run started — a revision
245
+ * alone does not prove content. Absent when no git probe ran. */
246
+ worktree_dirty?: boolean;
247
+ /** SHA-256 over the effective validated job (computeJobDigest). */
248
+ job_digest: string;
249
+ /** Absolute run directory the result was written into. A result found
250
+ * elsewhere has been moved or copied and fails evidence assessment. */
251
+ out_dir: string;
252
+ }
253
+
254
+ export const QA_RUN_STATUS_SCHEMA_VERSION = 1 as const;
255
+
256
+ export type QaRunStatusState = "launching" | "queued" | "running" | "completed";
257
+
258
+ /** Live status document (`run-status.json`) beside the result in every run
259
+ * directory. Written at start, every stage boundary, and on a heartbeat
260
+ * timer, so a client that lost its terminal (e.g. a Windows-to-WSL bridge
261
+ * disconnect) can mechanically distinguish a running job from a dead one:
262
+ * non-terminal state + dead PID + no result document = dead. The result
263
+ * document stays authoritative once the run completes. */
264
+ export interface QaRunStatusDocument {
265
+ schema_version: typeof QA_RUN_STATUS_SCHEMA_VERSION;
266
+ run_id: string;
267
+ /** PID of the process executing the matrix (the detach parent records the
268
+ * child's PID in its initial `launching` write; the child overwrites). */
269
+ pid: number;
270
+ state: QaRunStatusState;
271
+ /** Stage currently executing; null before plan and after completion. */
272
+ stage: QaRunStage | null;
273
+ started_at: string;
274
+ /** Heartbeat. Stage boundaries and a periodic timer both refresh it. */
275
+ updated_at: string;
276
+ /** Present only while state is "queued". */
277
+ queue?: { resource: string; waiting_since: string };
278
+ /** Present only once state is "completed". */
279
+ verdict?: QaRunVerdict;
280
+ }
281
+
282
+ /** Host-pressure sample. Captured at start, at every stage boundary, and at
283
+ * finish, so an incomplete run carries the load context that produced it and
284
+ * names the other heavy jobs it was competing with. */
285
+ export interface QaRunHostSample {
286
+ captured_at: string;
287
+ loadavg_1m: number;
288
+ free_mem_bytes: number;
289
+ total_mem_bytes: number;
290
+ cpu_count: number;
291
+ /** Other holders of the admission resource at sample time. Present only
292
+ * when the run queued; an empty array means the run had the host to
293
+ * itself. This is what turns "it was slow" into "it was slow because these
294
+ * three jobs held slots". */
295
+ competing?: Array<{ label: string; pid: number }>;
296
+ }
297
+
127
298
  export interface QaRunResult {
128
299
  schema_version: typeof QA_RUN_RESULT_SCHEMA_VERSION;
300
+ /** Runner-executed or hand-recorded. A manual result never reads passed. */
301
+ evidence_source: QaRunEvidenceSource;
302
+ run: QaRunIdentity;
303
+ host: {
304
+ start: QaRunHostSample;
305
+ finish: QaRunHostSample;
306
+ /** Sample taken as each stage began, so a stall is attributable to the
307
+ * stage that was running and the load at that moment. */
308
+ stages?: Partial<Record<QaRunStage, QaRunHostSample>>;
309
+ };
310
+ /** Null when the run never completed a stage cleanly (e.g. plan failed). */
311
+ last_completed_stage: QaRunStage | null;
129
312
  target: string;
130
313
  tested_revision?: string;
131
314
  mode: "signoff" | "review";
@@ -135,14 +318,27 @@ export interface QaRunResult {
135
318
  contexts: QaRunContext[];
136
319
  commands: QaRunCommandOutcome[];
137
320
  critique: QaRunCritiqueOutcome[];
321
+ /** Present once the judge stage ran (even when it skipped for lack of a
322
+ * provider); absent when the run stopped before it. */
323
+ critique_pool?: QaRunCritiquePool;
324
+ /** Present once the capture stage wrote at least one context. */
325
+ review_pack?: QaRunReviewPack;
138
326
  snapshot: { saved: boolean; path?: string };
139
327
  wall_time_ms: {
140
328
  plan: number;
141
329
  gates: number;
142
330
  interactions: number;
331
+ /** Browser time: rendering every context into the review pack. */
332
+ capture: number;
333
+ /** Judge time: vision calls over the pack, no browser open. */
143
334
  critique: number;
144
335
  snapshot: number;
336
+ /** Runner stages only — admission queue wait is deliberately excluded
337
+ * (see `queue`) so total stays pure runner time. */
145
338
  total: number;
339
+ /** Milliseconds spent waiting for a machine-wide admission slot before
340
+ * any browser work. Absent when the run did not queue. */
341
+ queue?: number;
146
342
  };
147
343
  blockers: QaRunBlocker[];
148
344
  verdict: QaRunVerdict;
@@ -306,6 +502,30 @@ export function validateQaRunJob(value: unknown): QaRunJobValidation {
306
502
  errors.push("policy.command_timeout_ms must be an integer ≥ 1000");
307
503
  }
308
504
  }
505
+ if (p?.run_deadline_ms !== undefined) {
506
+ const n = p.run_deadline_ms;
507
+ if (typeof n !== "number" || !Number.isInteger(n) || n < 10_000) {
508
+ errors.push("policy.run_deadline_ms must be an integer ≥ 10000");
509
+ }
510
+ }
511
+ if (p?.critique_max_tiles !== undefined) {
512
+ const n = p.critique_max_tiles;
513
+ if (typeof n !== "number" || !Number.isInteger(n) || n < 1 || n > 200) {
514
+ errors.push("policy.critique_max_tiles must be an integer between 1 and 200");
515
+ }
516
+ }
517
+ if (p?.critique_pool !== undefined) {
518
+ const n = p.critique_pool;
519
+ if (typeof n !== "number" || !Number.isInteger(n) || n < 1 || n > 16) {
520
+ errors.push("policy.critique_pool must be an integer between 1 and 16");
521
+ }
522
+ }
523
+ if (p?.review_pack_retention_minutes !== undefined) {
524
+ const n = p.review_pack_retention_minutes;
525
+ if (typeof n !== "number" || !Number.isInteger(n) || n < 1 || n > 43_200) {
526
+ errors.push("policy.review_pack_retention_minutes must be an integer between 1 and 43200");
527
+ }
528
+ }
309
529
  }
310
530
 
311
531
  scanForSecrets(job, "job", errors);
@@ -360,17 +580,160 @@ export function mergeCoverage(manifest: QaManifest, job: QaRunJob): QaRunContext
360
580
  * - signoff mode additionally requires the snapshot to have been saved.
361
581
  * `passed` is only reachable when every input proves out.
362
582
  */
583
+ // ---------------------------------------------------------------------------
584
+ // Run identity: job digest + evidence assessment
585
+ // ---------------------------------------------------------------------------
586
+
587
+ /** Recursively key-sort plain objects so the digest is stable under key
588
+ * order. Arrays keep their order — it is semantically meaningful (contexts,
589
+ * argv arrays). */
590
+ function canonicalize(value: unknown): unknown {
591
+ if (Array.isArray(value)) return value.map(canonicalize);
592
+ if (value && typeof value === "object") {
593
+ const record = value as Record<string, unknown>;
594
+ const sorted: Record<string, unknown> = {};
595
+ for (const key of Object.keys(record).sort()) sorted[key] = canonicalize(record[key]);
596
+ return sorted;
597
+ }
598
+ return value;
599
+ }
600
+
601
+ /** SHA-256 hex digest over the effective validated job (after the CLI merges
602
+ * its authoritative `target` and `mode`). `policy` is excluded: concurrency,
603
+ * timeout, and metered-critique knobs change how the run executes, not what
604
+ * it proves, and CLI flags mutate them after the job file is read — including
605
+ * them would make the same job file verify differently across invocations. */
606
+ export function computeJobDigest(job: QaRunJob): string {
607
+ const { policy: _policy, ...identityBearing } = job;
608
+ return createHash("sha256")
609
+ .update(JSON.stringify(canonicalize(identityBearing)))
610
+ .digest("hex");
611
+ }
612
+
613
+ export interface QaEvidenceExpectations {
614
+ /** Exact run ID the caller expects (from the invocation it just made). */
615
+ run_id?: string;
616
+ /** Revision the evidence must have tested. A result whose revision_source
617
+ * is `unknown` cannot satisfy a revision expectation — fail-closed. */
618
+ tested_revision?: string;
619
+ /** Digest of the effective job the evidence must have run. */
620
+ job_digest?: string;
621
+ /** ISO-8601 floor: a run started before this instant is stale. */
622
+ not_started_before?: string;
623
+ /** Maximum age of completed_at in milliseconds, evaluated against `now`. */
624
+ max_age_ms?: number;
625
+ /** Evaluation instant for max_age_ms (ISO-8601; default: current time). */
626
+ now?: string;
627
+ /** Directory the result file was read from. Compared against the recorded
628
+ * run.out_dir: a moved or copied result is not evidence for its new home. */
629
+ found_in_dir?: string;
630
+ }
631
+
632
+ export interface QaEvidenceAssessment {
633
+ fresh: boolean;
634
+ /** Empty when fresh; each entry names one independent staleness reason. */
635
+ reasons: string[];
636
+ run?: QaRunIdentity;
637
+ verdict?: QaRunVerdict;
638
+ }
639
+
640
+ /**
641
+ * Assess whether a result document is fresh evidence for the invocation the
642
+ * caller has in mind. Fail-closed: a document without a verifiable identity
643
+ * block (schema v1 or foreign JSON) is stale by definition, and every
644
+ * expectation mismatch is reported, not just the first.
645
+ */
646
+ export function assessQaRunEvidence(
647
+ document: unknown,
648
+ expectations: QaEvidenceExpectations = {},
649
+ ): QaEvidenceAssessment {
650
+ const reasons: string[] = [];
651
+ if (!document || typeof document !== "object" || Array.isArray(document)) {
652
+ return { fresh: false, reasons: ["document is not a result object"] };
653
+ }
654
+ const result = document as Partial<QaRunResult> & Record<string, unknown>;
655
+ if (result.schema_version !== QA_RUN_RESULT_SCHEMA_VERSION) {
656
+ return {
657
+ fresh: false,
658
+ reasons: [
659
+ `schema_version is ${JSON.stringify(result.schema_version)}, not ` +
660
+ `${QA_RUN_RESULT_SCHEMA_VERSION} — pre-identity results carry nothing to verify`,
661
+ ],
662
+ };
663
+ }
664
+ const run = result.run;
665
+ if (
666
+ !run ||
667
+ typeof run.run_id !== "string" ||
668
+ typeof run.started_at !== "string" ||
669
+ typeof run.completed_at !== "string" ||
670
+ typeof run.job_digest !== "string"
671
+ ) {
672
+ return { fresh: false, reasons: ["result carries no complete run-identity block"] };
673
+ }
674
+ if (expectations.run_id !== undefined && run.run_id !== expectations.run_id) {
675
+ reasons.push(`run_id ${run.run_id} is not the expected ${expectations.run_id}`);
676
+ }
677
+ if (expectations.job_digest !== undefined && run.job_digest !== expectations.job_digest) {
678
+ reasons.push("job_digest does not match the expected job definition");
679
+ }
680
+ if (expectations.tested_revision !== undefined) {
681
+ if (run.revision_source === "unknown" || run.tested_revision === undefined) {
682
+ reasons.push(
683
+ `a revision expectation was given but the run recorded revision_source "unknown"`,
684
+ );
685
+ } else if (run.tested_revision !== expectations.tested_revision) {
686
+ reasons.push(
687
+ `tested_revision ${run.tested_revision} is not the expected ${expectations.tested_revision}`,
688
+ );
689
+ }
690
+ }
691
+ if (expectations.not_started_before !== undefined) {
692
+ if (Date.parse(run.started_at) < Date.parse(expectations.not_started_before)) {
693
+ reasons.push(
694
+ `run started ${run.started_at}, before the freshness floor ${expectations.not_started_before}`,
695
+ );
696
+ }
697
+ }
698
+ if (expectations.max_age_ms !== undefined) {
699
+ const now = expectations.now !== undefined ? Date.parse(expectations.now) : Date.now();
700
+ const age = now - Date.parse(run.completed_at);
701
+ if (Number.isNaN(age) || age > expectations.max_age_ms) {
702
+ reasons.push(
703
+ `run completed ${run.completed_at}, older than the ${expectations.max_age_ms}ms maximum age`,
704
+ );
705
+ }
706
+ }
707
+ if (expectations.found_in_dir !== undefined && run.out_dir !== expectations.found_in_dir) {
708
+ reasons.push(
709
+ `result was found in ${expectations.found_in_dir} but records out_dir ${run.out_dir} — ` +
710
+ "a moved or copied result is not evidence for its new location",
711
+ );
712
+ }
713
+ return {
714
+ fresh: reasons.length === 0,
715
+ reasons,
716
+ run: run as QaRunIdentity,
717
+ ...(result.verdict !== undefined ? { verdict: result.verdict } : {}),
718
+ };
719
+ }
720
+
363
721
  export function computeVerdict(input: {
364
722
  mode: QaRunJob["mode"];
365
723
  blockers: QaRunBlocker[];
366
724
  commands: QaRunCommandOutcome[];
367
725
  critique: QaRunCritiqueOutcome[];
368
726
  snapshotSaved: boolean;
727
+ /** Defaults to `runner`. `manual` caps the verdict at incomplete. */
728
+ evidenceSource?: QaRunEvidenceSource;
369
729
  }): QaRunVerdict {
370
730
  const failed =
371
731
  input.commands.some((command) => command.outcome === "failed") ||
372
732
  input.critique.some((entry) => entry.outcome === "failed");
373
733
  if (failed) return "failed";
734
+ // Hand-recorded evidence can prove a defect but never its absence: nothing
735
+ // re-executable ran, so a manual result stops at incomplete by contract.
736
+ if (input.evidenceSource === "manual") return "incomplete";
374
737
  const unknown =
375
738
  input.commands.some((command) => command.outcome === "unknown") ||
376
739
  input.critique.some((entry) => entry.outcome === "unknown");