tickmarkr 2.5.8 → 2.5.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/adapters/qwen.js +30 -3
  2. package/dist/cli/commands/approve.js +15 -3
  3. package/dist/cli/commands/beat.d.ts +2 -0
  4. package/dist/cli/commands/beat.js +28 -29
  5. package/dist/cli/commands/compile.js +11 -0
  6. package/dist/cli/commands/fleet.js +61 -53
  7. package/dist/cli/commands/verify.d.ts +4 -1
  8. package/dist/cli/commands/verify.js +16 -5
  9. package/dist/cli/help.d.ts +2 -0
  10. package/dist/cli/help.js +6 -4
  11. package/dist/config/config.d.ts +41 -3
  12. package/dist/config/config.js +48 -17
  13. package/dist/config/fleet-overlay.js +47 -62
  14. package/dist/gates/baseline.d.ts +65 -0
  15. package/dist/gates/baseline.js +163 -9
  16. package/dist/gates/review.d.ts +6 -4
  17. package/dist/gates/review.js +62 -21
  18. package/dist/gates/run-gates.d.ts +4 -1
  19. package/dist/gates/run-gates.js +21 -11
  20. package/dist/gates/test-manifest.d.ts +11 -1
  21. package/dist/gates/test-manifest.js +41 -21
  22. package/dist/run/daemon.js +84 -41
  23. package/dist/run/journal.d.ts +17 -2
  24. package/dist/run/journal.js +99 -11
  25. package/dist/run/merge.d.ts +13 -2
  26. package/dist/run/merge.js +74 -12
  27. package/dist/run/protocol.d.ts +82 -0
  28. package/dist/run/protocol.js +35 -0
  29. package/dist/run/receipt-resolver.d.ts +18 -0
  30. package/dist/run/receipt-resolver.js +132 -0
  31. package/dist/run/supervision.d.ts +14 -1
  32. package/dist/run/supervision.js +122 -24
  33. package/dist/tui/cockpit/evidence-view.d.ts +10 -1
  34. package/dist/tui/cockpit/evidence-view.js +37 -5
  35. package/dist/tui/ink/fleet-app.d.ts +12 -22
  36. package/dist/tui/ink/fleet-app.js +520 -131
  37. package/package.json +1 -1
  38. package/skills/tickmarkr-overseer/SKILL.md +61 -36
  39. package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
6
6
  import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
- import { structuredFindings } from "../run/journal.js";
9
+ import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { marginalCostRank } from "../route/router.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
13
  import { resolveStateDir } from "./cache.js";
14
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
15
15
  import { classifyVerdictCause } from "./verdict-cause.js";
16
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
17
17
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -188,15 +188,17 @@ export function matchClosureId(candidate, target) {
188
188
  // OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
189
189
  // it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
190
190
  // `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
191
- const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
192
191
  if (typeof target === "string") {
193
- return normCandidate === target.replace(/\s+/g, "");
192
+ return reviewFingerprintMatches(candidate, target);
194
193
  }
195
- for (const fp of target) {
196
- if (typeof fp === "string" && normCandidate === fp.replace(/\s+/g, ""))
197
- return fp;
198
- }
199
- return undefined;
194
+ const matches = [...target].filter((finding) => {
195
+ const ids = typeof finding === "string" ? [finding] : observedReviewFingerprints(finding);
196
+ return ids.some((id) => reviewFingerprintMatches(candidate, id));
197
+ });
198
+ if (matches.length !== 1)
199
+ return undefined;
200
+ const match = matches[0];
201
+ return typeof match === "string" ? match : match.fingerprint;
200
202
  }
201
203
  /**
202
204
  * Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
@@ -205,10 +207,10 @@ export function matchClosureId(candidate, target) {
205
207
  export function isReviewClosureInvalid(v, priorIds) {
206
208
  const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
207
209
  const closureLists = [v?.resolved, v?.reraised];
208
- const allCandidateIds = [...(v?.resolved ?? []), ...(v?.reraised ?? [])];
210
+ const allCandidateIds = closureLists.flatMap((list) => Array.isArray(list) ? list : []);
209
211
  return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
210
212
  || new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
211
- || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
213
+ || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, priors) === (typeof id === "string" ? id : id.fingerprint))));
212
214
  }
213
215
  /**
214
216
  * OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
@@ -368,10 +370,17 @@ function daemonRepoRoot(worktree, artifactDir) {
368
370
  * closure on a typo and that read as malformed — the block is what a closure list is copied from.
369
371
  */
370
372
  export function renderPriorMaterials(priorMaterials) {
373
+ const fingerprints = priorMaterials.map((finding, i) => {
374
+ const observed = observedReviewFingerprints(finding);
375
+ const heading = observed.length > 1 ? `Finding ${i + 1} (choose one observed spelling):\n` : "";
376
+ return heading + observed.map((id) => `Fingerprint: ${id}`).join("\n");
377
+ }).join("\n");
371
378
  return `## Prior materials this attempt must close
372
- Copy each fingerprint below EXACTLY (they appear once, in this block) into resolved or reraised:
379
+ For each finding below, copy exactly ONE of its observed fingerprints into resolved or reraised.
380
+ Use only these observed spellings. For every findings entry that restates a reraised prior, whether at the same path or a new path, set its "reraised" field to the copied id; unrelated defects need separate entries.
381
+ The fingerprints appear once, in this block:
373
382
  \`\`\`text
374
- ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}`).join("\n")}
383
+ ${fingerprints}
375
384
  \`\`\`
376
385
  ${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
377
386
  }
@@ -383,7 +392,7 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
383
392
  // never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
384
393
  priorReviewers = [],
385
394
  // OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
386
- carriedAuthors = []) {
395
+ carriedAuthors = [], operatorContext) {
387
396
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
388
397
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
389
398
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -502,6 +511,10 @@ ${suiteBudget} Never run the whole suite (including an unfiltered npm test or vi
502
511
 
503
512
  ${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
504
513
 
514
+ ` : ""}${operatorContext?.trim() ? `## Operator context
515
+ Context only: this never substitutes for an acceptance criterion or closes a prior material.
516
+ ${operatorContext.trim()}
517
+
505
518
  ` : ""}## Diff
506
519
  \`\`\`diff
507
520
  ${diff}
@@ -578,7 +591,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
578
591
  const provider = modelProvider(reviewer.model, reviewer.vendor);
579
592
  const v = extractVerdictJson(raw, nonce);
580
593
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
581
- const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
594
+ const priorIds = priorMaterials;
582
595
  const closureInvalid = isReviewClosureInvalid(v, priorIds);
583
596
  const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
584
597
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
@@ -610,7 +623,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
610
623
  provider,
611
624
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
612
625
  cause,
613
- ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: [...priorIds] } : {}),
626
+ ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
614
627
  bytes, seatAuthoredBytes: bytes,
615
628
  ...(saved ? { rawPath: saved } : {}),
616
629
  ...(savedBrief ? { briefPath: savedBrief } : {}),
@@ -621,7 +634,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
621
634
  const decided = findings !== null
622
635
  ? classifyReviewFindings(findings)
623
636
  : classifyReviewIssues(v.approve, v.issues);
624
- const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, finding.fingerprint)));
637
+ const ownLines = [...decided.lines];
638
+ const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, [finding])));
625
639
  if (reraised.length) {
626
640
  if (decided.pass)
627
641
  decided.headline = "requested changes";
@@ -636,6 +650,36 @@ The top-level comments array is optional. Use it only for actionable line-anchor
636
650
  }
637
651
  const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
638
652
  const details = appendAnchoredReview(prose, v);
653
+ // Only the verdict's anchors may supply missing evidence, and only when unambiguous.
654
+ // Reuse the journal's path normalization without changing legacy details-only parsing.
655
+ const anchoredPaths = new Set(parseAnchoredComments(v).map((comment) => {
656
+ const anchor = structuredFindings("review", `- ${comment.path}:${comment.line} — anchor`)
657
+ .find((finding) => finding.class === "review:anchored");
658
+ return anchor?.path ?? comment.path;
659
+ }));
660
+ const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
661
+ const ownDetails = appendAnchoredReview(ownLines.join("\n"), v);
662
+ const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
663
+ .map((finding) => finding.path === UNIDENTIFIED && anchoredPath
664
+ ? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
665
+ : finding);
666
+ // A top-level reraised id identifies a prior, not an arbitrary new defect. Bind a restatement
667
+ // only when its own entry echoes that validated id (or its note explicitly contains the id).
668
+ const linkedRows = currentRows.map((finding) => {
669
+ if (finding.class !== "review:material")
670
+ return finding;
671
+ const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
672
+ const mentioned = reraised.filter((prior) => observedReviewFingerprints(prior)
673
+ .some((fp) => finding.note.includes(`\`${fp}\``) || finding.note.trim() === fp));
674
+ const id = matchClosureId(entry?.reraised, reraised)
675
+ ?? (mentioned.length === 1 ? mentioned[0].fingerprint : undefined);
676
+ return id ? { ...finding, reraisedFrom: id } : finding;
677
+ });
678
+ // Several rows claiming the same prior are ambiguous; none gets to erase the others.
679
+ const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
680
+ && linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
681
+ ? { ...finding, reraisedFrom: undefined } : finding);
682
+ const carriedRows = carryReviewFindings(reraised, unambiguousRows);
639
683
  return {
640
684
  gate: "review",
641
685
  pass: decided.pass,
@@ -649,10 +693,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
649
693
  resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
650
694
  reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
651
695
  } : {}),
652
- ...(reraised.length ? { findings: [
653
- ...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
654
- ...reraised,
655
- ] } : {}),
696
+ ...(!decided.pass ? { findings: carriedRows } : {}),
656
697
  ...(saved ? { rawPath: saved } : {}),
657
698
  ...(savedBrief ? { briefPath: savedBrief } : {}),
658
699
  },
@@ -2,7 +2,7 @@ import type { CommandReceiptAttribution } from "../run/protocol.js";
2
2
  import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
3
3
  import { type TickmarkrConfig } from "../config/config.js";
4
4
  import { type GateName, type Task } from "../graph/schema.js";
5
- import { type Baseline } from "./baseline.js";
5
+ import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
6
6
  import { type GateVia } from "./llm.js";
7
7
  import { type PriorReviewer } from "./review.js";
8
8
  import type { GateResult } from "./types.js";
@@ -46,6 +46,7 @@ export type GateEvent = {
46
46
  result?: GateResult;
47
47
  };
48
48
  export interface GateContext {
49
+ evidence?: GateEvidenceOptions;
49
50
  buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
50
51
  authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
51
52
  verificationScope?: VerificationScope;
@@ -61,6 +62,8 @@ export interface GateContext {
61
62
  cfg: TickmarkrConfig;
62
63
  via?: GateVia;
63
64
  carriedFindings?: readonly StructuredFinding[];
65
+ /** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
66
+ operatorContext?: string;
64
67
  excludeReviewers?: string[];
65
68
  demotedReviewers?: Set<string>;
66
69
  reviewNoVerdicts?: Map<string, string[]>;
@@ -222,22 +222,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
222
222
  longestFile: entry?.longestFile,
223
223
  overallCeilingMs: effectiveCeilingMs(entry),
224
224
  artifactDir,
225
+ evidence: retry.evidence,
225
226
  });
226
227
  const reportPath = outcome.reportPath;
228
+ const evidence = { evidenceReceipt: outcome.evidenceReceipt, evidenceReceipts: outcome.evidenceReceipts };
227
229
  if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
228
230
  && outcome.meta?.retryable !== false) {
229
231
  const waitedMs = await waitForCalmWindow(executionSignal());
230
232
  if (!calmWindowReady())
231
- return { gate: "test", pass: false, details: outcome.details,
233
+ return { ...evidence, gate: "test", pass: false, details: outcome.details,
232
234
  meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
233
235
  if (!retry.authorizeRetry("infra")) {
234
- return { gate: "test", pass: false, details: outcome.details,
236
+ return { ...evidence, gate: "test", pass: false, details: outcome.details,
235
237
  meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
236
238
  }
237
239
  const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
238
- return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
240
+ return { ...result, evidenceReceipts: [...(outcome.evidenceReceipts ?? []), ...(result.evidenceReceipts ?? [])], meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
239
241
  }
240
242
  return {
243
+ ...evidence,
241
244
  gate: "test",
242
245
  pass: outcome.pass,
243
246
  details: outcome.details,
@@ -259,6 +262,12 @@ function classifySignalOnlyTest(g) {
259
262
  }
260
263
  export async function runGates(task, ctx) {
261
264
  const results = [];
265
+ const evidence = {
266
+ artifactDir: ctx.artifactDir,
267
+ runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
268
+ taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
269
+ ...ctx.evidence,
270
+ };
262
271
  // Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
263
272
  // allocates a new invocation, including retries whose local spawn counter starts at one again.
264
273
  let currentBuild;
@@ -658,8 +667,8 @@ export async function runGates(task, ctx) {
658
667
  // other scripted test command keeps today's exit-code contract byte-identically.
659
668
  const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
660
669
  r = useManifest
661
- ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
662
- : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
670
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence }))
671
+ : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), evidence, ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
663
672
  }
664
673
  finally {
665
674
  await receiptNotes;
@@ -676,7 +685,7 @@ export async function runGates(task, ctx) {
676
685
  if (!cached && r.pass && commands[g]) {
677
686
  const dirt = await dirtyWorktree();
678
687
  if (dirt) {
679
- await record(await dirtyRefusal(g, dirt, commands[g]));
688
+ await record({ ...await dirtyRefusal(g, dirt, commands[g]), evidenceReceipt: r.evidenceReceipt, evidenceReceipts: r.evidenceReceipts });
680
689
  return;
681
690
  }
682
691
  }
@@ -903,7 +912,7 @@ export async function runGates(task, ctx) {
903
912
  const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
904
913
  const carriedAuthors = ctx.carriedAuthors ?? [];
905
914
  let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
906
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
915
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
907
916
  // OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
908
917
  // a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
909
918
  // verdict never enters results; an exhausted pool preserves its cause.
@@ -936,7 +945,7 @@ export async function runGates(task, ctx) {
936
945
  const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
937
946
  const exclusion = crossAdapter ? "adapter" : "channel";
938
947
  exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
939
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
948
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
940
949
  if (second.meta?.noEligibleReviewer !== true) {
941
950
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
942
951
  const route = exclusion === "adapter"
@@ -1104,8 +1113,8 @@ export async function runGates(task, ctx) {
1104
1113
  if (!full) {
1105
1114
  const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
1106
1115
  full = fullUsesManifest
1107
- ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
1108
- : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
1116
+ ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, { ...retryOptions(identity), evidence }))
1117
+ : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], { ...retryOptions(identity), evidence })))[0];
1109
1118
  }
1110
1119
  fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
1111
1120
  const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
@@ -1115,8 +1124,9 @@ export async function runGates(task, ctx) {
1115
1124
  verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
1116
1125
  }
1117
1126
  const merged = withTelemetry(dirt
1118
- ? await dirtyRefusal("test", dirt, ctx.commands.test)
1127
+ ? { ...await dirtyRefusal("test", dirt, ctx.commands.test), evidenceReceipt: full.evidenceReceipt, evidenceReceipts: full.evidenceReceipts }
1119
1128
  : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
1129
+ merged.evidenceReceipts = [...(heldTest?.evidenceReceipts ?? []), ...(full?.evidenceReceipts ?? [])];
1120
1130
  results[results.findIndex((r) => r.gate === "test")] = merged;
1121
1131
  heldTest = undefined;
1122
1132
  await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
@@ -1,4 +1,5 @@
1
- import type { BaselineFileDuration } from "./baseline.js";
1
+ import { type GateEvidenceOptions, type BaselineFileDuration } from "./baseline.js";
2
+ import type { GateEvidenceReceipt } from "../run/protocol.js";
2
3
  export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
3
4
  /** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
4
5
  * --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
@@ -78,6 +79,8 @@ export declare const FILE_HANG_SLACK = 3;
78
79
  export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
79
80
  export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
80
81
  export interface ManifestRunResult {
82
+ evidenceReceipt: GateEvidenceReceipt;
83
+ evidenceReceipts: GateEvidenceReceipt[];
81
84
  exitCode: number | undefined;
82
85
  stdout: string;
83
86
  stderr: string;
@@ -91,6 +94,7 @@ export interface ManifestRunResult {
91
94
  /** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
92
95
  * timeout kills the detached process group, including descendants holding the output pipes. */
93
96
  export declare function runManifestedTest(cmd: string, cwd: string, opts: {
97
+ evidence?: GateEvidenceOptions;
94
98
  manifest: readonly string[];
95
99
  nonce: string;
96
100
  reportPath: string;
@@ -101,6 +105,8 @@ export declare function runManifestedTest(cmd: string, cwd: string, opts: {
101
105
  overallCeilingMs?: number;
102
106
  }): Promise<ManifestRunResult>;
103
107
  export interface ManifestGateOutcome {
108
+ evidenceReceipt?: GateEvidenceReceipt;
109
+ evidenceReceipts?: GateEvidenceReceipt[];
104
110
  pass: boolean;
105
111
  kind: ManifestVerdictKind;
106
112
  details: string;
@@ -115,6 +121,8 @@ export interface DiscoveredManifest {
115
121
  separator: string;
116
122
  listingExit: number | undefined;
117
123
  listingStdout: string;
124
+ evidenceReceipt: GateEvidenceReceipt;
125
+ evidenceReceipts: GateEvidenceReceipt[];
118
126
  }
119
127
  /** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
120
128
  * from its own listing and never from a stdout summary. The gate and the baseline capture share it,
@@ -125,6 +133,7 @@ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
125
133
  nonce: string;
126
134
  env: NodeJS.ProcessEnv;
127
135
  overallCeilingMs?: number;
136
+ evidence?: GateEvidenceOptions;
128
137
  }): Promise<DiscoveredManifest>;
129
138
  /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
130
139
  * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
@@ -137,4 +146,5 @@ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
137
146
  longestFile?: BaselineFileDuration | null;
138
147
  overallCeilingMs?: number;
139
148
  artifactDir?: string;
149
+ evidence?: GateEvidenceOptions;
140
150
  }): Promise<ManifestGateOutcome>;
@@ -4,6 +4,7 @@ import { tmpdir } from "node:os";
4
4
  import { isAbsolute, join, relative, sep } from "node:path";
5
5
  import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
6
6
  import { shq } from "../adapters/types.js";
7
+ import { beginGateEvidence, redactGateOutput } from "./baseline.js";
7
8
  import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
8
9
  /**
9
10
  * VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
@@ -330,6 +331,7 @@ export function runManifestedTest(cmd, cwd, opts) {
330
331
  const pollMs = opts.pollMs ?? 20;
331
332
  const overallCeilingMs = usable(opts.overallCeilingMs) ? opts.overallCeilingMs : DEFAULT_FILE_HANG_BUDGET_MS;
332
333
  const env = { ...(opts.env ?? process.env), TICKMARKR_TEST_REPORT: opts.reportPath, TICKMARKR_TEST_NONCE: opts.nonce };
334
+ const evidence = beginGateEvidence(cwd, "test", cmd, { ...opts.evidence, env }, opts.nonce);
333
335
  const controller = new AbortController();
334
336
  let pid;
335
337
  let killedFile;
@@ -355,6 +357,7 @@ export function runManifestedTest(cmd, cwd, opts) {
355
357
  }
356
358
  };
357
359
  return shell(cmd, cwd, overallCeilingMs, false, {
360
+ onReceipt: receipt => evidence.observe(killedFile && receipt.outcome === "cancelled" ? { ...receipt, outcome: "timed-out" } : receipt),
358
361
  env, signal: controller.signal, onTimeout: () => checkHang(true),
359
362
  onSpawn: (childPid) => {
360
363
  pid = childPid;
@@ -362,11 +365,21 @@ export function runManifestedTest(cmd, cwd, opts) {
362
365
  clearInterval(poll);
363
366
  poll = setInterval(checkHang, pollMs);
364
367
  },
365
- }).then((result) => ({
366
- exitCode: result.signalExit ? undefined : result.code,
367
- stdout: result.stdout, stderr: result.stderr,
368
- report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
369
- })).finally(() => { clearInterval(poll); });
368
+ }).then((result) => {
369
+ const evidenceReceipt = evidence.finish(result.stdout, result.stderr);
370
+ return {
371
+ evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt],
372
+ exitCode: result.signalExit ? undefined : result.code,
373
+ stdout: result.stdout, stderr: result.stderr,
374
+ report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
375
+ };
376
+ }).catch((error) => {
377
+ if (error instanceof Error) {
378
+ const evidenceReceipt = evidence.finish();
379
+ Object.assign(error, { evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt] });
380
+ }
381
+ throw error;
382
+ }).finally(() => { clearInterval(poll); });
370
383
  }
371
384
  /** The child environment every manifest invocation (listing and run) receives, and the lifecycle
372
385
  * protocol it records. R41: the policy is whatever THIS process was launched with (npm reads
@@ -391,11 +404,12 @@ function manifestEnvironment(cwd) {
391
404
  export async function discoverTestManifest(cmd, cwd, opts) {
392
405
  const invocation = runnerInvocation(cmd, cwd);
393
406
  const listed = await runManifestedTest(invocation.listing, cwd, {
394
- manifest: [], nonce: opts.nonce, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
407
+ evidence: opts.evidence, manifest: [], nonce: `${opts.nonce}-listing`, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
395
408
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
396
409
  });
410
+ const discoveryError = (message) => Object.assign(new Error(message), { evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts });
397
411
  if (listed.exitCode !== 0)
398
- throw new Error(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
412
+ throw discoveryError(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
399
413
  let files;
400
414
  try {
401
415
  const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
@@ -404,11 +418,11 @@ export async function discoverTestManifest(cmd, cwd, opts) {
404
418
  files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
405
419
  }
406
420
  catch {
407
- throw new Error(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
421
+ throw discoveryError(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
408
422
  }
409
423
  if (!files.length)
410
- throw new Error("vitest cannot list files: empty manifest");
411
- return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout };
424
+ throw discoveryError("vitest cannot list files: empty manifest");
425
+ return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout, evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts };
412
426
  }
413
427
  /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
414
428
  * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
@@ -430,11 +444,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
430
444
  const nonce = randomBytes(16).toString("hex");
431
445
  const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
432
446
  const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
447
+ const evidenceReceipts = [];
433
448
  let spawnedCommand = cmd;
434
449
  let manifestPath;
435
450
  const { env, verification } = manifestEnvironment(cwd);
436
451
  try {
437
- const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs });
452
+ const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
453
+ evidenceReceipts.push(...invocation.evidenceReceipts);
438
454
  const files = invocation.files;
439
455
  // R41: the EXPECTED manifest is evidence in its own right — persisted beside the report with the
440
456
  // exact discovery invocation, so a later reader can tell what this invocation was asked to prove
@@ -448,7 +464,7 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
448
464
  writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
449
465
  spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
450
466
  const invoked = await runManifestedTest(spawnedCommand, cwd, {
451
- manifest: files, nonce, reportPath, env,
467
+ evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: files, nonce, reportPath, env,
452
468
  baselineDurations: opts.baselineDurations?.map((d) => ({ ...d, file: toManifestPath(d.file, cwd) })),
453
469
  longestFile: opts.longestFile,
454
470
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
@@ -457,12 +473,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
457
473
  const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
458
474
  report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
459
475
  // Preserve the validator's verdict and classification; runner evidence only explains it.
460
- const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
461
- const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
462
- const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
463
- const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
464
- writeFileSync(stdoutPath, stdoutTail);
465
- writeFileSync(stderrPath, stderrTail);
476
+ const evidenceReceipt = invoked.evidenceReceipt;
477
+ evidenceReceipts.push(...invoked.evidenceReceipts);
478
+ const evidenceRoot = opts.evidence?.artifactDir ?? dir;
479
+ const stdoutPath = join(evidenceRoot, evidenceReceipt.stdout.path);
480
+ const stderrPath = join(evidenceRoot, evidenceReceipt.stderr.path);
481
+ const stdoutTail = Buffer.from(redactGateOutput(invoked.stdout, env)).subarray(-16 * 1024);
482
+ const stderrTail = Buffer.from(redactGateOutput(invoked.stderr, env)).subarray(-16 * 1024);
466
483
  const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
467
484
  const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
468
485
  const errors = report?.certificate?.errors;
@@ -470,11 +487,11 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
470
487
  const reportedDiagnostics = report?.certificate?.diagnostics;
471
488
  const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
472
489
  const diagnostics = !verdict.pass
473
- ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
490
+ ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}; runner vitest`
474
491
  + [...runnerErrors,
475
492
  stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
476
493
  : "";
477
- return { pass: verdict.pass, kind: verdict.kind,
494
+ return { evidenceReceipt, evidenceReceipts, pass: verdict.pass, kind: verdict.kind,
478
495
  details: verdict.details + diagnostics,
479
496
  classification: verdict.meta.classification,
480
497
  meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
@@ -482,7 +499,10 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
482
499
  exitCode: invoked.exitCode ?? -1, reportPath };
483
500
  }
484
501
  catch (error) {
485
- return { pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
502
+ const evidenceReceipt = error?.evidenceReceipt;
503
+ if (evidenceReceipt)
504
+ evidenceReceipts.push(...(error.evidenceReceipts ?? [evidenceReceipt]));
505
+ return { evidenceReceipt, evidenceReceipts, pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
486
506
  details: error instanceof Error ? error.message : String(error),
487
507
  meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
488
508
  }