tickmarkr 2.5.7 → 2.5.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/dist/adapters/qwen.js +30 -3
  2. package/dist/cli/commands/approve.js +23 -4
  3. package/dist/cli/commands/beat.d.ts +2 -0
  4. package/dist/cli/commands/beat.js +28 -29
  5. package/dist/cli/commands/compile.js +18 -0
  6. package/dist/cli/commands/fleet.js +61 -53
  7. package/dist/cli/commands/plan.d.ts +5 -0
  8. package/dist/cli/commands/plan.js +28 -23
  9. package/dist/cli/commands/resume.js +1 -1
  10. package/dist/cli/commands/run.js +1 -1
  11. package/dist/cli/commands/verify.d.ts +4 -1
  12. package/dist/cli/commands/verify.js +16 -5
  13. package/dist/cli/help.d.ts +2 -0
  14. package/dist/cli/help.js +6 -4
  15. package/dist/compile/native.js +68 -7
  16. package/dist/compile/retired-literals.d.ts +22 -0
  17. package/dist/compile/retired-literals.js +271 -0
  18. package/dist/config/config.d.ts +41 -3
  19. package/dist/config/config.js +48 -17
  20. package/dist/config/fleet-overlay.js +47 -62
  21. package/dist/drivers/index.d.ts +4 -2
  22. package/dist/drivers/index.js +54 -6
  23. package/dist/gates/baseline.d.ts +65 -0
  24. package/dist/gates/baseline.js +163 -9
  25. package/dist/gates/review.d.ts +6 -4
  26. package/dist/gates/review.js +62 -21
  27. package/dist/gates/run-gates.d.ts +6 -1
  28. package/dist/gates/run-gates.js +25 -13
  29. package/dist/gates/test-manifest.d.ts +20 -1
  30. package/dist/gates/test-manifest.js +50 -22
  31. package/dist/gates/test-reporter.js +22 -1
  32. package/dist/graph/schema.d.ts +28 -0
  33. package/dist/graph/schema.js +13 -1
  34. package/dist/run/daemon.d.ts +19 -0
  35. package/dist/run/daemon.js +363 -118
  36. package/dist/run/journal.d.ts +54 -3
  37. package/dist/run/journal.js +142 -11
  38. package/dist/run/merge.d.ts +15 -2
  39. package/dist/run/merge.js +74 -11
  40. package/dist/run/protocol.d.ts +82 -0
  41. package/dist/run/protocol.js +35 -0
  42. package/dist/run/receipt-resolver.d.ts +18 -0
  43. package/dist/run/receipt-resolver.js +132 -0
  44. package/dist/run/repair-disposition.d.ts +41 -0
  45. package/dist/run/repair-disposition.js +77 -0
  46. package/dist/run/supervision.d.ts +14 -1
  47. package/dist/run/supervision.js +122 -24
  48. package/dist/tui/cockpit/evidence-view.d.ts +10 -1
  49. package/dist/tui/cockpit/evidence-view.js +37 -5
  50. package/dist/tui/cockpit/home-view.js +45 -30
  51. package/dist/tui/cockpit/live-store.d.ts +18 -0
  52. package/dist/tui/ink/fleet-app.d.ts +12 -22
  53. package/dist/tui/ink/fleet-app.js +520 -131
  54. package/package.json +1 -1
  55. package/schema/rungraph.schema.json +54 -0
  56. package/skills/tickmarkr-overseer/SKILL.md +168 -36
  57. package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
6
6
  import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
7
7
  import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
- import { structuredFindings } from "../run/journal.js";
9
+ import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { marginalCostRank } from "../route/router.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
13
  import { resolveStateDir } from "./cache.js";
14
- import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
+ import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
15
15
  import { classifyVerdictCause } from "./verdict-cause.js";
16
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
17
17
  export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
@@ -188,15 +188,17 @@ export function matchClosureId(candidate, target) {
188
188
  // OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
189
189
  // it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
190
190
  // `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
191
- const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
192
191
  if (typeof target === "string") {
193
- return normCandidate === target.replace(/\s+/g, "");
192
+ return reviewFingerprintMatches(candidate, target);
194
193
  }
195
- for (const fp of target) {
196
- if (typeof fp === "string" && normCandidate === fp.replace(/\s+/g, ""))
197
- return fp;
198
- }
199
- return undefined;
194
+ const matches = [...target].filter((finding) => {
195
+ const ids = typeof finding === "string" ? [finding] : observedReviewFingerprints(finding);
196
+ return ids.some((id) => reviewFingerprintMatches(candidate, id));
197
+ });
198
+ if (matches.length !== 1)
199
+ return undefined;
200
+ const match = matches[0];
201
+ return typeof match === "string" ? match : match.fingerprint;
200
202
  }
201
203
  /**
202
204
  * Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
@@ -205,10 +207,10 @@ export function matchClosureId(candidate, target) {
205
207
  export function isReviewClosureInvalid(v, priorIds) {
206
208
  const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
207
209
  const closureLists = [v?.resolved, v?.reraised];
208
- const allCandidateIds = [...(v?.resolved ?? []), ...(v?.reraised ?? [])];
210
+ const allCandidateIds = closureLists.flatMap((list) => Array.isArray(list) ? list : []);
209
211
  return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
210
212
  || new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
211
- || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
213
+ || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, priors) === (typeof id === "string" ? id : id.fingerprint))));
212
214
  }
213
215
  /**
214
216
  * OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
@@ -368,10 +370,17 @@ function daemonRepoRoot(worktree, artifactDir) {
368
370
  * closure on a typo and that read as malformed — the block is what a closure list is copied from.
369
371
  */
370
372
  export function renderPriorMaterials(priorMaterials) {
373
+ const fingerprints = priorMaterials.map((finding, i) => {
374
+ const observed = observedReviewFingerprints(finding);
375
+ const heading = observed.length > 1 ? `Finding ${i + 1} (choose one observed spelling):\n` : "";
376
+ return heading + observed.map((id) => `Fingerprint: ${id}`).join("\n");
377
+ }).join("\n");
371
378
  return `## Prior materials this attempt must close
372
- Copy each fingerprint below EXACTLY (they appear once, in this block) into resolved or reraised:
379
+ For each finding below, copy exactly ONE of its observed fingerprints into resolved or reraised.
380
+ Use only these observed spellings. For every findings entry that restates a reraised prior, whether at the same path or a new path, set its "reraised" field to the copied id; unrelated defects need separate entries.
381
+ The fingerprints appear once, in this block:
373
382
  \`\`\`text
374
- ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}`).join("\n")}
383
+ ${fingerprints}
375
384
  \`\`\`
376
385
  ${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
377
386
  }
@@ -383,7 +392,7 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
383
392
  // never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
384
393
  priorReviewers = [],
385
394
  // OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
386
- carriedAuthors = []) {
395
+ carriedAuthors = [], operatorContext) {
387
396
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
388
397
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
389
398
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -502,6 +511,10 @@ ${suiteBudget} Never run the whole suite (including an unfiltered npm test or vi
502
511
 
503
512
  ${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
504
513
 
514
+ ` : ""}${operatorContext?.trim() ? `## Operator context
515
+ Context only: this never substitutes for an acceptance criterion or closes a prior material.
516
+ ${operatorContext.trim()}
517
+
505
518
  ` : ""}## Diff
506
519
  \`\`\`diff
507
520
  ${diff}
@@ -578,7 +591,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
578
591
  const provider = modelProvider(reviewer.model, reviewer.vendor);
579
592
  const v = extractVerdictJson(raw, nonce);
580
593
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
581
- const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
594
+ const priorIds = priorMaterials;
582
595
  const closureInvalid = isReviewClosureInvalid(v, priorIds);
583
596
  const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
584
597
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
@@ -610,7 +623,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
610
623
  provider,
611
624
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
612
625
  cause,
613
- ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: [...priorIds] } : {}),
626
+ ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
614
627
  bytes, seatAuthoredBytes: bytes,
615
628
  ...(saved ? { rawPath: saved } : {}),
616
629
  ...(savedBrief ? { briefPath: savedBrief } : {}),
@@ -621,7 +634,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
621
634
  const decided = findings !== null
622
635
  ? classifyReviewFindings(findings)
623
636
  : classifyReviewIssues(v.approve, v.issues);
624
- const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, finding.fingerprint)));
637
+ const ownLines = [...decided.lines];
638
+ const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, [finding])));
625
639
  if (reraised.length) {
626
640
  if (decided.pass)
627
641
  decided.headline = "requested changes";
@@ -636,6 +650,36 @@ The top-level comments array is optional. Use it only for actionable line-anchor
636
650
  }
637
651
  const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
638
652
  const details = appendAnchoredReview(prose, v);
653
+ // Only the verdict's anchors may supply missing evidence, and only when unambiguous.
654
+ // Reuse the journal's path normalization without changing legacy details-only parsing.
655
+ const anchoredPaths = new Set(parseAnchoredComments(v).map((comment) => {
656
+ const anchor = structuredFindings("review", `- ${comment.path}:${comment.line} — anchor`)
657
+ .find((finding) => finding.class === "review:anchored");
658
+ return anchor?.path ?? comment.path;
659
+ }));
660
+ const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
661
+ const ownDetails = appendAnchoredReview(ownLines.join("\n"), v);
662
+ const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
663
+ .map((finding) => finding.path === UNIDENTIFIED && anchoredPath
664
+ ? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
665
+ : finding);
666
+ // A top-level reraised id identifies a prior, not an arbitrary new defect. Bind a restatement
667
+ // only when its own entry echoes that validated id (or its note explicitly contains the id).
668
+ const linkedRows = currentRows.map((finding) => {
669
+ if (finding.class !== "review:material")
670
+ return finding;
671
+ const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
672
+ const mentioned = reraised.filter((prior) => observedReviewFingerprints(prior)
673
+ .some((fp) => finding.note.includes(`\`${fp}\``) || finding.note.trim() === fp));
674
+ const id = matchClosureId(entry?.reraised, reraised)
675
+ ?? (mentioned.length === 1 ? mentioned[0].fingerprint : undefined);
676
+ return id ? { ...finding, reraisedFrom: id } : finding;
677
+ });
678
+ // Several rows claiming the same prior are ambiguous; none gets to erase the others.
679
+ const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
680
+ && linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
681
+ ? { ...finding, reraisedFrom: undefined } : finding);
682
+ const carriedRows = carryReviewFindings(reraised, unambiguousRows);
639
683
  return {
640
684
  gate: "review",
641
685
  pass: decided.pass,
@@ -649,10 +693,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
649
693
  resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
650
694
  reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
651
695
  } : {}),
652
- ...(reraised.length ? { findings: [
653
- ...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
654
- ...reraised,
655
- ] } : {}),
696
+ ...(!decided.pass ? { findings: carriedRows } : {}),
656
697
  ...(saved ? { rawPath: saved } : {}),
657
698
  ...(savedBrief ? { briefPath: savedBrief } : {}),
658
699
  },
@@ -2,7 +2,7 @@ import type { CommandReceiptAttribution } from "../run/protocol.js";
2
2
  import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
3
3
  import { type TickmarkrConfig } from "../config/config.js";
4
4
  import { type GateName, type Task } from "../graph/schema.js";
5
- import { type Baseline } from "./baseline.js";
5
+ import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
6
6
  import { type GateVia } from "./llm.js";
7
7
  import { type PriorReviewer } from "./review.js";
8
8
  import type { GateResult } from "./types.js";
@@ -46,6 +46,7 @@ export type GateEvent = {
46
46
  result?: GateResult;
47
47
  };
48
48
  export interface GateContext {
49
+ evidence?: GateEvidenceOptions;
49
50
  buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
50
51
  authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
51
52
  verificationScope?: VerificationScope;
@@ -61,10 +62,14 @@ export interface GateContext {
61
62
  cfg: TickmarkrConfig;
62
63
  via?: GateVia;
63
64
  carriedFindings?: readonly StructuredFinding[];
65
+ /** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
66
+ operatorContext?: string;
64
67
  excludeReviewers?: string[];
65
68
  demotedReviewers?: Set<string>;
66
69
  reviewNoVerdicts?: Map<string, string[]>;
67
70
  recheck?: boolean;
71
+ /** Explicit worker funding requires fresh red measurements, never a gate waiver. */
72
+ cachedRedBypass?: "operator-rerun";
68
73
  carriedAuthors?: readonly string[];
69
74
  reviewHistory?: string[];
70
75
  priorReviewers?: PriorReviewer[];
@@ -222,22 +222,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
222
222
  longestFile: entry?.longestFile,
223
223
  overallCeilingMs: effectiveCeilingMs(entry),
224
224
  artifactDir,
225
+ evidence: retry.evidence,
225
226
  });
226
227
  const reportPath = outcome.reportPath;
228
+ const evidence = { evidenceReceipt: outcome.evidenceReceipt, evidenceReceipts: outcome.evidenceReceipts };
227
229
  if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
228
230
  && outcome.meta?.retryable !== false) {
229
231
  const waitedMs = await waitForCalmWindow(executionSignal());
230
232
  if (!calmWindowReady())
231
- return { gate: "test", pass: false, details: outcome.details,
233
+ return { ...evidence, gate: "test", pass: false, details: outcome.details,
232
234
  meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
233
235
  if (!retry.authorizeRetry("infra")) {
234
- return { gate: "test", pass: false, details: outcome.details,
236
+ return { ...evidence, gate: "test", pass: false, details: outcome.details,
235
237
  meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
236
238
  }
237
239
  const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
238
- return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
240
+ return { ...result, evidenceReceipts: [...(outcome.evidenceReceipts ?? []), ...(result.evidenceReceipts ?? [])], meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
239
241
  }
240
242
  return {
243
+ ...evidence,
241
244
  gate: "test",
242
245
  pass: outcome.pass,
243
246
  details: outcome.details,
@@ -259,6 +262,12 @@ function classifySignalOnlyTest(g) {
259
262
  }
260
263
  export async function runGates(task, ctx) {
261
264
  const results = [];
265
+ const evidence = {
266
+ artifactDir: ctx.artifactDir,
267
+ runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
268
+ taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
269
+ ...ctx.evidence,
270
+ };
262
271
  // Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
263
272
  // allocates a new invocation, including retries whose local spawn counter starts at one again.
264
273
  let currentBuild;
@@ -310,9 +319,11 @@ export async function runGates(task, ctx) {
310
319
  // OBS-1055: on a recheck a cached red is the answer the operator just said was wrongly given; it is
311
320
  // journaled as discarded and the gate runs. Returns true when the hit must NOT be reused.
312
321
  const discardCachedRed = async (gate, hit) => {
313
- if (!ctx.recheck || hit.pass)
322
+ const reason = ctx.recheck ? "recheck" : ctx.cachedRedBypass;
323
+ if (!reason || hit.pass)
314
324
  return false;
315
- await ctx.onGate?.({ phase: "note", gate, name: "recheck-rerun", payload: { gate, reason: "cached-red-discarded" }, result: hit });
325
+ await ctx.onGate?.({ phase: "note", gate, name: reason === "recheck" ? "recheck-rerun" : "gate-rerun",
326
+ payload: { gate, reason: "cached-red-discarded", ...(reason === "recheck" ? {} : { bypass: reason }) }, result: hit });
316
327
  return true;
317
328
  };
318
329
  const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
@@ -656,8 +667,8 @@ export async function runGates(task, ctx) {
656
667
  // other scripted test command keeps today's exit-code contract byte-identically.
657
668
  const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
658
669
  r = useManifest
659
- ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
660
- : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
670
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence }))
671
+ : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), evidence, ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
661
672
  }
662
673
  finally {
663
674
  await receiptNotes;
@@ -674,7 +685,7 @@ export async function runGates(task, ctx) {
674
685
  if (!cached && r.pass && commands[g]) {
675
686
  const dirt = await dirtyWorktree();
676
687
  if (dirt) {
677
- await record(await dirtyRefusal(g, dirt, commands[g]));
688
+ await record({ ...await dirtyRefusal(g, dirt, commands[g]), evidenceReceipt: r.evidenceReceipt, evidenceReceipts: r.evidenceReceipts });
678
689
  return;
679
690
  }
680
691
  }
@@ -901,7 +912,7 @@ export async function runGates(task, ctx) {
901
912
  const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
902
913
  const carriedAuthors = ctx.carriedAuthors ?? [];
903
914
  let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
904
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
915
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
905
916
  // OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
906
917
  // a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
907
918
  // verdict never enters results; an exhausted pool preserves its cause.
@@ -934,7 +945,7 @@ export async function runGates(task, ctx) {
934
945
  const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
935
946
  const exclusion = crossAdapter ? "adapter" : "channel";
936
947
  exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
937
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
948
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
938
949
  if (second.meta?.noEligibleReviewer !== true) {
939
950
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
940
951
  const route = exclusion === "adapter"
@@ -1102,8 +1113,8 @@ export async function runGates(task, ctx) {
1102
1113
  if (!full) {
1103
1114
  const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
1104
1115
  full = fullUsesManifest
1105
- ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
1106
- : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
1116
+ ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, { ...retryOptions(identity), evidence }))
1117
+ : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], { ...retryOptions(identity), evidence })))[0];
1107
1118
  }
1108
1119
  fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
1109
1120
  const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
@@ -1113,8 +1124,9 @@ export async function runGates(task, ctx) {
1113
1124
  verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
1114
1125
  }
1115
1126
  const merged = withTelemetry(dirt
1116
- ? await dirtyRefusal("test", dirt, ctx.commands.test)
1127
+ ? { ...await dirtyRefusal("test", dirt, ctx.commands.test), evidenceReceipt: full.evidenceReceipt, evidenceReceipts: full.evidenceReceipts }
1117
1128
  : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
1129
+ merged.evidenceReceipts = [...(heldTest?.evidenceReceipts ?? []), ...(full?.evidenceReceipts ?? [])];
1118
1130
  results[results.findIndex((r) => r.gate === "test")] = merged;
1119
1131
  heldTest = undefined;
1120
1132
  await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
@@ -1,4 +1,5 @@
1
- import type { BaselineFileDuration } from "./baseline.js";
1
+ import { type GateEvidenceOptions, type BaselineFileDuration } from "./baseline.js";
2
+ import type { GateEvidenceReceipt } from "../run/protocol.js";
2
3
  export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
3
4
  /** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
4
5
  * --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
@@ -20,6 +21,15 @@ export interface TestReportCompletion {
20
21
  failed: number;
21
22
  skipped: number;
22
23
  };
24
+ /** Bounded (4096 bytes) runner evidence per failed test — diff, actual/expected or stack head.
25
+ * Never part of the failure's identity: `failures` alone feeds details and fingerprints. */
26
+ evidence?: FailureEvidence[];
27
+ }
28
+ export interface FailureEvidence {
29
+ test: string;
30
+ text: string;
31
+ truncated?: true;
32
+ unavailable?: true;
23
33
  }
24
34
  /** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
25
35
  export interface TestReport {
@@ -69,6 +79,8 @@ export declare const FILE_HANG_SLACK = 3;
69
79
  export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
70
80
  export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
71
81
  export interface ManifestRunResult {
82
+ evidenceReceipt: GateEvidenceReceipt;
83
+ evidenceReceipts: GateEvidenceReceipt[];
72
84
  exitCode: number | undefined;
73
85
  stdout: string;
74
86
  stderr: string;
@@ -82,6 +94,7 @@ export interface ManifestRunResult {
82
94
  /** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
83
95
  * timeout kills the detached process group, including descendants holding the output pipes. */
84
96
  export declare function runManifestedTest(cmd: string, cwd: string, opts: {
97
+ evidence?: GateEvidenceOptions;
85
98
  manifest: readonly string[];
86
99
  nonce: string;
87
100
  reportPath: string;
@@ -92,6 +105,8 @@ export declare function runManifestedTest(cmd: string, cwd: string, opts: {
92
105
  overallCeilingMs?: number;
93
106
  }): Promise<ManifestRunResult>;
94
107
  export interface ManifestGateOutcome {
108
+ evidenceReceipt?: GateEvidenceReceipt;
109
+ evidenceReceipts?: GateEvidenceReceipt[];
95
110
  pass: boolean;
96
111
  kind: ManifestVerdictKind;
97
112
  details: string;
@@ -106,6 +121,8 @@ export interface DiscoveredManifest {
106
121
  separator: string;
107
122
  listingExit: number | undefined;
108
123
  listingStdout: string;
124
+ evidenceReceipt: GateEvidenceReceipt;
125
+ evidenceReceipts: GateEvidenceReceipt[];
109
126
  }
110
127
  /** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
111
128
  * from its own listing and never from a stdout summary. The gate and the baseline capture share it,
@@ -116,6 +133,7 @@ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
116
133
  nonce: string;
117
134
  env: NodeJS.ProcessEnv;
118
135
  overallCeilingMs?: number;
136
+ evidence?: GateEvidenceOptions;
119
137
  }): Promise<DiscoveredManifest>;
120
138
  /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
121
139
  * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
@@ -128,4 +146,5 @@ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
128
146
  longestFile?: BaselineFileDuration | null;
129
147
  overallCeilingMs?: number;
130
148
  artifactDir?: string;
149
+ evidence?: GateEvidenceOptions;
131
150
  }): Promise<ManifestGateOutcome>;
@@ -4,6 +4,7 @@ import { tmpdir } from "node:os";
4
4
  import { isAbsolute, join, relative, sep } from "node:path";
5
5
  import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
6
6
  import { shq } from "../adapters/types.js";
7
+ import { beginGateEvidence, redactGateOutput } from "./baseline.js";
7
8
  import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
8
9
  /**
9
10
  * VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
@@ -103,6 +104,12 @@ export function toManifestPath(file, cwd) {
103
104
  catch { /* cwd unreadable — best effort with the given path */ }
104
105
  return relative(resolvedCwd, file).split(sep).join("/");
105
106
  }
107
+ function isFailureEvidence(v) {
108
+ if (typeof v !== "object" || v === null)
109
+ return false;
110
+ const e = v;
111
+ return typeof e.test === "string" && typeof e.text === "string";
112
+ }
106
113
  function isTestReportShape(v) {
107
114
  if (typeof v !== "object" || v === null)
108
115
  return false;
@@ -228,10 +235,12 @@ export function verifyManifestReport(opts) {
228
235
  const failedCompletions = Object.entries(report.completed).filter(([, c]) => c.status === "failed");
229
236
  const failingFiles = failedCompletions.map(([file]) => file).sort();
230
237
  const failures = failedCompletions.flatMap(([, c]) => c.failures?.length ? c.failures : ["<unnamed failure>"]);
238
+ // Parsed defensively: a malformed entry is dropped, never a verdict change — evidence is diagnostics only.
239
+ const failureEvidence = failedCompletions.flatMap(([, c]) => Array.isArray(c.evidence) ? c.evidence.filter(isFailureEvidence) : []);
231
240
  if (failures.length)
232
241
  return { kind: "work", pass: false,
233
242
  details: `test report names failing fingerprint(s):\n${failures.join("\n")}`,
234
- meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode } };
243
+ meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode, failureEvidence } };
235
244
  if (exitCode === undefined) {
236
245
  return {
237
246
  kind: "fail-closed",
@@ -322,6 +331,7 @@ export function runManifestedTest(cmd, cwd, opts) {
322
331
  const pollMs = opts.pollMs ?? 20;
323
332
  const overallCeilingMs = usable(opts.overallCeilingMs) ? opts.overallCeilingMs : DEFAULT_FILE_HANG_BUDGET_MS;
324
333
  const env = { ...(opts.env ?? process.env), TICKMARKR_TEST_REPORT: opts.reportPath, TICKMARKR_TEST_NONCE: opts.nonce };
334
+ const evidence = beginGateEvidence(cwd, "test", cmd, { ...opts.evidence, env }, opts.nonce);
325
335
  const controller = new AbortController();
326
336
  let pid;
327
337
  let killedFile;
@@ -347,6 +357,7 @@ export function runManifestedTest(cmd, cwd, opts) {
347
357
  }
348
358
  };
349
359
  return shell(cmd, cwd, overallCeilingMs, false, {
360
+ onReceipt: receipt => evidence.observe(killedFile && receipt.outcome === "cancelled" ? { ...receipt, outcome: "timed-out" } : receipt),
350
361
  env, signal: controller.signal, onTimeout: () => checkHang(true),
351
362
  onSpawn: (childPid) => {
352
363
  pid = childPid;
@@ -354,11 +365,21 @@ export function runManifestedTest(cmd, cwd, opts) {
354
365
  clearInterval(poll);
355
366
  poll = setInterval(checkHang, pollMs);
356
367
  },
357
- }).then((result) => ({
358
- exitCode: result.signalExit ? undefined : result.code,
359
- stdout: result.stdout, stderr: result.stderr,
360
- report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
361
- })).finally(() => { clearInterval(poll); });
368
+ }).then((result) => {
369
+ const evidenceReceipt = evidence.finish(result.stdout, result.stderr);
370
+ return {
371
+ evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt],
372
+ exitCode: result.signalExit ? undefined : result.code,
373
+ stdout: result.stdout, stderr: result.stderr,
374
+ report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
375
+ };
376
+ }).catch((error) => {
377
+ if (error instanceof Error) {
378
+ const evidenceReceipt = evidence.finish();
379
+ Object.assign(error, { evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt] });
380
+ }
381
+ throw error;
382
+ }).finally(() => { clearInterval(poll); });
362
383
  }
363
384
  /** The child environment every manifest invocation (listing and run) receives, and the lifecycle
364
385
  * protocol it records. R41: the policy is whatever THIS process was launched with (npm reads
@@ -383,11 +404,12 @@ function manifestEnvironment(cwd) {
383
404
  export async function discoverTestManifest(cmd, cwd, opts) {
384
405
  const invocation = runnerInvocation(cmd, cwd);
385
406
  const listed = await runManifestedTest(invocation.listing, cwd, {
386
- manifest: [], nonce: opts.nonce, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
407
+ evidence: opts.evidence, manifest: [], nonce: `${opts.nonce}-listing`, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
387
408
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
388
409
  });
410
+ const discoveryError = (message) => Object.assign(new Error(message), { evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts });
389
411
  if (listed.exitCode !== 0)
390
- throw new Error(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
412
+ throw discoveryError(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
391
413
  let files;
392
414
  try {
393
415
  const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
@@ -396,11 +418,11 @@ export async function discoverTestManifest(cmd, cwd, opts) {
396
418
  files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
397
419
  }
398
420
  catch {
399
- throw new Error(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
421
+ throw discoveryError(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
400
422
  }
401
423
  if (!files.length)
402
- throw new Error("vitest cannot list files: empty manifest");
403
- return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout };
424
+ throw discoveryError("vitest cannot list files: empty manifest");
425
+ return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout, evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts };
404
426
  }
405
427
  /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
406
428
  * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
@@ -422,11 +444,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
422
444
  const nonce = randomBytes(16).toString("hex");
423
445
  const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
424
446
  const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
447
+ const evidenceReceipts = [];
425
448
  let spawnedCommand = cmd;
426
449
  let manifestPath;
427
450
  const { env, verification } = manifestEnvironment(cwd);
428
451
  try {
429
- const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs });
452
+ const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
453
+ evidenceReceipts.push(...invocation.evidenceReceipts);
430
454
  const files = invocation.files;
431
455
  // R41: the EXPECTED manifest is evidence in its own right — persisted beside the report with the
432
456
  // exact discovery invocation, so a later reader can tell what this invocation was asked to prove
@@ -440,7 +464,7 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
440
464
  writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
441
465
  spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
442
466
  const invoked = await runManifestedTest(spawnedCommand, cwd, {
443
- manifest: files, nonce, reportPath, env,
467
+ evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: files, nonce, reportPath, env,
444
468
  baselineDurations: opts.baselineDurations?.map((d) => ({ ...d, file: toManifestPath(d.file, cwd) })),
445
469
  longestFile: opts.longestFile,
446
470
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
@@ -449,12 +473,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
449
473
  const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
450
474
  report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
451
475
  // Preserve the validator's verdict and classification; runner evidence only explains it.
452
- const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
453
- const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
454
- const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
455
- const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
456
- writeFileSync(stdoutPath, stdoutTail);
457
- writeFileSync(stderrPath, stderrTail);
476
+ const evidenceReceipt = invoked.evidenceReceipt;
477
+ evidenceReceipts.push(...invoked.evidenceReceipts);
478
+ const evidenceRoot = opts.evidence?.artifactDir ?? dir;
479
+ const stdoutPath = join(evidenceRoot, evidenceReceipt.stdout.path);
480
+ const stderrPath = join(evidenceRoot, evidenceReceipt.stderr.path);
481
+ const stdoutTail = Buffer.from(redactGateOutput(invoked.stdout, env)).subarray(-16 * 1024);
482
+ const stderrTail = Buffer.from(redactGateOutput(invoked.stderr, env)).subarray(-16 * 1024);
458
483
  const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
459
484
  const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
460
485
  const errors = report?.certificate?.errors;
@@ -462,11 +487,11 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
462
487
  const reportedDiagnostics = report?.certificate?.diagnostics;
463
488
  const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
464
489
  const diagnostics = !verdict.pass
465
- ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
490
+ ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}; runner vitest`
466
491
  + [...runnerErrors,
467
492
  stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
468
493
  : "";
469
- return { pass: verdict.pass, kind: verdict.kind,
494
+ return { evidenceReceipt, evidenceReceipts, pass: verdict.pass, kind: verdict.kind,
470
495
  details: verdict.details + diagnostics,
471
496
  classification: verdict.meta.classification,
472
497
  meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
@@ -474,7 +499,10 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
474
499
  exitCode: invoked.exitCode ?? -1, reportPath };
475
500
  }
476
501
  catch (error) {
477
- return { pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
502
+ const evidenceReceipt = error?.evidenceReceipt;
503
+ if (evidenceReceipt)
504
+ evidenceReceipts.push(...(error.evidenceReceipts ?? [evidenceReceipt]));
505
+ return { evidenceReceipt, evidenceReceipts, pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
478
506
  details: error instanceof Error ? error.message : String(error),
479
507
  meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
480
508
  }
@@ -11,6 +11,24 @@ export default class TickmarkrReporter {
11
11
  this.cwd = realpathSync(process.cwd());
12
12
  }
13
13
  file(module) { return relative(this.cwd, module.moduleId).split(sep).join('/'); }
14
+ // Bounded assertion evidence, kept BESIDE the failure fingerprint and never inside it: the
15
+ // fingerprint is the failure's identity (the repeated-failure cap compares it), the evidence is
16
+ // what the runner knew and the message elided. ONE 4096-byte budget per failed test, shared by
17
+ // all of its errors (expect.soft yields several); never invented.
18
+ evidence(test, errors) {
19
+ const parts = [];
20
+ for (const e of errors) {
21
+ if (e && typeof e.diff === 'string' && e.diff) parts.push(e.diff);
22
+ else for (const k of ['actual', 'expected']) if (e && e[k] !== undefined && e[k] !== null) parts.push(k + ': ' + (typeof e[k] === 'string' ? e[k] : JSON.stringify(e[k])));
23
+ if (e && typeof e.stack === 'string' && e.stack) parts.push(e.stack.split('\n').slice(0, 8).join('\n'));
24
+ }
25
+ if (!parts.length) return { test, text: '', unavailable: true };
26
+ const full = parts.join('\n').replace(/\x1b\[[0-9;]*m/g, '');
27
+ if (Buffer.byteLength(full) <= 4096) return { test, text: full };
28
+ let text = full.slice(0, 4096);
29
+ while (Buffer.byteLength(text) > 4096) text = text.slice(0, -1);
30
+ return { test, text, truncated: true };
31
+ }
14
32
  save() {
15
33
  writeFileSync(this.path + '.tmp', JSON.stringify(this.report));
16
34
  renameSync(this.path + '.tmp', this.path);
@@ -27,6 +45,7 @@ export default class TickmarkrReporter {
27
45
  const file = this.file(module);
28
46
  const failed = module.state() === 'failed';
29
47
  const failures = [];
48
+ const evidence = [];
30
49
  // R41: count test bodies by their own state so a module whose every test was skipped (a
31
50
  // describe.skipIf gate) is recorded as SKIPPED — present in the lifecycle, but never as
32
51
  // executed test-body success. 'passed'/'failed' executed; anything else did not run.
@@ -37,6 +56,8 @@ export default class TickmarkrReporter {
37
56
  tests.failed++;
38
57
  if (failed) {
39
58
  const errors = test.result().errors || [];
59
+ const name = file + ' > ' + test.fullName;
60
+ evidence.push(this.evidence(name, errors));
40
61
  failures.push(...(errors.length ? errors.map(e => 'FAIL ' + file + ' > ' + test.fullName + ': ' + e.message) : ['FAIL ' + file + ' > ' + test.fullName]));
41
62
  }
42
63
  } else if (state === 'passed') tests.passed++;
@@ -45,7 +66,7 @@ export default class TickmarkrReporter {
45
66
  if (failed && !failures.length) failures.push('FAIL ' + file);
46
67
  const status = failed ? 'failed' : tests.passed + tests.failed === 0 ? 'skipped' : 'passed';
47
68
  if (file in this.report.completed) this.report.duplicateCompletions.push(file);
48
- this.report.completed[file] = { at: Date.now(), status, failures, tests };
69
+ this.report.completed[file] = { at: Date.now(), status, failures, tests, ...(evidence.length ? { evidence } : {}) };
49
70
  this.save();
50
71
  }
51
72
  onTestRunEnd(modules, errors, reason) {