tickmarkr 2.5.7 → 2.5.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/qwen.js +30 -3
- package/dist/cli/commands/approve.js +23 -4
- package/dist/cli/commands/beat.d.ts +2 -0
- package/dist/cli/commands/beat.js +28 -29
- package/dist/cli/commands/compile.js +18 -0
- package/dist/cli/commands/fleet.js +61 -53
- package/dist/cli/commands/plan.d.ts +5 -0
- package/dist/cli/commands/plan.js +28 -23
- package/dist/cli/commands/resume.js +1 -1
- package/dist/cli/commands/run.js +1 -1
- package/dist/cli/commands/verify.d.ts +4 -1
- package/dist/cli/commands/verify.js +16 -5
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +6 -4
- package/dist/compile/native.js +68 -7
- package/dist/compile/retired-literals.d.ts +22 -0
- package/dist/compile/retired-literals.js +271 -0
- package/dist/config/config.d.ts +41 -3
- package/dist/config/config.js +48 -17
- package/dist/config/fleet-overlay.js +47 -62
- package/dist/drivers/index.d.ts +4 -2
- package/dist/drivers/index.js +54 -6
- package/dist/gates/baseline.d.ts +65 -0
- package/dist/gates/baseline.js +163 -9
- package/dist/gates/review.d.ts +6 -4
- package/dist/gates/review.js +62 -21
- package/dist/gates/run-gates.d.ts +6 -1
- package/dist/gates/run-gates.js +25 -13
- package/dist/gates/test-manifest.d.ts +20 -1
- package/dist/gates/test-manifest.js +50 -22
- package/dist/gates/test-reporter.js +22 -1
- package/dist/graph/schema.d.ts +28 -0
- package/dist/graph/schema.js +13 -1
- package/dist/run/daemon.d.ts +19 -0
- package/dist/run/daemon.js +363 -118
- package/dist/run/journal.d.ts +54 -3
- package/dist/run/journal.js +142 -11
- package/dist/run/merge.d.ts +15 -2
- package/dist/run/merge.js +74 -11
- package/dist/run/protocol.d.ts +82 -0
- package/dist/run/protocol.js +35 -0
- package/dist/run/receipt-resolver.d.ts +18 -0
- package/dist/run/receipt-resolver.js +132 -0
- package/dist/run/repair-disposition.d.ts +41 -0
- package/dist/run/repair-disposition.js +77 -0
- package/dist/run/supervision.d.ts +14 -1
- package/dist/run/supervision.js +122 -24
- package/dist/tui/cockpit/evidence-view.d.ts +10 -1
- package/dist/tui/cockpit/evidence-view.js +37 -5
- package/dist/tui/cockpit/home-view.js +45 -30
- package/dist/tui/cockpit/live-store.d.ts +18 -0
- package/dist/tui/ink/fleet-app.d.ts +12 -22
- package/dist/tui/ink/fleet-app.js +520 -131
- package/package.json +1 -1
- package/schema/rungraph.schema.json +54 -0
- package/skills/tickmarkr-overseer/SKILL.md +168 -36
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
package/dist/gates/review.js
CHANGED
|
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
|
|
|
6
6
|
import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
|
-
import { structuredFindings } from "../run/journal.js";
|
|
9
|
+
import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
|
|
10
10
|
import { redactSecrets } from "../run/redact.js";
|
|
11
11
|
import { marginalCostRank } from "../route/router.js";
|
|
12
12
|
import { modelProvider } from "../route/preference.js";
|
|
13
13
|
import { resolveStateDir } from "./cache.js";
|
|
14
|
-
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
14
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
15
15
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
16
16
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
17
17
|
export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
|
|
@@ -188,15 +188,17 @@ export function matchClosureId(candidate, target) {
|
|
|
188
188
|
// OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
|
|
189
189
|
// it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
|
|
190
190
|
// `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
|
|
191
|
-
const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
|
|
192
191
|
if (typeof target === "string") {
|
|
193
|
-
return
|
|
192
|
+
return reviewFingerprintMatches(candidate, target);
|
|
194
193
|
}
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
}
|
|
199
|
-
|
|
194
|
+
const matches = [...target].filter((finding) => {
|
|
195
|
+
const ids = typeof finding === "string" ? [finding] : observedReviewFingerprints(finding);
|
|
196
|
+
return ids.some((id) => reviewFingerprintMatches(candidate, id));
|
|
197
|
+
});
|
|
198
|
+
if (matches.length !== 1)
|
|
199
|
+
return undefined;
|
|
200
|
+
const match = matches[0];
|
|
201
|
+
return typeof match === "string" ? match : match.fingerprint;
|
|
200
202
|
}
|
|
201
203
|
/**
|
|
202
204
|
* Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
|
|
@@ -205,10 +207,10 @@ export function matchClosureId(candidate, target) {
|
|
|
205
207
|
export function isReviewClosureInvalid(v, priorIds) {
|
|
206
208
|
const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
|
|
207
209
|
const closureLists = [v?.resolved, v?.reraised];
|
|
208
|
-
const allCandidateIds =
|
|
210
|
+
const allCandidateIds = closureLists.flatMap((list) => Array.isArray(list) ? list : []);
|
|
209
211
|
return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
|
|
210
212
|
|| new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
|
|
211
|
-
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
|
|
213
|
+
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, priors) === (typeof id === "string" ? id : id.fingerprint))));
|
|
212
214
|
}
|
|
213
215
|
/**
|
|
214
216
|
* OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
|
|
@@ -368,10 +370,17 @@ function daemonRepoRoot(worktree, artifactDir) {
|
|
|
368
370
|
* closure on a typo and that read as malformed — the block is what a closure list is copied from.
|
|
369
371
|
*/
|
|
370
372
|
export function renderPriorMaterials(priorMaterials) {
|
|
373
|
+
const fingerprints = priorMaterials.map((finding, i) => {
|
|
374
|
+
const observed = observedReviewFingerprints(finding);
|
|
375
|
+
const heading = observed.length > 1 ? `Finding ${i + 1} (choose one observed spelling):\n` : "";
|
|
376
|
+
return heading + observed.map((id) => `Fingerprint: ${id}`).join("\n");
|
|
377
|
+
}).join("\n");
|
|
371
378
|
return `## Prior materials this attempt must close
|
|
372
|
-
|
|
379
|
+
For each finding below, copy exactly ONE of its observed fingerprints into resolved or reraised.
|
|
380
|
+
Use only these observed spellings. For every findings entry that restates a reraised prior, whether at the same path or a new path, set its "reraised" field to the copied id; unrelated defects need separate entries.
|
|
381
|
+
The fingerprints appear once, in this block:
|
|
373
382
|
\`\`\`text
|
|
374
|
-
${
|
|
383
|
+
${fingerprints}
|
|
375
384
|
\`\`\`
|
|
376
385
|
${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
|
|
377
386
|
}
|
|
@@ -383,7 +392,7 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
|
|
|
383
392
|
// never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
|
|
384
393
|
priorReviewers = [],
|
|
385
394
|
// OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
|
|
386
|
-
carriedAuthors = []) {
|
|
395
|
+
carriedAuthors = [], operatorContext) {
|
|
387
396
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
388
397
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
389
398
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -502,6 +511,10 @@ ${suiteBudget} Never run the whole suite (including an unfiltered npm test or vi
|
|
|
502
511
|
|
|
503
512
|
${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
|
|
504
513
|
|
|
514
|
+
` : ""}${operatorContext?.trim() ? `## Operator context
|
|
515
|
+
Context only: this never substitutes for an acceptance criterion or closes a prior material.
|
|
516
|
+
${operatorContext.trim()}
|
|
517
|
+
|
|
505
518
|
` : ""}## Diff
|
|
506
519
|
\`\`\`diff
|
|
507
520
|
${diff}
|
|
@@ -578,7 +591,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
578
591
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
579
592
|
const v = extractVerdictJson(raw, nonce);
|
|
580
593
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
581
|
-
const priorIds =
|
|
594
|
+
const priorIds = priorMaterials;
|
|
582
595
|
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
583
596
|
const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
|
|
584
597
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
@@ -610,7 +623,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
610
623
|
provider,
|
|
611
624
|
...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
|
|
612
625
|
cause,
|
|
613
|
-
...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints:
|
|
626
|
+
...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
|
|
614
627
|
bytes, seatAuthoredBytes: bytes,
|
|
615
628
|
...(saved ? { rawPath: saved } : {}),
|
|
616
629
|
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
@@ -621,7 +634,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
621
634
|
const decided = findings !== null
|
|
622
635
|
? classifyReviewFindings(findings)
|
|
623
636
|
: classifyReviewIssues(v.approve, v.issues);
|
|
624
|
-
const
|
|
637
|
+
const ownLines = [...decided.lines];
|
|
638
|
+
const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, [finding])));
|
|
625
639
|
if (reraised.length) {
|
|
626
640
|
if (decided.pass)
|
|
627
641
|
decided.headline = "requested changes";
|
|
@@ -636,6 +650,36 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
636
650
|
}
|
|
637
651
|
const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
|
|
638
652
|
const details = appendAnchoredReview(prose, v);
|
|
653
|
+
// Only the verdict's anchors may supply missing evidence, and only when unambiguous.
|
|
654
|
+
// Reuse the journal's path normalization without changing legacy details-only parsing.
|
|
655
|
+
const anchoredPaths = new Set(parseAnchoredComments(v).map((comment) => {
|
|
656
|
+
const anchor = structuredFindings("review", `- ${comment.path}:${comment.line} — anchor`)
|
|
657
|
+
.find((finding) => finding.class === "review:anchored");
|
|
658
|
+
return anchor?.path ?? comment.path;
|
|
659
|
+
}));
|
|
660
|
+
const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
|
|
661
|
+
const ownDetails = appendAnchoredReview(ownLines.join("\n"), v);
|
|
662
|
+
const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
|
|
663
|
+
.map((finding) => finding.path === UNIDENTIFIED && anchoredPath
|
|
664
|
+
? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
|
|
665
|
+
: finding);
|
|
666
|
+
// A top-level reraised id identifies a prior, not an arbitrary new defect. Bind a restatement
|
|
667
|
+
// only when its own entry echoes that validated id (or its note explicitly contains the id).
|
|
668
|
+
const linkedRows = currentRows.map((finding) => {
|
|
669
|
+
if (finding.class !== "review:material")
|
|
670
|
+
return finding;
|
|
671
|
+
const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
|
|
672
|
+
const mentioned = reraised.filter((prior) => observedReviewFingerprints(prior)
|
|
673
|
+
.some((fp) => finding.note.includes(`\`${fp}\``) || finding.note.trim() === fp));
|
|
674
|
+
const id = matchClosureId(entry?.reraised, reraised)
|
|
675
|
+
?? (mentioned.length === 1 ? mentioned[0].fingerprint : undefined);
|
|
676
|
+
return id ? { ...finding, reraisedFrom: id } : finding;
|
|
677
|
+
});
|
|
678
|
+
// Several rows claiming the same prior are ambiguous; none gets to erase the others.
|
|
679
|
+
const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
|
|
680
|
+
&& linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
|
|
681
|
+
? { ...finding, reraisedFrom: undefined } : finding);
|
|
682
|
+
const carriedRows = carryReviewFindings(reraised, unambiguousRows);
|
|
639
683
|
return {
|
|
640
684
|
gate: "review",
|
|
641
685
|
pass: decided.pass,
|
|
@@ -649,10 +693,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
649
693
|
resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
650
694
|
reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
651
695
|
} : {}),
|
|
652
|
-
...(
|
|
653
|
-
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
654
|
-
...reraised,
|
|
655
|
-
] } : {}),
|
|
696
|
+
...(!decided.pass ? { findings: carriedRows } : {}),
|
|
656
697
|
...(saved ? { rawPath: saved } : {}),
|
|
657
698
|
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
658
699
|
},
|
|
@@ -2,7 +2,7 @@ import type { CommandReceiptAttribution } from "../run/protocol.js";
|
|
|
2
2
|
import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
|
|
3
3
|
import { type TickmarkrConfig } from "../config/config.js";
|
|
4
4
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
5
|
-
import { type Baseline } from "./baseline.js";
|
|
5
|
+
import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
|
|
6
6
|
import { type GateVia } from "./llm.js";
|
|
7
7
|
import { type PriorReviewer } from "./review.js";
|
|
8
8
|
import type { GateResult } from "./types.js";
|
|
@@ -46,6 +46,7 @@ export type GateEvent = {
|
|
|
46
46
|
result?: GateResult;
|
|
47
47
|
};
|
|
48
48
|
export interface GateContext {
|
|
49
|
+
evidence?: GateEvidenceOptions;
|
|
49
50
|
buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
|
|
50
51
|
authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
|
|
51
52
|
verificationScope?: VerificationScope;
|
|
@@ -61,10 +62,14 @@ export interface GateContext {
|
|
|
61
62
|
cfg: TickmarkrConfig;
|
|
62
63
|
via?: GateVia;
|
|
63
64
|
carriedFindings?: readonly StructuredFinding[];
|
|
65
|
+
/** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
|
|
66
|
+
operatorContext?: string;
|
|
64
67
|
excludeReviewers?: string[];
|
|
65
68
|
demotedReviewers?: Set<string>;
|
|
66
69
|
reviewNoVerdicts?: Map<string, string[]>;
|
|
67
70
|
recheck?: boolean;
|
|
71
|
+
/** Explicit worker funding requires fresh red measurements, never a gate waiver. */
|
|
72
|
+
cachedRedBypass?: "operator-rerun";
|
|
68
73
|
carriedAuthors?: readonly string[];
|
|
69
74
|
reviewHistory?: string[];
|
|
70
75
|
priorReviewers?: PriorReviewer[];
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -222,22 +222,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
|
|
|
222
222
|
longestFile: entry?.longestFile,
|
|
223
223
|
overallCeilingMs: effectiveCeilingMs(entry),
|
|
224
224
|
artifactDir,
|
|
225
|
+
evidence: retry.evidence,
|
|
225
226
|
});
|
|
226
227
|
const reportPath = outcome.reportPath;
|
|
228
|
+
const evidence = { evidenceReceipt: outcome.evidenceReceipt, evidenceReceipts: outcome.evidenceReceipts };
|
|
227
229
|
if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
|
|
228
230
|
&& outcome.meta?.retryable !== false) {
|
|
229
231
|
const waitedMs = await waitForCalmWindow(executionSignal());
|
|
230
232
|
if (!calmWindowReady())
|
|
231
|
-
return { gate: "test", pass: false, details: outcome.details,
|
|
233
|
+
return { ...evidence, gate: "test", pass: false, details: outcome.details,
|
|
232
234
|
meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
|
|
233
235
|
if (!retry.authorizeRetry("infra")) {
|
|
234
|
-
return { gate: "test", pass: false, details: outcome.details,
|
|
236
|
+
return { ...evidence, gate: "test", pass: false, details: outcome.details,
|
|
235
237
|
meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
|
|
236
238
|
}
|
|
237
239
|
const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
|
|
238
|
-
return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
|
|
240
|
+
return { ...result, evidenceReceipts: [...(outcome.evidenceReceipts ?? []), ...(result.evidenceReceipts ?? [])], meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
|
|
239
241
|
}
|
|
240
242
|
return {
|
|
243
|
+
...evidence,
|
|
241
244
|
gate: "test",
|
|
242
245
|
pass: outcome.pass,
|
|
243
246
|
details: outcome.details,
|
|
@@ -259,6 +262,12 @@ function classifySignalOnlyTest(g) {
|
|
|
259
262
|
}
|
|
260
263
|
export async function runGates(task, ctx) {
|
|
261
264
|
const results = [];
|
|
265
|
+
const evidence = {
|
|
266
|
+
artifactDir: ctx.artifactDir,
|
|
267
|
+
runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
|
|
268
|
+
taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
|
|
269
|
+
...ctx.evidence,
|
|
270
|
+
};
|
|
262
271
|
// Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
|
|
263
272
|
// allocates a new invocation, including retries whose local spawn counter starts at one again.
|
|
264
273
|
let currentBuild;
|
|
@@ -310,9 +319,11 @@ export async function runGates(task, ctx) {
|
|
|
310
319
|
// OBS-1055: on a recheck a cached red is the answer the operator just said was wrongly given; it is
|
|
311
320
|
// journaled as discarded and the gate runs. Returns true when the hit must NOT be reused.
|
|
312
321
|
const discardCachedRed = async (gate, hit) => {
|
|
313
|
-
|
|
322
|
+
const reason = ctx.recheck ? "recheck" : ctx.cachedRedBypass;
|
|
323
|
+
if (!reason || hit.pass)
|
|
314
324
|
return false;
|
|
315
|
-
await ctx.onGate?.({ phase: "note", gate, name: "recheck-rerun"
|
|
325
|
+
await ctx.onGate?.({ phase: "note", gate, name: reason === "recheck" ? "recheck-rerun" : "gate-rerun",
|
|
326
|
+
payload: { gate, reason: "cached-red-discarded", ...(reason === "recheck" ? {} : { bypass: reason }) }, result: hit });
|
|
316
327
|
return true;
|
|
317
328
|
};
|
|
318
329
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
@@ -656,8 +667,8 @@ export async function runGates(task, ctx) {
|
|
|
656
667
|
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
657
668
|
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
658
669
|
r = useManifest
|
|
659
|
-
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
|
|
660
|
-
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
670
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence }))
|
|
671
|
+
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), evidence, ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
661
672
|
}
|
|
662
673
|
finally {
|
|
663
674
|
await receiptNotes;
|
|
@@ -674,7 +685,7 @@ export async function runGates(task, ctx) {
|
|
|
674
685
|
if (!cached && r.pass && commands[g]) {
|
|
675
686
|
const dirt = await dirtyWorktree();
|
|
676
687
|
if (dirt) {
|
|
677
|
-
await record(await dirtyRefusal(g, dirt, commands[g]));
|
|
688
|
+
await record({ ...await dirtyRefusal(g, dirt, commands[g]), evidenceReceipt: r.evidenceReceipt, evidenceReceipts: r.evidenceReceipts });
|
|
678
689
|
return;
|
|
679
690
|
}
|
|
680
691
|
}
|
|
@@ -901,7 +912,7 @@ export async function runGates(task, ctx) {
|
|
|
901
912
|
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
902
913
|
const carriedAuthors = ctx.carriedAuthors ?? [];
|
|
903
914
|
let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
|
|
904
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
|
|
915
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
|
|
905
916
|
// OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
|
|
906
917
|
// a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
|
|
907
918
|
// verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -934,7 +945,7 @@ export async function runGates(task, ctx) {
|
|
|
934
945
|
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
|
|
935
946
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
936
947
|
exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
937
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
|
|
948
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
|
|
938
949
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
939
950
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
940
951
|
const route = exclusion === "adapter"
|
|
@@ -1102,8 +1113,8 @@ export async function runGates(task, ctx) {
|
|
|
1102
1113
|
if (!full) {
|
|
1103
1114
|
const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
|
|
1104
1115
|
full = fullUsesManifest
|
|
1105
|
-
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
|
|
1106
|
-
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
|
|
1116
|
+
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, { ...retryOptions(identity), evidence }))
|
|
1117
|
+
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], { ...retryOptions(identity), evidence })))[0];
|
|
1107
1118
|
}
|
|
1108
1119
|
fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
|
|
1109
1120
|
const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
|
|
@@ -1113,8 +1124,9 @@ export async function runGates(task, ctx) {
|
|
|
1113
1124
|
verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
1114
1125
|
}
|
|
1115
1126
|
const merged = withTelemetry(dirt
|
|
1116
|
-
? await dirtyRefusal("test", dirt, ctx.commands.test)
|
|
1127
|
+
? { ...await dirtyRefusal("test", dirt, ctx.commands.test), evidenceReceipt: full.evidenceReceipt, evidenceReceipts: full.evidenceReceipts }
|
|
1117
1128
|
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
|
|
1129
|
+
merged.evidenceReceipts = [...(heldTest?.evidenceReceipts ?? []), ...(full?.evidenceReceipts ?? [])];
|
|
1118
1130
|
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
1119
1131
|
heldTest = undefined;
|
|
1120
1132
|
await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import type
|
|
1
|
+
import { type GateEvidenceOptions, type BaselineFileDuration } from "./baseline.js";
|
|
2
|
+
import type { GateEvidenceReceipt } from "../run/protocol.js";
|
|
2
3
|
export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
|
|
3
4
|
/** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
|
|
4
5
|
* --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
|
|
@@ -20,6 +21,15 @@ export interface TestReportCompletion {
|
|
|
20
21
|
failed: number;
|
|
21
22
|
skipped: number;
|
|
22
23
|
};
|
|
24
|
+
/** Bounded (4096 bytes) runner evidence per failed test — diff, actual/expected or stack head.
|
|
25
|
+
* Never part of the failure's identity: `failures` alone feeds details and fingerprints. */
|
|
26
|
+
evidence?: FailureEvidence[];
|
|
27
|
+
}
|
|
28
|
+
export interface FailureEvidence {
|
|
29
|
+
test: string;
|
|
30
|
+
text: string;
|
|
31
|
+
truncated?: true;
|
|
32
|
+
unavailable?: true;
|
|
23
33
|
}
|
|
24
34
|
/** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
|
|
25
35
|
export interface TestReport {
|
|
@@ -69,6 +79,8 @@ export declare const FILE_HANG_SLACK = 3;
|
|
|
69
79
|
export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
|
|
70
80
|
export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
|
|
71
81
|
export interface ManifestRunResult {
|
|
82
|
+
evidenceReceipt: GateEvidenceReceipt;
|
|
83
|
+
evidenceReceipts: GateEvidenceReceipt[];
|
|
72
84
|
exitCode: number | undefined;
|
|
73
85
|
stdout: string;
|
|
74
86
|
stderr: string;
|
|
@@ -82,6 +94,7 @@ export interface ManifestRunResult {
|
|
|
82
94
|
/** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
|
|
83
95
|
* timeout kills the detached process group, including descendants holding the output pipes. */
|
|
84
96
|
export declare function runManifestedTest(cmd: string, cwd: string, opts: {
|
|
97
|
+
evidence?: GateEvidenceOptions;
|
|
85
98
|
manifest: readonly string[];
|
|
86
99
|
nonce: string;
|
|
87
100
|
reportPath: string;
|
|
@@ -92,6 +105,8 @@ export declare function runManifestedTest(cmd: string, cwd: string, opts: {
|
|
|
92
105
|
overallCeilingMs?: number;
|
|
93
106
|
}): Promise<ManifestRunResult>;
|
|
94
107
|
export interface ManifestGateOutcome {
|
|
108
|
+
evidenceReceipt?: GateEvidenceReceipt;
|
|
109
|
+
evidenceReceipts?: GateEvidenceReceipt[];
|
|
95
110
|
pass: boolean;
|
|
96
111
|
kind: ManifestVerdictKind;
|
|
97
112
|
details: string;
|
|
@@ -106,6 +121,8 @@ export interface DiscoveredManifest {
|
|
|
106
121
|
separator: string;
|
|
107
122
|
listingExit: number | undefined;
|
|
108
123
|
listingStdout: string;
|
|
124
|
+
evidenceReceipt: GateEvidenceReceipt;
|
|
125
|
+
evidenceReceipts: GateEvidenceReceipt[];
|
|
109
126
|
}
|
|
110
127
|
/** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
|
|
111
128
|
* from its own listing and never from a stdout summary. The gate and the baseline capture share it,
|
|
@@ -116,6 +133,7 @@ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
|
|
|
116
133
|
nonce: string;
|
|
117
134
|
env: NodeJS.ProcessEnv;
|
|
118
135
|
overallCeilingMs?: number;
|
|
136
|
+
evidence?: GateEvidenceOptions;
|
|
119
137
|
}): Promise<DiscoveredManifest>;
|
|
120
138
|
/** The baseline capture's reading of the same seam: the manifest's file count, or null when the
|
|
121
139
|
* runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
|
|
@@ -128,4 +146,5 @@ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
|
|
|
128
146
|
longestFile?: BaselineFileDuration | null;
|
|
129
147
|
overallCeilingMs?: number;
|
|
130
148
|
artifactDir?: string;
|
|
149
|
+
evidence?: GateEvidenceOptions;
|
|
131
150
|
}): Promise<ManifestGateOutcome>;
|
|
@@ -4,6 +4,7 @@ import { tmpdir } from "node:os";
|
|
|
4
4
|
import { isAbsolute, join, relative, sep } from "node:path";
|
|
5
5
|
import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
|
|
6
6
|
import { shq } from "../adapters/types.js";
|
|
7
|
+
import { beginGateEvidence, redactGateOutput } from "./baseline.js";
|
|
7
8
|
import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
8
9
|
/**
|
|
9
10
|
* VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
|
|
@@ -103,6 +104,12 @@ export function toManifestPath(file, cwd) {
|
|
|
103
104
|
catch { /* cwd unreadable — best effort with the given path */ }
|
|
104
105
|
return relative(resolvedCwd, file).split(sep).join("/");
|
|
105
106
|
}
|
|
107
|
+
function isFailureEvidence(v) {
|
|
108
|
+
if (typeof v !== "object" || v === null)
|
|
109
|
+
return false;
|
|
110
|
+
const e = v;
|
|
111
|
+
return typeof e.test === "string" && typeof e.text === "string";
|
|
112
|
+
}
|
|
106
113
|
function isTestReportShape(v) {
|
|
107
114
|
if (typeof v !== "object" || v === null)
|
|
108
115
|
return false;
|
|
@@ -228,10 +235,12 @@ export function verifyManifestReport(opts) {
|
|
|
228
235
|
const failedCompletions = Object.entries(report.completed).filter(([, c]) => c.status === "failed");
|
|
229
236
|
const failingFiles = failedCompletions.map(([file]) => file).sort();
|
|
230
237
|
const failures = failedCompletions.flatMap(([, c]) => c.failures?.length ? c.failures : ["<unnamed failure>"]);
|
|
238
|
+
// Parsed defensively: a malformed entry is dropped, never a verdict change — evidence is diagnostics only.
|
|
239
|
+
const failureEvidence = failedCompletions.flatMap(([, c]) => Array.isArray(c.evidence) ? c.evidence.filter(isFailureEvidence) : []);
|
|
231
240
|
if (failures.length)
|
|
232
241
|
return { kind: "work", pass: false,
|
|
233
242
|
details: `test report names failing fingerprint(s):\n${failures.join("\n")}`,
|
|
234
|
-
meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode } };
|
|
243
|
+
meta: { classification: "regression", failingTests: failures, failingFiles, processExit: exitCode, failureEvidence } };
|
|
235
244
|
if (exitCode === undefined) {
|
|
236
245
|
return {
|
|
237
246
|
kind: "fail-closed",
|
|
@@ -322,6 +331,7 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
322
331
|
const pollMs = opts.pollMs ?? 20;
|
|
323
332
|
const overallCeilingMs = usable(opts.overallCeilingMs) ? opts.overallCeilingMs : DEFAULT_FILE_HANG_BUDGET_MS;
|
|
324
333
|
const env = { ...(opts.env ?? process.env), TICKMARKR_TEST_REPORT: opts.reportPath, TICKMARKR_TEST_NONCE: opts.nonce };
|
|
334
|
+
const evidence = beginGateEvidence(cwd, "test", cmd, { ...opts.evidence, env }, opts.nonce);
|
|
325
335
|
const controller = new AbortController();
|
|
326
336
|
let pid;
|
|
327
337
|
let killedFile;
|
|
@@ -347,6 +357,7 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
347
357
|
}
|
|
348
358
|
};
|
|
349
359
|
return shell(cmd, cwd, overallCeilingMs, false, {
|
|
360
|
+
onReceipt: receipt => evidence.observe(killedFile && receipt.outcome === "cancelled" ? { ...receipt, outcome: "timed-out" } : receipt),
|
|
350
361
|
env, signal: controller.signal, onTimeout: () => checkHang(true),
|
|
351
362
|
onSpawn: (childPid) => {
|
|
352
363
|
pid = childPid;
|
|
@@ -354,11 +365,21 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
354
365
|
clearInterval(poll);
|
|
355
366
|
poll = setInterval(checkHang, pollMs);
|
|
356
367
|
},
|
|
357
|
-
}).then((result) =>
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
368
|
+
}).then((result) => {
|
|
369
|
+
const evidenceReceipt = evidence.finish(result.stdout, result.stderr);
|
|
370
|
+
return {
|
|
371
|
+
evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt],
|
|
372
|
+
exitCode: result.signalExit ? undefined : result.code,
|
|
373
|
+
stdout: result.stdout, stderr: result.stderr,
|
|
374
|
+
report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
|
|
375
|
+
};
|
|
376
|
+
}).catch((error) => {
|
|
377
|
+
if (error instanceof Error) {
|
|
378
|
+
const evidenceReceipt = evidence.finish();
|
|
379
|
+
Object.assign(error, { evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt] });
|
|
380
|
+
}
|
|
381
|
+
throw error;
|
|
382
|
+
}).finally(() => { clearInterval(poll); });
|
|
362
383
|
}
|
|
363
384
|
/** The child environment every manifest invocation (listing and run) receives, and the lifecycle
|
|
364
385
|
* protocol it records. R41: the policy is whatever THIS process was launched with (npm reads
|
|
@@ -383,11 +404,12 @@ function manifestEnvironment(cwd) {
|
|
|
383
404
|
export async function discoverTestManifest(cmd, cwd, opts) {
|
|
384
405
|
const invocation = runnerInvocation(cmd, cwd);
|
|
385
406
|
const listed = await runManifestedTest(invocation.listing, cwd, {
|
|
386
|
-
manifest: [], nonce: opts.nonce
|
|
407
|
+
evidence: opts.evidence, manifest: [], nonce: `${opts.nonce}-listing`, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
|
|
387
408
|
overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
|
|
388
409
|
});
|
|
410
|
+
const discoveryError = (message) => Object.assign(new Error(message), { evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts });
|
|
389
411
|
if (listed.exitCode !== 0)
|
|
390
|
-
throw
|
|
412
|
+
throw discoveryError(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
|
|
391
413
|
let files;
|
|
392
414
|
try {
|
|
393
415
|
const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
|
|
@@ -396,11 +418,11 @@ export async function discoverTestManifest(cmd, cwd, opts) {
|
|
|
396
418
|
files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
|
|
397
419
|
}
|
|
398
420
|
catch {
|
|
399
|
-
throw
|
|
421
|
+
throw discoveryError(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
|
|
400
422
|
}
|
|
401
423
|
if (!files.length)
|
|
402
|
-
throw
|
|
403
|
-
return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout };
|
|
424
|
+
throw discoveryError("vitest cannot list files: empty manifest");
|
|
425
|
+
return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout, evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts };
|
|
404
426
|
}
|
|
405
427
|
/** The baseline capture's reading of the same seam: the manifest's file count, or null when the
|
|
406
428
|
* runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
|
|
@@ -422,11 +444,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
422
444
|
const nonce = randomBytes(16).toString("hex");
|
|
423
445
|
const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
|
|
424
446
|
const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
|
|
447
|
+
const evidenceReceipts = [];
|
|
425
448
|
let spawnedCommand = cmd;
|
|
426
449
|
let manifestPath;
|
|
427
450
|
const { env, verification } = manifestEnvironment(cwd);
|
|
428
451
|
try {
|
|
429
|
-
const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs });
|
|
452
|
+
const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
|
|
453
|
+
evidenceReceipts.push(...invocation.evidenceReceipts);
|
|
430
454
|
const files = invocation.files;
|
|
431
455
|
// R41: the EXPECTED manifest is evidence in its own right — persisted beside the report with the
|
|
432
456
|
// exact discovery invocation, so a later reader can tell what this invocation was asked to prove
|
|
@@ -440,7 +464,7 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
440
464
|
writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
|
|
441
465
|
spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
|
|
442
466
|
const invoked = await runManifestedTest(spawnedCommand, cwd, {
|
|
443
|
-
manifest: files, nonce, reportPath, env,
|
|
467
|
+
evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: files, nonce, reportPath, env,
|
|
444
468
|
baselineDurations: opts.baselineDurations?.map((d) => ({ ...d, file: toManifestPath(d.file, cwd) })),
|
|
445
469
|
longestFile: opts.longestFile,
|
|
446
470
|
overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
|
|
@@ -449,12 +473,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
449
473
|
const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
|
|
450
474
|
report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
|
|
451
475
|
// Preserve the validator's verdict and classification; runner evidence only explains it.
|
|
452
|
-
const
|
|
453
|
-
|
|
454
|
-
const
|
|
455
|
-
const
|
|
456
|
-
|
|
457
|
-
|
|
476
|
+
const evidenceReceipt = invoked.evidenceReceipt;
|
|
477
|
+
evidenceReceipts.push(...invoked.evidenceReceipts);
|
|
478
|
+
const evidenceRoot = opts.evidence?.artifactDir ?? dir;
|
|
479
|
+
const stdoutPath = join(evidenceRoot, evidenceReceipt.stdout.path);
|
|
480
|
+
const stderrPath = join(evidenceRoot, evidenceReceipt.stderr.path);
|
|
481
|
+
const stdoutTail = Buffer.from(redactGateOutput(invoked.stdout, env)).subarray(-16 * 1024);
|
|
482
|
+
const stderrTail = Buffer.from(redactGateOutput(invoked.stderr, env)).subarray(-16 * 1024);
|
|
458
483
|
const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
|
|
459
484
|
const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
|
|
460
485
|
const errors = report?.certificate?.errors;
|
|
@@ -462,11 +487,11 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
462
487
|
const reportedDiagnostics = report?.certificate?.diagnostics;
|
|
463
488
|
const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
|
|
464
489
|
const diagnostics = !verdict.pass
|
|
465
|
-
? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
|
|
490
|
+
? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}; runner vitest`
|
|
466
491
|
+ [...runnerErrors,
|
|
467
492
|
stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
|
|
468
493
|
: "";
|
|
469
|
-
return { pass: verdict.pass, kind: verdict.kind,
|
|
494
|
+
return { evidenceReceipt, evidenceReceipts, pass: verdict.pass, kind: verdict.kind,
|
|
470
495
|
details: verdict.details + diagnostics,
|
|
471
496
|
classification: verdict.meta.classification,
|
|
472
497
|
meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
|
|
@@ -474,7 +499,10 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
474
499
|
exitCode: invoked.exitCode ?? -1, reportPath };
|
|
475
500
|
}
|
|
476
501
|
catch (error) {
|
|
477
|
-
|
|
502
|
+
const evidenceReceipt = error?.evidenceReceipt;
|
|
503
|
+
if (evidenceReceipt)
|
|
504
|
+
evidenceReceipts.push(...(error.evidenceReceipts ?? [evidenceReceipt]));
|
|
505
|
+
return { evidenceReceipt, evidenceReceipts, pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
|
|
478
506
|
details: error instanceof Error ? error.message : String(error),
|
|
479
507
|
meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
|
|
480
508
|
}
|
|
@@ -11,6 +11,24 @@ export default class TickmarkrReporter {
|
|
|
11
11
|
this.cwd = realpathSync(process.cwd());
|
|
12
12
|
}
|
|
13
13
|
file(module) { return relative(this.cwd, module.moduleId).split(sep).join('/'); }
|
|
14
|
+
// Bounded assertion evidence, kept BESIDE the failure fingerprint and never inside it: the
|
|
15
|
+
// fingerprint is the failure's identity (the repeated-failure cap compares it), the evidence is
|
|
16
|
+
// what the runner knew and the message elided. ONE 4096-byte budget per failed test, shared by
|
|
17
|
+
// all of its errors (expect.soft yields several); never invented.
|
|
18
|
+
evidence(test, errors) {
|
|
19
|
+
const parts = [];
|
|
20
|
+
for (const e of errors) {
|
|
21
|
+
if (e && typeof e.diff === 'string' && e.diff) parts.push(e.diff);
|
|
22
|
+
else for (const k of ['actual', 'expected']) if (e && e[k] !== undefined && e[k] !== null) parts.push(k + ': ' + (typeof e[k] === 'string' ? e[k] : JSON.stringify(e[k])));
|
|
23
|
+
if (e && typeof e.stack === 'string' && e.stack) parts.push(e.stack.split('\n').slice(0, 8).join('\n'));
|
|
24
|
+
}
|
|
25
|
+
if (!parts.length) return { test, text: '', unavailable: true };
|
|
26
|
+
const full = parts.join('\n').replace(/\x1b\[[0-9;]*m/g, '');
|
|
27
|
+
if (Buffer.byteLength(full) <= 4096) return { test, text: full };
|
|
28
|
+
let text = full.slice(0, 4096);
|
|
29
|
+
while (Buffer.byteLength(text) > 4096) text = text.slice(0, -1);
|
|
30
|
+
return { test, text, truncated: true };
|
|
31
|
+
}
|
|
14
32
|
save() {
|
|
15
33
|
writeFileSync(this.path + '.tmp', JSON.stringify(this.report));
|
|
16
34
|
renameSync(this.path + '.tmp', this.path);
|
|
@@ -27,6 +45,7 @@ export default class TickmarkrReporter {
|
|
|
27
45
|
const file = this.file(module);
|
|
28
46
|
const failed = module.state() === 'failed';
|
|
29
47
|
const failures = [];
|
|
48
|
+
const evidence = [];
|
|
30
49
|
// R41: count test bodies by their own state so a module whose every test was skipped (a
|
|
31
50
|
// describe.skipIf gate) is recorded as SKIPPED — present in the lifecycle, but never as
|
|
32
51
|
// executed test-body success. 'passed'/'failed' executed; anything else did not run.
|
|
@@ -37,6 +56,8 @@ export default class TickmarkrReporter {
|
|
|
37
56
|
tests.failed++;
|
|
38
57
|
if (failed) {
|
|
39
58
|
const errors = test.result().errors || [];
|
|
59
|
+
const name = file + ' > ' + test.fullName;
|
|
60
|
+
evidence.push(this.evidence(name, errors));
|
|
40
61
|
failures.push(...(errors.length ? errors.map(e => 'FAIL ' + file + ' > ' + test.fullName + ': ' + e.message) : ['FAIL ' + file + ' > ' + test.fullName]));
|
|
41
62
|
}
|
|
42
63
|
} else if (state === 'passed') tests.passed++;
|
|
@@ -45,7 +66,7 @@ export default class TickmarkrReporter {
|
|
|
45
66
|
if (failed && !failures.length) failures.push('FAIL ' + file);
|
|
46
67
|
const status = failed ? 'failed' : tests.passed + tests.failed === 0 ? 'skipped' : 'passed';
|
|
47
68
|
if (file in this.report.completed) this.report.duplicateCompletions.push(file);
|
|
48
|
-
this.report.completed[file] = { at: Date.now(), status, failures, tests };
|
|
69
|
+
this.report.completed[file] = { at: Date.now(), status, failures, tests, ...(evidence.length ? { evidence } : {}) };
|
|
49
70
|
this.save();
|
|
50
71
|
}
|
|
51
72
|
onTestRunEnd(modules, errors, reason) {
|