tickmarkr 2.5.8 → 2.5.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/qwen.js +30 -3
- package/dist/cli/commands/approve.js +15 -3
- package/dist/cli/commands/beat.d.ts +2 -0
- package/dist/cli/commands/beat.js +28 -29
- package/dist/cli/commands/compile.js +11 -0
- package/dist/cli/commands/fleet.js +61 -53
- package/dist/cli/commands/verify.d.ts +4 -1
- package/dist/cli/commands/verify.js +16 -5
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +6 -4
- package/dist/config/config.d.ts +41 -3
- package/dist/config/config.js +48 -17
- package/dist/config/fleet-overlay.js +47 -62
- package/dist/gates/baseline.d.ts +65 -0
- package/dist/gates/baseline.js +163 -9
- package/dist/gates/review.d.ts +6 -4
- package/dist/gates/review.js +62 -21
- package/dist/gates/run-gates.d.ts +4 -1
- package/dist/gates/run-gates.js +21 -11
- package/dist/gates/test-manifest.d.ts +11 -1
- package/dist/gates/test-manifest.js +41 -21
- package/dist/run/daemon.js +84 -41
- package/dist/run/journal.d.ts +17 -2
- package/dist/run/journal.js +99 -11
- package/dist/run/merge.d.ts +13 -2
- package/dist/run/merge.js +74 -12
- package/dist/run/protocol.d.ts +82 -0
- package/dist/run/protocol.js +35 -0
- package/dist/run/receipt-resolver.d.ts +18 -0
- package/dist/run/receipt-resolver.js +132 -0
- package/dist/run/supervision.d.ts +14 -1
- package/dist/run/supervision.js +122 -24
- package/dist/tui/cockpit/evidence-view.d.ts +10 -1
- package/dist/tui/cockpit/evidence-view.js +37 -5
- package/dist/tui/ink/fleet-app.d.ts +12 -22
- package/dist/tui/ink/fleet-app.js +520 -131
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +61 -36
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
package/dist/gates/review.js
CHANGED
|
@@ -6,12 +6,12 @@ import { filesGlob } from "../graph/files-glob.js";
|
|
|
6
6
|
import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
|
-
import { structuredFindings } from "../run/journal.js";
|
|
9
|
+
import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
|
|
10
10
|
import { redactSecrets } from "../run/redact.js";
|
|
11
11
|
import { marginalCostRank } from "../route/router.js";
|
|
12
12
|
import { modelProvider } from "../route/preference.js";
|
|
13
13
|
import { resolveStateDir } from "./cache.js";
|
|
14
|
-
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
14
|
+
import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
|
|
15
15
|
import { classifyVerdictCause } from "./verdict-cause.js";
|
|
16
16
|
import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
|
|
17
17
|
export { isProtectedEvidence, PROTECTED_EVIDENCE_PREFIXES, REGENERABLE_CAPTURE_PATHS, setAsideReceiptPath, setAsideRegenerableCaptures, } from "./artifact-manifest.js";
|
|
@@ -188,15 +188,17 @@ export function matchClosureId(candidate, target) {
|
|
|
188
188
|
// OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
|
|
189
189
|
// it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
|
|
190
190
|
// `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
|
|
191
|
-
const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
|
|
192
191
|
if (typeof target === "string") {
|
|
193
|
-
return
|
|
192
|
+
return reviewFingerprintMatches(candidate, target);
|
|
194
193
|
}
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
}
|
|
199
|
-
|
|
194
|
+
const matches = [...target].filter((finding) => {
|
|
195
|
+
const ids = typeof finding === "string" ? [finding] : observedReviewFingerprints(finding);
|
|
196
|
+
return ids.some((id) => reviewFingerprintMatches(candidate, id));
|
|
197
|
+
});
|
|
198
|
+
if (matches.length !== 1)
|
|
199
|
+
return undefined;
|
|
200
|
+
const match = matches[0];
|
|
201
|
+
return typeof match === "string" ? match : match.fingerprint;
|
|
200
202
|
}
|
|
201
203
|
/**
|
|
202
204
|
* Validates closure ids in a review verdict: membership, duplication, and coverage of every prior id
|
|
@@ -205,10 +207,10 @@ export function matchClosureId(candidate, target) {
|
|
|
205
207
|
export function isReviewClosureInvalid(v, priorIds) {
|
|
206
208
|
const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
|
|
207
209
|
const closureLists = [v?.resolved, v?.reraised];
|
|
208
|
-
const allCandidateIds =
|
|
210
|
+
const allCandidateIds = closureLists.flatMap((list) => Array.isArray(list) ? list : []);
|
|
209
211
|
return !!v && (priors.size > 0 || closureLists.some((list) => list !== undefined)) && (closureLists.some((list) => !Array.isArray(list) || list.some((id) => !matchClosureId(id, priors)))
|
|
210
212
|
|| new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
|
|
211
|
-
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
|
|
213
|
+
|| [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, priors) === (typeof id === "string" ? id : id.fingerprint))));
|
|
212
214
|
}
|
|
213
215
|
/**
|
|
214
216
|
* OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
|
|
@@ -368,10 +370,17 @@ function daemonRepoRoot(worktree, artifactDir) {
|
|
|
368
370
|
* closure on a typo and that read as malformed — the block is what a closure list is copied from.
|
|
369
371
|
*/
|
|
370
372
|
export function renderPriorMaterials(priorMaterials) {
|
|
373
|
+
const fingerprints = priorMaterials.map((finding, i) => {
|
|
374
|
+
const observed = observedReviewFingerprints(finding);
|
|
375
|
+
const heading = observed.length > 1 ? `Finding ${i + 1} (choose one observed spelling):\n` : "";
|
|
376
|
+
return heading + observed.map((id) => `Fingerprint: ${id}`).join("\n");
|
|
377
|
+
}).join("\n");
|
|
371
378
|
return `## Prior materials this attempt must close
|
|
372
|
-
|
|
379
|
+
For each finding below, copy exactly ONE of its observed fingerprints into resolved or reraised.
|
|
380
|
+
Use only these observed spellings. For every findings entry that restates a reraised prior, whether at the same path or a new path, set its "reraised" field to the copied id; unrelated defects need separate entries.
|
|
381
|
+
The fingerprints appear once, in this block:
|
|
373
382
|
\`\`\`text
|
|
374
|
-
${
|
|
383
|
+
${fingerprints}
|
|
375
384
|
\`\`\`
|
|
376
385
|
${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
|
|
377
386
|
}
|
|
@@ -383,7 +392,7 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
|
|
|
383
392
|
// never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
|
|
384
393
|
priorReviewers = [],
|
|
385
394
|
// OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
|
|
386
|
-
carriedAuthors = []) {
|
|
395
|
+
carriedAuthors = [], operatorContext) {
|
|
387
396
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
388
397
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
389
398
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -502,6 +511,10 @@ ${suiteBudget} Never run the whole suite (including an unfiltered npm test or vi
|
|
|
502
511
|
|
|
503
512
|
${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
|
|
504
513
|
|
|
514
|
+
` : ""}${operatorContext?.trim() ? `## Operator context
|
|
515
|
+
Context only: this never substitutes for an acceptance criterion or closes a prior material.
|
|
516
|
+
${operatorContext.trim()}
|
|
517
|
+
|
|
505
518
|
` : ""}## Diff
|
|
506
519
|
\`\`\`diff
|
|
507
520
|
${diff}
|
|
@@ -578,7 +591,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
578
591
|
const provider = modelProvider(reviewer.model, reviewer.vendor);
|
|
579
592
|
const v = extractVerdictJson(raw, nonce);
|
|
580
593
|
const findings = v && Array.isArray(v.findings) ? v.findings : null;
|
|
581
|
-
const priorIds =
|
|
594
|
+
const priorIds = priorMaterials;
|
|
582
595
|
const closureInvalid = isReviewClosureInvalid(v, priorIds);
|
|
583
596
|
const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
|
|
584
597
|
// findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
|
|
@@ -610,7 +623,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
610
623
|
provider,
|
|
611
624
|
...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
|
|
612
625
|
cause,
|
|
613
|
-
...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints:
|
|
626
|
+
...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: priorIds.flatMap(observedReviewFingerprints) } : {}),
|
|
614
627
|
bytes, seatAuthoredBytes: bytes,
|
|
615
628
|
...(saved ? { rawPath: saved } : {}),
|
|
616
629
|
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
@@ -621,7 +634,8 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
621
634
|
const decided = findings !== null
|
|
622
635
|
? classifyReviewFindings(findings)
|
|
623
636
|
: classifyReviewIssues(v.approve, v.issues);
|
|
624
|
-
const
|
|
637
|
+
const ownLines = [...decided.lines];
|
|
638
|
+
const reraised = priorMaterials.filter((finding) => v.reraised?.some((id) => matchClosureId(id, [finding])));
|
|
625
639
|
if (reraised.length) {
|
|
626
640
|
if (decided.pass)
|
|
627
641
|
decided.headline = "requested changes";
|
|
@@ -636,6 +650,36 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
636
650
|
}
|
|
637
651
|
const prose = `reviewer ${reviewer.adapter}:${reviewer.model} (vendor: ${reviewer.vendor}; provider: ${provider}): ${decided.headline}${decided.lines.length ? "\n" + decided.lines.join("\n") : ""}`;
|
|
638
652
|
const details = appendAnchoredReview(prose, v);
|
|
653
|
+
// Only the verdict's anchors may supply missing evidence, and only when unambiguous.
|
|
654
|
+
// Reuse the journal's path normalization without changing legacy details-only parsing.
|
|
655
|
+
const anchoredPaths = new Set(parseAnchoredComments(v).map((comment) => {
|
|
656
|
+
const anchor = structuredFindings("review", `- ${comment.path}:${comment.line} — anchor`)
|
|
657
|
+
.find((finding) => finding.class === "review:anchored");
|
|
658
|
+
return anchor?.path ?? comment.path;
|
|
659
|
+
}));
|
|
660
|
+
const anchoredPath = anchoredPaths.size === 1 ? [...anchoredPaths][0] : undefined;
|
|
661
|
+
const ownDetails = appendAnchoredReview(ownLines.join("\n"), v);
|
|
662
|
+
const currentRows = (ownDetails.trim() ? structuredFindings("review", ownDetails) : [])
|
|
663
|
+
.map((finding) => finding.path === UNIDENTIFIED && anchoredPath
|
|
664
|
+
? { ...finding, path: anchoredPath, fingerprint: `${finding.class}|${anchoredPath}|${finding.symbol}` }
|
|
665
|
+
: finding);
|
|
666
|
+
// A top-level reraised id identifies a prior, not an arbitrary new defect. Bind a restatement
|
|
667
|
+
// only when its own entry echoes that validated id (or its note explicitly contains the id).
|
|
668
|
+
const linkedRows = currentRows.map((finding) => {
|
|
669
|
+
if (finding.class !== "review:material")
|
|
670
|
+
return finding;
|
|
671
|
+
const entry = v.findings?.find((entry) => entry && typeof entry === "object" && entry.note === finding.note);
|
|
672
|
+
const mentioned = reraised.filter((prior) => observedReviewFingerprints(prior)
|
|
673
|
+
.some((fp) => finding.note.includes(`\`${fp}\``) || finding.note.trim() === fp));
|
|
674
|
+
const id = matchClosureId(entry?.reraised, reraised)
|
|
675
|
+
?? (mentioned.length === 1 ? mentioned[0].fingerprint : undefined);
|
|
676
|
+
return id ? { ...finding, reraisedFrom: id } : finding;
|
|
677
|
+
});
|
|
678
|
+
// Several rows claiming the same prior are ambiguous; none gets to erase the others.
|
|
679
|
+
const unambiguousRows = linkedRows.map((finding) => finding.reraisedFrom
|
|
680
|
+
&& linkedRows.filter((row) => row.reraisedFrom === finding.reraisedFrom).length > 1
|
|
681
|
+
? { ...finding, reraisedFrom: undefined } : finding);
|
|
682
|
+
const carriedRows = carryReviewFindings(reraised, unambiguousRows);
|
|
639
683
|
return {
|
|
640
684
|
gate: "review",
|
|
641
685
|
pass: decided.pass,
|
|
@@ -649,10 +693,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
649
693
|
resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
650
694
|
reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
651
695
|
} : {}),
|
|
652
|
-
...(
|
|
653
|
-
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
654
|
-
...reraised,
|
|
655
|
-
] } : {}),
|
|
696
|
+
...(!decided.pass ? { findings: carriedRows } : {}),
|
|
656
697
|
...(saved ? { rawPath: saved } : {}),
|
|
657
698
|
...(savedBrief ? { briefPath: savedBrief } : {}),
|
|
658
699
|
},
|
|
@@ -2,7 +2,7 @@ import type { CommandReceiptAttribution } from "../run/protocol.js";
|
|
|
2
2
|
import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
|
|
3
3
|
import { type TickmarkrConfig } from "../config/config.js";
|
|
4
4
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
5
|
-
import { type Baseline } from "./baseline.js";
|
|
5
|
+
import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
|
|
6
6
|
import { type GateVia } from "./llm.js";
|
|
7
7
|
import { type PriorReviewer } from "./review.js";
|
|
8
8
|
import type { GateResult } from "./types.js";
|
|
@@ -46,6 +46,7 @@ export type GateEvent = {
|
|
|
46
46
|
result?: GateResult;
|
|
47
47
|
};
|
|
48
48
|
export interface GateContext {
|
|
49
|
+
evidence?: GateEvidenceOptions;
|
|
49
50
|
buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
|
|
50
51
|
authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
|
|
51
52
|
verificationScope?: VerificationScope;
|
|
@@ -61,6 +62,8 @@ export interface GateContext {
|
|
|
61
62
|
cfg: TickmarkrConfig;
|
|
62
63
|
via?: GateVia;
|
|
63
64
|
carriedFindings?: readonly StructuredFinding[];
|
|
65
|
+
/** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
|
|
66
|
+
operatorContext?: string;
|
|
64
67
|
excludeReviewers?: string[];
|
|
65
68
|
demotedReviewers?: Set<string>;
|
|
66
69
|
reviewNoVerdicts?: Map<string, string[]>;
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -222,22 +222,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
|
|
|
222
222
|
longestFile: entry?.longestFile,
|
|
223
223
|
overallCeilingMs: effectiveCeilingMs(entry),
|
|
224
224
|
artifactDir,
|
|
225
|
+
evidence: retry.evidence,
|
|
225
226
|
});
|
|
226
227
|
const reportPath = outcome.reportPath;
|
|
228
|
+
const evidence = { evidenceReceipt: outcome.evidenceReceipt, evidenceReceipts: outcome.evidenceReceipts };
|
|
227
229
|
if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
|
|
228
230
|
&& outcome.meta?.retryable !== false) {
|
|
229
231
|
const waitedMs = await waitForCalmWindow(executionSignal());
|
|
230
232
|
if (!calmWindowReady())
|
|
231
|
-
return { gate: "test", pass: false, details: outcome.details,
|
|
233
|
+
return { ...evidence, gate: "test", pass: false, details: outcome.details,
|
|
232
234
|
meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
|
|
233
235
|
if (!retry.authorizeRetry("infra")) {
|
|
234
|
-
return { gate: "test", pass: false, details: outcome.details,
|
|
236
|
+
return { ...evidence, gate: "test", pass: false, details: outcome.details,
|
|
235
237
|
meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
|
|
236
238
|
}
|
|
237
239
|
const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
|
|
238
|
-
return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
|
|
240
|
+
return { ...result, evidenceReceipts: [...(outcome.evidenceReceipts ?? []), ...(result.evidenceReceipts ?? [])], meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
|
|
239
241
|
}
|
|
240
242
|
return {
|
|
243
|
+
...evidence,
|
|
241
244
|
gate: "test",
|
|
242
245
|
pass: outcome.pass,
|
|
243
246
|
details: outcome.details,
|
|
@@ -259,6 +262,12 @@ function classifySignalOnlyTest(g) {
|
|
|
259
262
|
}
|
|
260
263
|
export async function runGates(task, ctx) {
|
|
261
264
|
const results = [];
|
|
265
|
+
const evidence = {
|
|
266
|
+
artifactDir: ctx.artifactDir,
|
|
267
|
+
runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
|
|
268
|
+
taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
|
|
269
|
+
...ctx.evidence,
|
|
270
|
+
};
|
|
262
271
|
// Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
|
|
263
272
|
// allocates a new invocation, including retries whose local spawn counter starts at one again.
|
|
264
273
|
let currentBuild;
|
|
@@ -658,8 +667,8 @@ export async function runGates(task, ctx) {
|
|
|
658
667
|
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
659
668
|
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
660
669
|
r = useManifest
|
|
661
|
-
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
|
|
662
|
-
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
670
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence }))
|
|
671
|
+
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), evidence, ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
|
|
663
672
|
}
|
|
664
673
|
finally {
|
|
665
674
|
await receiptNotes;
|
|
@@ -676,7 +685,7 @@ export async function runGates(task, ctx) {
|
|
|
676
685
|
if (!cached && r.pass && commands[g]) {
|
|
677
686
|
const dirt = await dirtyWorktree();
|
|
678
687
|
if (dirt) {
|
|
679
|
-
await record(await dirtyRefusal(g, dirt, commands[g]));
|
|
688
|
+
await record({ ...await dirtyRefusal(g, dirt, commands[g]), evidenceReceipt: r.evidenceReceipt, evidenceReceipts: r.evidenceReceipts });
|
|
680
689
|
return;
|
|
681
690
|
}
|
|
682
691
|
}
|
|
@@ -903,7 +912,7 @@ export async function runGates(task, ctx) {
|
|
|
903
912
|
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
904
913
|
const carriedAuthors = ctx.carriedAuthors ?? [];
|
|
905
914
|
let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
|
|
906
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
|
|
915
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
|
|
907
916
|
// OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
|
|
908
917
|
// a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
|
|
909
918
|
// verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -936,7 +945,7 @@ export async function runGates(task, ctx) {
|
|
|
936
945
|
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
|
|
937
946
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
938
947
|
exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
939
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
|
|
948
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
|
|
940
949
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
941
950
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
942
951
|
const route = exclusion === "adapter"
|
|
@@ -1104,8 +1113,8 @@ export async function runGates(task, ctx) {
|
|
|
1104
1113
|
if (!full) {
|
|
1105
1114
|
const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
|
|
1106
1115
|
full = fullUsesManifest
|
|
1107
|
-
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
|
|
1108
|
-
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
|
|
1116
|
+
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, { ...retryOptions(identity), evidence }))
|
|
1117
|
+
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], { ...retryOptions(identity), evidence })))[0];
|
|
1109
1118
|
}
|
|
1110
1119
|
fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
|
|
1111
1120
|
const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
|
|
@@ -1115,8 +1124,9 @@ export async function runGates(task, ctx) {
|
|
|
1115
1124
|
verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
1116
1125
|
}
|
|
1117
1126
|
const merged = withTelemetry(dirt
|
|
1118
|
-
? await dirtyRefusal("test", dirt, ctx.commands.test)
|
|
1127
|
+
? { ...await dirtyRefusal("test", dirt, ctx.commands.test), evidenceReceipt: full.evidenceReceipt, evidenceReceipts: full.evidenceReceipts }
|
|
1119
1128
|
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
|
|
1129
|
+
merged.evidenceReceipts = [...(heldTest?.evidenceReceipts ?? []), ...(full?.evidenceReceipts ?? [])];
|
|
1120
1130
|
results[results.findIndex((r) => r.gate === "test")] = merged;
|
|
1121
1131
|
heldTest = undefined;
|
|
1122
1132
|
await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
|
|
@@ -1,4 +1,5 @@
|
|
|
1
|
-
import type
|
|
1
|
+
import { type GateEvidenceOptions, type BaselineFileDuration } from "./baseline.js";
|
|
2
|
+
import type { GateEvidenceReceipt } from "../run/protocol.js";
|
|
2
3
|
export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
|
|
3
4
|
/** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
|
|
4
5
|
* --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
|
|
@@ -78,6 +79,8 @@ export declare const FILE_HANG_SLACK = 3;
|
|
|
78
79
|
export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
|
|
79
80
|
export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
|
|
80
81
|
export interface ManifestRunResult {
|
|
82
|
+
evidenceReceipt: GateEvidenceReceipt;
|
|
83
|
+
evidenceReceipts: GateEvidenceReceipt[];
|
|
81
84
|
exitCode: number | undefined;
|
|
82
85
|
stdout: string;
|
|
83
86
|
stderr: string;
|
|
@@ -91,6 +94,7 @@ export interface ManifestRunResult {
|
|
|
91
94
|
/** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
|
|
92
95
|
* timeout kills the detached process group, including descendants holding the output pipes. */
|
|
93
96
|
export declare function runManifestedTest(cmd: string, cwd: string, opts: {
|
|
97
|
+
evidence?: GateEvidenceOptions;
|
|
94
98
|
manifest: readonly string[];
|
|
95
99
|
nonce: string;
|
|
96
100
|
reportPath: string;
|
|
@@ -101,6 +105,8 @@ export declare function runManifestedTest(cmd: string, cwd: string, opts: {
|
|
|
101
105
|
overallCeilingMs?: number;
|
|
102
106
|
}): Promise<ManifestRunResult>;
|
|
103
107
|
export interface ManifestGateOutcome {
|
|
108
|
+
evidenceReceipt?: GateEvidenceReceipt;
|
|
109
|
+
evidenceReceipts?: GateEvidenceReceipt[];
|
|
104
110
|
pass: boolean;
|
|
105
111
|
kind: ManifestVerdictKind;
|
|
106
112
|
details: string;
|
|
@@ -115,6 +121,8 @@ export interface DiscoveredManifest {
|
|
|
115
121
|
separator: string;
|
|
116
122
|
listingExit: number | undefined;
|
|
117
123
|
listingStdout: string;
|
|
124
|
+
evidenceReceipt: GateEvidenceReceipt;
|
|
125
|
+
evidenceReceipts: GateEvidenceReceipt[];
|
|
118
126
|
}
|
|
119
127
|
/** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
|
|
120
128
|
* from its own listing and never from a stdout summary. The gate and the baseline capture share it,
|
|
@@ -125,6 +133,7 @@ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
|
|
|
125
133
|
nonce: string;
|
|
126
134
|
env: NodeJS.ProcessEnv;
|
|
127
135
|
overallCeilingMs?: number;
|
|
136
|
+
evidence?: GateEvidenceOptions;
|
|
128
137
|
}): Promise<DiscoveredManifest>;
|
|
129
138
|
/** The baseline capture's reading of the same seam: the manifest's file count, or null when the
|
|
130
139
|
* runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
|
|
@@ -137,4 +146,5 @@ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
|
|
|
137
146
|
longestFile?: BaselineFileDuration | null;
|
|
138
147
|
overallCeilingMs?: number;
|
|
139
148
|
artifactDir?: string;
|
|
149
|
+
evidence?: GateEvidenceOptions;
|
|
140
150
|
}): Promise<ManifestGateOutcome>;
|
|
@@ -4,6 +4,7 @@ import { tmpdir } from "node:os";
|
|
|
4
4
|
import { isAbsolute, join, relative, sep } from "node:path";
|
|
5
5
|
import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
|
|
6
6
|
import { shq } from "../adapters/types.js";
|
|
7
|
+
import { beginGateEvidence, redactGateOutput } from "./baseline.js";
|
|
7
8
|
import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
8
9
|
/**
|
|
9
10
|
* VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
|
|
@@ -330,6 +331,7 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
330
331
|
const pollMs = opts.pollMs ?? 20;
|
|
331
332
|
const overallCeilingMs = usable(opts.overallCeilingMs) ? opts.overallCeilingMs : DEFAULT_FILE_HANG_BUDGET_MS;
|
|
332
333
|
const env = { ...(opts.env ?? process.env), TICKMARKR_TEST_REPORT: opts.reportPath, TICKMARKR_TEST_NONCE: opts.nonce };
|
|
334
|
+
const evidence = beginGateEvidence(cwd, "test", cmd, { ...opts.evidence, env }, opts.nonce);
|
|
333
335
|
const controller = new AbortController();
|
|
334
336
|
let pid;
|
|
335
337
|
let killedFile;
|
|
@@ -355,6 +357,7 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
355
357
|
}
|
|
356
358
|
};
|
|
357
359
|
return shell(cmd, cwd, overallCeilingMs, false, {
|
|
360
|
+
onReceipt: receipt => evidence.observe(killedFile && receipt.outcome === "cancelled" ? { ...receipt, outcome: "timed-out" } : receipt),
|
|
358
361
|
env, signal: controller.signal, onTimeout: () => checkHang(true),
|
|
359
362
|
onSpawn: (childPid) => {
|
|
360
363
|
pid = childPid;
|
|
@@ -362,11 +365,21 @@ export function runManifestedTest(cmd, cwd, opts) {
|
|
|
362
365
|
clearInterval(poll);
|
|
363
366
|
poll = setInterval(checkHang, pollMs);
|
|
364
367
|
},
|
|
365
|
-
}).then((result) =>
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
368
|
+
}).then((result) => {
|
|
369
|
+
const evidenceReceipt = evidence.finish(result.stdout, result.stderr);
|
|
370
|
+
return {
|
|
371
|
+
evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt],
|
|
372
|
+
exitCode: result.signalExit ? undefined : result.code,
|
|
373
|
+
stdout: result.stdout, stderr: result.stderr,
|
|
374
|
+
report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
|
|
375
|
+
};
|
|
376
|
+
}).catch((error) => {
|
|
377
|
+
if (error instanceof Error) {
|
|
378
|
+
const evidenceReceipt = evidence.finish();
|
|
379
|
+
Object.assign(error, { evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt] });
|
|
380
|
+
}
|
|
381
|
+
throw error;
|
|
382
|
+
}).finally(() => { clearInterval(poll); });
|
|
370
383
|
}
|
|
371
384
|
/** The child environment every manifest invocation (listing and run) receives, and the lifecycle
|
|
372
385
|
* protocol it records. R41: the policy is whatever THIS process was launched with (npm reads
|
|
@@ -391,11 +404,12 @@ function manifestEnvironment(cwd) {
|
|
|
391
404
|
export async function discoverTestManifest(cmd, cwd, opts) {
|
|
392
405
|
const invocation = runnerInvocation(cmd, cwd);
|
|
393
406
|
const listed = await runManifestedTest(invocation.listing, cwd, {
|
|
394
|
-
manifest: [], nonce: opts.nonce
|
|
407
|
+
evidence: opts.evidence, manifest: [], nonce: `${opts.nonce}-listing`, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
|
|
395
408
|
overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
|
|
396
409
|
});
|
|
410
|
+
const discoveryError = (message) => Object.assign(new Error(message), { evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts });
|
|
397
411
|
if (listed.exitCode !== 0)
|
|
398
|
-
throw
|
|
412
|
+
throw discoveryError(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
|
|
399
413
|
let files;
|
|
400
414
|
try {
|
|
401
415
|
const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
|
|
@@ -404,11 +418,11 @@ export async function discoverTestManifest(cmd, cwd, opts) {
|
|
|
404
418
|
files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
|
|
405
419
|
}
|
|
406
420
|
catch {
|
|
407
|
-
throw
|
|
421
|
+
throw discoveryError(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
|
|
408
422
|
}
|
|
409
423
|
if (!files.length)
|
|
410
|
-
throw
|
|
411
|
-
return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout };
|
|
424
|
+
throw discoveryError("vitest cannot list files: empty manifest");
|
|
425
|
+
return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout, evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts };
|
|
412
426
|
}
|
|
413
427
|
/** The baseline capture's reading of the same seam: the manifest's file count, or null when the
|
|
414
428
|
* runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
|
|
@@ -430,11 +444,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
430
444
|
const nonce = randomBytes(16).toString("hex");
|
|
431
445
|
const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
|
|
432
446
|
const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
|
|
447
|
+
const evidenceReceipts = [];
|
|
433
448
|
let spawnedCommand = cmd;
|
|
434
449
|
let manifestPath;
|
|
435
450
|
const { env, verification } = manifestEnvironment(cwd);
|
|
436
451
|
try {
|
|
437
|
-
const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs });
|
|
452
|
+
const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
|
|
453
|
+
evidenceReceipts.push(...invocation.evidenceReceipts);
|
|
438
454
|
const files = invocation.files;
|
|
439
455
|
// R41: the EXPECTED manifest is evidence in its own right — persisted beside the report with the
|
|
440
456
|
// exact discovery invocation, so a later reader can tell what this invocation was asked to prove
|
|
@@ -448,7 +464,7 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
448
464
|
writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
|
|
449
465
|
spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
|
|
450
466
|
const invoked = await runManifestedTest(spawnedCommand, cwd, {
|
|
451
|
-
manifest: files, nonce, reportPath, env,
|
|
467
|
+
evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: files, nonce, reportPath, env,
|
|
452
468
|
baselineDurations: opts.baselineDurations?.map((d) => ({ ...d, file: toManifestPath(d.file, cwd) })),
|
|
453
469
|
longestFile: opts.longestFile,
|
|
454
470
|
overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
|
|
@@ -457,12 +473,13 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
457
473
|
const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
|
|
458
474
|
report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
|
|
459
475
|
// Preserve the validator's verdict and classification; runner evidence only explains it.
|
|
460
|
-
const
|
|
461
|
-
|
|
462
|
-
const
|
|
463
|
-
const
|
|
464
|
-
|
|
465
|
-
|
|
476
|
+
const evidenceReceipt = invoked.evidenceReceipt;
|
|
477
|
+
evidenceReceipts.push(...invoked.evidenceReceipts);
|
|
478
|
+
const evidenceRoot = opts.evidence?.artifactDir ?? dir;
|
|
479
|
+
const stdoutPath = join(evidenceRoot, evidenceReceipt.stdout.path);
|
|
480
|
+
const stderrPath = join(evidenceRoot, evidenceReceipt.stderr.path);
|
|
481
|
+
const stdoutTail = Buffer.from(redactGateOutput(invoked.stdout, env)).subarray(-16 * 1024);
|
|
482
|
+
const stderrTail = Buffer.from(redactGateOutput(invoked.stderr, env)).subarray(-16 * 1024);
|
|
466
483
|
const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
|
|
467
484
|
const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
|
|
468
485
|
const errors = report?.certificate?.errors;
|
|
@@ -470,11 +487,11 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
470
487
|
const reportedDiagnostics = report?.certificate?.diagnostics;
|
|
471
488
|
const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
|
|
472
489
|
const diagnostics = !verdict.pass
|
|
473
|
-
? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
|
|
490
|
+
? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}; runner vitest`
|
|
474
491
|
+ [...runnerErrors,
|
|
475
492
|
stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
|
|
476
493
|
: "";
|
|
477
|
-
return { pass: verdict.pass, kind: verdict.kind,
|
|
494
|
+
return { evidenceReceipt, evidenceReceipts, pass: verdict.pass, kind: verdict.kind,
|
|
478
495
|
details: verdict.details + diagnostics,
|
|
479
496
|
classification: verdict.meta.classification,
|
|
480
497
|
meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
|
|
@@ -482,7 +499,10 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
|
|
|
482
499
|
exitCode: invoked.exitCode ?? -1, reportPath };
|
|
483
500
|
}
|
|
484
501
|
catch (error) {
|
|
485
|
-
|
|
502
|
+
const evidenceReceipt = error?.evidenceReceipt;
|
|
503
|
+
if (evidenceReceipt)
|
|
504
|
+
evidenceReceipts.push(...(error.evidenceReceipts ?? [evidenceReceipt]));
|
|
505
|
+
return { evidenceReceipt, evidenceReceipts, pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
|
|
486
506
|
details: error instanceof Error ? error.message : String(error),
|
|
487
507
|
meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
|
|
488
508
|
}
|