tickmarkr 2.5.3 → 2.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/registry.js +6 -1
- package/dist/adapters/types.d.ts +3 -0
- package/dist/cli/commands/approve.d.ts +1 -0
- package/dist/cli/commands/approve.js +54 -6
- package/dist/cli/commands/doctor.d.ts +4 -0
- package/dist/cli/commands/doctor.js +44 -29
- package/dist/cli/commands/fleet.js +195 -33
- package/dist/cli/commands/init.js +196 -6
- package/dist/cli/commands/plan.js +144 -3
- package/dist/cli/commands/resume.js +6 -2
- package/dist/cli/commands/run.js +21 -2
- package/dist/cli/commands/verify.js +1 -1
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +3 -1
- package/dist/compile/collateral.d.ts +2 -0
- package/dist/compile/collateral.js +50 -9
- package/dist/compile/ownership.d.ts +7 -0
- package/dist/compile/ownership.js +59 -22
- package/dist/config/config.d.ts +23 -8
- package/dist/config/config.js +42 -28
- package/dist/config/fleet-overlay.d.ts +3 -9
- package/dist/config/fleet-overlay.js +68 -11
- package/dist/config/fleet-why.d.ts +7 -0
- package/dist/config/fleet-why.js +5 -0
- package/dist/drivers/herdr.d.ts +5 -0
- package/dist/drivers/herdr.js +14 -3
- package/dist/drivers/index.d.ts +15 -1
- package/dist/drivers/index.js +38 -10
- package/dist/drivers/orca.d.ts +99 -10
- package/dist/drivers/orca.js +586 -97
- package/dist/drivers/types.d.ts +1 -0
- package/dist/gates/baseline.d.ts +3 -0
- package/dist/gates/baseline.js +2 -1
- package/dist/gates/cache.d.ts +100 -0
- package/dist/gates/cache.js +389 -0
- package/dist/gates/review.d.ts +35 -2
- package/dist/gates/review.js +64 -16
- package/dist/gates/run-gates.d.ts +5 -0
- package/dist/gates/run-gates.js +134 -18
- package/dist/gates/test-manifest.d.ts +99 -0
- package/dist/gates/test-manifest.js +389 -0
- package/dist/gates/test-reporter.d.ts +4 -0
- package/dist/gates/test-reporter.js +49 -0
- package/dist/route/preference.d.ts +22 -1
- package/dist/route/preference.js +123 -25
- package/dist/route/router.d.ts +13 -0
- package/dist/route/router.js +95 -17
- package/dist/run/daemon.d.ts +13 -0
- package/dist/run/daemon.js +398 -96
- package/dist/run/git.d.ts +35 -1
- package/dist/run/git.js +111 -10
- package/dist/run/journal.d.ts +14 -2
- package/dist/run/journal.js +78 -10
- package/dist/run/lease.d.ts +14 -0
- package/dist/run/lease.js +87 -0
- package/dist/run/merge.d.ts +2 -0
- package/dist/run/merge.js +91 -3
- package/dist/run/operator-state.d.ts +11 -0
- package/dist/run/operator-state.js +17 -3
- package/dist/tui/cockpit/board.d.ts +96 -0
- package/dist/tui/cockpit/board.js +346 -0
- package/dist/tui/cockpit/decision-actions.js +2 -0
- package/dist/tui/cockpit/layout.d.ts +5 -1
- package/dist/tui/cockpit/layout.js +8 -3
- package/dist/tui/cockpit/live-runtime.js +83 -31
- package/dist/tui/cockpit/run-view.d.ts +7 -5
- package/dist/tui/cockpit/run-view.js +12 -11
- package/dist/tui/ink/fleet-app.d.ts +41 -27
- package/dist/tui/ink/fleet-app.js +204 -31
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +180 -99
package/dist/gates/review.js
CHANGED
|
@@ -3,7 +3,7 @@ import { join } from "node:path";
|
|
|
3
3
|
import { channelKey, shq } from "../adapters/types.js";
|
|
4
4
|
import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
|
|
5
5
|
import { filesGlob } from "../graph/files-glob.js";
|
|
6
|
-
import { renderAcceptanceItem } from "../graph/schema.js";
|
|
6
|
+
import { renderAcceptanceItem, TIERS } from "../graph/schema.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { shOk } from "../run/git.js";
|
|
9
9
|
import { structuredFindings } from "../run/journal.js";
|
|
@@ -200,9 +200,52 @@ function reviewPreferIndex(c, prefer) {
|
|
|
200
200
|
const i = prefer.findIndex((p) => p === c.adapter || p === channelKey(c));
|
|
201
201
|
return i === -1 ? prefer.length : i;
|
|
202
202
|
}
|
|
203
|
+
/**
|
|
204
|
+
* RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
|
|
205
|
+
* task-declared floor, a configured `review.floor` tier and, on a second round or a retry, the prior
|
|
206
|
+
* reviewer's tier. The cause names the input that reached the maximum (earlier inputs win a tie, so a
|
|
207
|
+
* floor the author's tier already satisfies is attributed to the author).
|
|
208
|
+
*/
|
|
209
|
+
export function resolveReviewerFloor(authorTier, taskFloor, configFloor, priorReviewerTier) {
|
|
210
|
+
const inputs = [
|
|
211
|
+
[authorTier, "author-tier"], [taskFloor, "task-floor"], [configFloor, "config"], [priorReviewerTier, "prior-reviewer"],
|
|
212
|
+
];
|
|
213
|
+
let best = { floor: authorTier, cause: "author-tier" };
|
|
214
|
+
for (const [tier, cause] of inputs) {
|
|
215
|
+
if (tier !== undefined && TIER_RANK[tier] > TIER_RANK[best.floor])
|
|
216
|
+
best = { floor: tier, cause };
|
|
217
|
+
}
|
|
218
|
+
return best;
|
|
219
|
+
}
|
|
220
|
+
/**
|
|
221
|
+
* The highest tier among the task's prior reviewers — the prior reviewer's tier for RF-1. A recorded
|
|
222
|
+
* dispatch tier is historical evidence and wins over the current pool; an unrecorded one falls back to
|
|
223
|
+
* the seat's channel; a seat neither establishes (it left the pool on resume, or the journal holds
|
|
224
|
+
* garbage) holds frontier — fail closed, never silently dropped.
|
|
225
|
+
*/
|
|
226
|
+
export function priorReviewerTier(channels, priorReviewers = []) {
|
|
227
|
+
let top;
|
|
228
|
+
for (const p of priorReviewers) {
|
|
229
|
+
const key = typeof p === "string" ? p : p.reviewer;
|
|
230
|
+
const seen = (typeof p === "string" ? undefined : p.tier) ?? channels.find((ch) => channelKey(ch) === key)?.tier;
|
|
231
|
+
const tier = TIERS.includes(seen) ? seen : "frontier";
|
|
232
|
+
if (top === undefined || TIER_RANK[tier] > TIER_RANK[top])
|
|
233
|
+
top = tier;
|
|
234
|
+
}
|
|
235
|
+
return top;
|
|
236
|
+
}
|
|
237
|
+
/**
|
|
238
|
+
* The gate's floor: author tier, task floor, review.floor (a tier — `worker` names none) and the seats
|
|
239
|
+
* the caller names as THIS TASK's prior reviewers (earlier rounds' seats, a flaked seat). Eligibility
|
|
240
|
+
* exclusions are NOT evidence — a retry bans a flaked seat's whole adapter, and those sibling channels
|
|
241
|
+
* never reviewed — and neither is the run-scoped LRU rotation history, which names unrelated tasks' seats.
|
|
242
|
+
*/
|
|
243
|
+
export function gateReviewerFloor(task, cfg, author, channels, priorReviewers = []) {
|
|
244
|
+
return resolveReviewerFloor(author.tier, task.routingHints?.floor, cfg.review.floor === "worker" ? undefined : cfg.review.floor, priorReviewerTier(channels, priorReviewers));
|
|
245
|
+
}
|
|
203
246
|
export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
204
247
|
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
205
|
-
floor, // task
|
|
248
|
+
floor, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
|
|
206
249
|
history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
|
|
207
250
|
onSeat, demoted = new Set()) {
|
|
208
251
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
@@ -213,6 +256,8 @@ onSeat, demoted = new Set()) {
|
|
|
213
256
|
if (!authorChannel)
|
|
214
257
|
return null;
|
|
215
258
|
const authorProvider = modelProvider(author.model, authorChannel.vendor);
|
|
259
|
+
// RF-1: every caller inherits the author-tier floor — a reviewer is never seated below its author.
|
|
260
|
+
const effectiveFloor = resolveReviewerFloor(author.tier, floor).floor;
|
|
216
261
|
const ranked = channels
|
|
217
262
|
// Three independent axes: different vendor, different resolved provider identity (OBS-946: on initial pick
|
|
218
263
|
// as well as failover, so an aggregator channel stamped "mixed" never seats the author's own provider),
|
|
@@ -222,7 +267,7 @@ onSeat, demoted = new Set()) {
|
|
|
222
267
|
&& modelProvider(c.model, c.vendor) !== authorProvider
|
|
223
268
|
&& modelId(c.model) !== modelId(author.model)
|
|
224
269
|
&& !exclude.includes(channelKey(c))
|
|
225
|
-
&&
|
|
270
|
+
&& TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
|
|
226
271
|
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
|
|
227
272
|
const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
|
|
228
273
|
|| history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
|
|
@@ -247,7 +292,10 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
|
|
|
247
292
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
248
293
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
249
294
|
// direct tests) skips persistence and changes nothing else.
|
|
250
|
-
artifactDir, reviewHistory, demotedReviewers, carriedFindings = []
|
|
295
|
+
artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
|
|
296
|
+
// RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
|
|
297
|
+
// never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
|
|
298
|
+
priorReviewers = []) {
|
|
251
299
|
// R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
|
|
252
300
|
// files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
|
|
253
301
|
// retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
|
|
@@ -317,20 +365,19 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = []) {
|
|
|
317
365
|
policy: "full",
|
|
318
366
|
...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
|
|
319
367
|
};
|
|
320
|
-
//
|
|
321
|
-
//
|
|
322
|
-
const reviewerFloor = task
|
|
368
|
+
// RF-1: the floor is max(author tier, task floor, review.floor tier, prior reviewer's tier). Only
|
|
369
|
+
// review.floor is read from config — cfg.routing.floors governs workers and never moves review seats.
|
|
370
|
+
const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
|
|
371
|
+
const floorMeta = { reviewerFloor, reviewerFloorCause };
|
|
323
372
|
let rotationSeat;
|
|
324
373
|
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
|
|
325
374
|
if (!reviewer) {
|
|
326
375
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
327
376
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
328
|
-
const reason = reviewerFloor
|
|
329
|
-
? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
|
|
330
|
-
: "no cross-vendor reviewer available (diversity rule)";
|
|
377
|
+
const reason = `no cross-vendor reviewer available at or above ${reviewerFloor} floor (${reviewerFloorCause}; diversity rule)`;
|
|
331
378
|
return cfg.review.required || priorMaterials.length > 0
|
|
332
|
-
? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...
|
|
333
|
-
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...
|
|
379
|
+
? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...floorMeta } }
|
|
380
|
+
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...floorMeta } };
|
|
334
381
|
}
|
|
335
382
|
reviewHistory?.push(channelKey(reviewer));
|
|
336
383
|
const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
|
|
@@ -369,6 +416,7 @@ Classify every concern as "material" (a correctness, security, or acceptance-cri
|
|
|
369
416
|
block the merge) or "minor" (style, naming, or preference that should not block). ONLY material findings
|
|
370
417
|
block approval. For a minor concern you have decided not to block on, set "defer": true and give a
|
|
371
418
|
one-line "rationale" — it is recorded in the review, never dropped.
|
|
419
|
+
A fix you prescribe that would break suites outside the task's declared write scope (files[]) is a scope finding, never a material one.
|
|
372
420
|
|
|
373
421
|
Respond with ONLY this JSON:
|
|
374
422
|
{"nonce": "${nonce}", "approve": true|false, "resolved": [], "reraised": [], "findings": [{"note": "...", "severity": "material"|"minor", "defer": false, "rationale": ""}], "comments": [{"path": "path/to/file", "line": 42, "body": "actionable feedback"}]}
|
|
@@ -453,7 +501,9 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
453
501
|
meta: {
|
|
454
502
|
...policyMeta,
|
|
455
503
|
...rotationMeta,
|
|
504
|
+
...floorMeta,
|
|
456
505
|
reviewer: channelKey(reviewer),
|
|
506
|
+
reviewerTier: reviewer.tier,
|
|
457
507
|
vendor: reviewer.vendor,
|
|
458
508
|
provider,
|
|
459
509
|
...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
|
|
@@ -488,15 +538,13 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
488
538
|
pass: decided.pass,
|
|
489
539
|
details,
|
|
490
540
|
meta: {
|
|
491
|
-
...policyMeta, ...rotationMeta, reviewer: channelKey(reviewer), vendor: reviewer.vendor, provider,
|
|
541
|
+
...policyMeta, ...rotationMeta, ...floorMeta, reviewer: channelKey(reviewer), reviewerTier: reviewer.tier, vendor: reviewer.vendor, provider,
|
|
542
|
+
// OBS-990 b: the verbatim ids plus ONE normalised copy of each list — never a third alias.
|
|
492
543
|
...(priorMaterials.length ? {
|
|
493
544
|
resolved: v.resolved,
|
|
494
545
|
reraised: v.reraised,
|
|
495
|
-
normalisedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
496
546
|
resolvedMatches: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
497
547
|
reraisedMatches: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
498
|
-
normalisedResolved: (v.resolved ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
499
|
-
normalisedReraised: (v.reraised ?? []).map((id) => matchClosureId(id, priorIds)).filter((id) => id !== undefined),
|
|
500
548
|
} : {}),
|
|
501
549
|
...(reraised.length ? { findings: [
|
|
502
550
|
...structuredFindings("review", details).filter((finding) => !reraised.some((prior) => prior.note === finding.note)),
|
|
@@ -3,8 +3,10 @@ import { type TickmarkrConfig } from "../config/config.js";
|
|
|
3
3
|
import { type GateName, type Task } from "../graph/schema.js";
|
|
4
4
|
import { type Baseline } from "./baseline.js";
|
|
5
5
|
import { type GateVia } from "./llm.js";
|
|
6
|
+
import { type PriorReviewer } from "./review.js";
|
|
6
7
|
import type { GateResult } from "./types.js";
|
|
7
8
|
import { type StructuredFinding } from "../run/journal.js";
|
|
9
|
+
import { type VerificationScope } from "./cache.js";
|
|
8
10
|
export type LoadProvider = () => number;
|
|
9
11
|
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
10
12
|
export declare function setLoadProviderForTests(provider: LoadProvider): void;
|
|
@@ -42,6 +44,7 @@ export type GateEvent = {
|
|
|
42
44
|
result: GateResult;
|
|
43
45
|
};
|
|
44
46
|
export interface GateContext {
|
|
47
|
+
verificationScope?: VerificationScope;
|
|
45
48
|
worktree: string;
|
|
46
49
|
baseRef: string;
|
|
47
50
|
result: WorkerResult;
|
|
@@ -57,11 +60,13 @@ export interface GateContext {
|
|
|
57
60
|
excludeReviewers?: string[];
|
|
58
61
|
demotedReviewers?: Set<string>;
|
|
59
62
|
reviewHistory?: string[];
|
|
63
|
+
priorReviewers?: PriorReviewer[];
|
|
60
64
|
artifactDir?: string;
|
|
61
65
|
pipeline?: "v185" | "legacy";
|
|
62
66
|
selectTests?: boolean;
|
|
63
67
|
collateral?: ReadonlyArray<string>;
|
|
64
68
|
onGate?: (e: GateEvent) => void | Promise<void>;
|
|
69
|
+
stateDir?: string;
|
|
65
70
|
}
|
|
66
71
|
/**
|
|
67
72
|
* The configured test command narrowed to these files. Mirrors testFiltered's `--` rule (acceptance.ts:104):
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -6,15 +6,17 @@ import { TIER_RANK } from "../config/config.js";
|
|
|
6
6
|
import { getAdapter } from "../adapters/registry.js";
|
|
7
7
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
8
8
|
import { acceptanceGate } from "./acceptance.js";
|
|
9
|
-
import { compareToBaseline } from "./baseline.js";
|
|
9
|
+
import { compareToBaseline, effectiveCeilingMs } from "./baseline.js";
|
|
10
10
|
import { evidenceGate } from "./evidence.js";
|
|
11
11
|
import { captureLlmOutput } from "./llm.js";
|
|
12
12
|
import { disallowedBy } from "../route/preference.js";
|
|
13
13
|
import { marginalCostRank } from "../route/router.js";
|
|
14
|
-
import { pickReviewer, reviewGate } from "./review.js";
|
|
14
|
+
import { gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
15
15
|
import { scopeGate } from "./scope.js";
|
|
16
|
-
import {
|
|
16
|
+
import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
|
|
17
|
+
import { shGit, resolvedCapacity } from "../run/git.js";
|
|
17
18
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
19
|
+
import { computeVerificationIdentity, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
18
20
|
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
19
21
|
let loadProvider = productionLoadProvider;
|
|
20
22
|
/** Test seam — inject deterministic load samples; production always reads os.loadavg. */
|
|
@@ -192,9 +194,45 @@ export function testCommandForFiles(testCmd, files) {
|
|
|
192
194
|
const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
|
|
193
195
|
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
194
196
|
}
|
|
197
|
+
/** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
|
|
198
|
+
async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir) {
|
|
199
|
+
const entry = baseline.commands.test;
|
|
200
|
+
const outcome = await evaluateManifestedTest(cmd, worktree, {
|
|
201
|
+
baselineDurations: entry?.fileDurations,
|
|
202
|
+
longestFile: entry?.longestFile,
|
|
203
|
+
overallCeilingMs: effectiveCeilingMs(entry),
|
|
204
|
+
artifactDir,
|
|
205
|
+
});
|
|
206
|
+
const reportPath = outcome.reportPath;
|
|
207
|
+
return {
|
|
208
|
+
gate: "test",
|
|
209
|
+
pass: outcome.pass,
|
|
210
|
+
details: outcome.details,
|
|
211
|
+
meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
|
|
212
|
+
};
|
|
213
|
+
}
|
|
214
|
+
const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
|
|
215
|
+
const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
|
|
216
|
+
/** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
|
|
217
|
+
* classification happens after persistence, too late to keep a scripted runner's non-verdict out.
|
|
218
|
+
* Keep named failures as work verdicts and preserve details for failure-policy fingerprinting. */
|
|
219
|
+
function classifySignalOnlyTest(g) {
|
|
220
|
+
if (g.gate !== "test" || g.pass || g.meta?.infra === true || !SIGNAL_EXIT_RE.test(g.details))
|
|
221
|
+
return;
|
|
222
|
+
const named = Array.isArray(g.meta?.failingTests) && g.meta.failingTests.length > 0;
|
|
223
|
+
if (named || FAILURE_IDENTITY_RE.test(g.details))
|
|
224
|
+
return;
|
|
225
|
+
g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
|
|
226
|
+
}
|
|
195
227
|
export async function runGates(task, ctx) {
|
|
196
228
|
const results = [];
|
|
197
229
|
let commits = [];
|
|
230
|
+
const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
|
|
231
|
+
const verdictStore = getVerdictStore(stateDir);
|
|
232
|
+
// VC-1: a reused verdict is journaled as its own row (the daemon appends every note by name) so
|
|
233
|
+
// the ledger names the reuse and the identity even where the gate-result row's details must stay
|
|
234
|
+
// the fresh verdict's (see formatReusedRow).
|
|
235
|
+
const noteReuse = (gate, r, id) => ctx.onGate?.({ phase: "note", gate, name: "gate-reused-verdict", payload: { gate, pass: r.pass, details: r.meta?.reusedDetails, ...reusedIdentity(id) }, result: r });
|
|
198
236
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
199
237
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
200
238
|
const failed = () => results.some((r) => !r.pass);
|
|
@@ -359,7 +397,7 @@ export async function runGates(task, ctx) {
|
|
|
359
397
|
const runBattery = async (commands, selected, gates = toolGates) => {
|
|
360
398
|
if (!gates.length)
|
|
361
399
|
return;
|
|
362
|
-
if (!v185) {
|
|
400
|
+
if (!v185 && !(commands.test && isVitestTestCommand(commands.test, ctx.worktree))) {
|
|
363
401
|
// ponytail: compareToBaseline batches adjacent tools — their starts are emitted at iteration,
|
|
364
402
|
// not at true execution start. They are collectively sub-second (measured), so the debounce
|
|
365
403
|
// suppresses them anyway; split compareToBaseline only if a tool gate ever gets slow.
|
|
@@ -386,25 +424,60 @@ export async function runGates(task, ctx) {
|
|
|
386
424
|
// any later tool before anyone reads its verdict.
|
|
387
425
|
for (const g of gates) {
|
|
388
426
|
await emitStart(g);
|
|
389
|
-
const
|
|
427
|
+
const cmd = commands[g];
|
|
428
|
+
let r;
|
|
429
|
+
let cached = false;
|
|
430
|
+
let identity;
|
|
431
|
+
if (cmd !== undefined) {
|
|
432
|
+
identity = await computeVerificationIdentity({
|
|
433
|
+
worktree: ctx.worktree,
|
|
434
|
+
gate: g,
|
|
435
|
+
scope: ctx.verificationScope,
|
|
436
|
+
command: cmd,
|
|
437
|
+
baseline: ctx.baseline,
|
|
438
|
+
selectedSet: g === "test" ? selected : undefined,
|
|
439
|
+
capacity: resolvedCapacity(),
|
|
440
|
+
});
|
|
441
|
+
const hit = verdictStore.get(identity);
|
|
442
|
+
if (hit)
|
|
443
|
+
classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
|
|
444
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
|
|
445
|
+
r = formatReusedRow(hit, identity);
|
|
446
|
+
cached = true;
|
|
447
|
+
await noteReuse(g, r, identity);
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
if (!r) {
|
|
451
|
+
// VL-1: a detected vitest test command is judged by its own invocation-bound report — the
|
|
452
|
+
// stdout-count/file-count path (compareToBaseline's fileCountDeficit) never runs for it. Any
|
|
453
|
+
// other scripted test command keeps today's exit-code contract byte-identically.
|
|
454
|
+
const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
|
|
455
|
+
r = useManifest
|
|
456
|
+
? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir))
|
|
457
|
+
: (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], g === "test" && selected ? { selected } : {})))[0];
|
|
458
|
+
}
|
|
390
459
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
391
460
|
if (g === "test" && selected)
|
|
392
|
-
selectedDurationMs = spans.get("test")
|
|
461
|
+
selectedDurationMs = spans.get("test")?.durationMs ?? 0;
|
|
393
462
|
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
394
463
|
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
395
464
|
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
396
465
|
// command, the last one included, and fail the gate whose command did it. (A red command needs
|
|
397
466
|
// no check: it already ends the round, and its own output is the truer verdict.)
|
|
398
|
-
if (r.pass && commands[g]) {
|
|
467
|
+
if (!cached && r.pass && commands[g]) {
|
|
399
468
|
const dirt = await dirtyWorktree();
|
|
400
469
|
if (dirt) {
|
|
401
470
|
await record(dirtyRefusal(g, dirt, commands[g]));
|
|
402
471
|
return;
|
|
403
472
|
}
|
|
404
473
|
}
|
|
474
|
+
if (r)
|
|
475
|
+
classifySignalOnlyTest(r);
|
|
476
|
+
if (!cached && identity && r && !isInfraResult(r)) {
|
|
477
|
+
verdictStore.set(identity, { ...r, meta: { ...r.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
478
|
+
}
|
|
405
479
|
if (g === "test" && selected) {
|
|
406
480
|
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
407
|
-
// green: held (see heldTest) so the full suite below can supersede it with ONE verdict.
|
|
408
481
|
if (!screened.pass)
|
|
409
482
|
await record(screened);
|
|
410
483
|
else {
|
|
@@ -602,7 +675,11 @@ export async function runGates(task, ctx) {
|
|
|
602
675
|
}
|
|
603
676
|
return rv;
|
|
604
677
|
};
|
|
605
|
-
|
|
678
|
+
// RF-1: THIS task's prior reviewers — earlier rounds' seats plus the seats that produced garbage for
|
|
679
|
+
// it (excludeReviewers names only dispatched seats). Kept apart from the eligibility exclusions the
|
|
680
|
+
// retry below adds for a flaked seat's whole adapter: those sibling channels never reviewed.
|
|
681
|
+
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
682
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, ctx.excludeReviewers, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers));
|
|
606
683
|
// OBS-193/574: an unparseable review verdict retries the REVIEW exactly once, preferring a
|
|
607
684
|
// different adapter. Only a single-adapter eligible pool may fall back to another channel on the
|
|
608
685
|
// flaked adapter. The flaked verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -622,10 +699,14 @@ export async function runGates(task, ctx) {
|
|
|
622
699
|
const priorExclusions = ctx.excludeReviewers ?? [];
|
|
623
700
|
const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
|
|
624
701
|
const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
|
|
625
|
-
|
|
702
|
+
// RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
|
|
703
|
+
// and the prior reviewers' tiers, the flaked seat's own included, so a retry never drops a tier.
|
|
704
|
+
const retryPrior = [...priorReviewers, flaked];
|
|
705
|
+
const retryFloor = gateReviewerFloor(task, ctx.cfg, ctx.author, ctx.channels, retryPrior).floor;
|
|
706
|
+
const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...priorExclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor);
|
|
626
707
|
const exclusion = crossAdapter ? "adapter" : "channel";
|
|
627
708
|
const retryExclusions = [...priorExclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
|
|
628
|
-
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings));
|
|
709
|
+
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, retryExclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior));
|
|
629
710
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
630
711
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
631
712
|
const route = exclusion === "adapter"
|
|
@@ -639,10 +720,11 @@ export async function runGates(task, ctx) {
|
|
|
639
720
|
meta: { ...second.meta, reviewRetry: { flaked, retried, exclusion } },
|
|
640
721
|
};
|
|
641
722
|
}
|
|
642
|
-
else
|
|
643
|
-
// Preserve the original no-answer cause when no replacement exists, but name the
|
|
644
|
-
// floor that correctly refused a lower-tier fallback.
|
|
645
|
-
|
|
723
|
+
else {
|
|
724
|
+
// Preserve the original no-answer cause when no replacement exists, but name the resolved
|
|
725
|
+
// floor that correctly refused a lower-tier fallback — in details AND in the row's meta.
|
|
726
|
+
const { reviewerFloor, reviewerFloorCause } = second.meta ?? {};
|
|
727
|
+
rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}`, meta: { ...rv.meta, reviewerFloor, reviewerFloorCause } };
|
|
646
728
|
}
|
|
647
729
|
}
|
|
648
730
|
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
|
@@ -739,9 +821,43 @@ export async function runGates(task, ctx) {
|
|
|
739
821
|
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
740
822
|
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
741
823
|
// a clean tree HERE is what makes "the gated commit is the tested tree" true at merge time.
|
|
742
|
-
|
|
743
|
-
|
|
744
|
-
|
|
824
|
+
// VL-1: the merge-candidate's manifest is the FULL set — a full suite whose report lacks one
|
|
825
|
+
// manifest file never reaches the pass branch below, so a selected-only green can never merge.
|
|
826
|
+
let full;
|
|
827
|
+
let cached = false;
|
|
828
|
+
let identity;
|
|
829
|
+
if (ctx.commands.test !== undefined) {
|
|
830
|
+
identity = await computeVerificationIdentity({
|
|
831
|
+
worktree: ctx.worktree,
|
|
832
|
+
gate: "test",
|
|
833
|
+
scope: ctx.verificationScope,
|
|
834
|
+
command: ctx.commands.test,
|
|
835
|
+
baseline: ctx.baseline,
|
|
836
|
+
selectedSet: undefined,
|
|
837
|
+
capacity: resolvedCapacity(),
|
|
838
|
+
});
|
|
839
|
+
const hit = verdictStore.get(identity);
|
|
840
|
+
if (hit)
|
|
841
|
+
classifySignalOnlyTest(hit);
|
|
842
|
+
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")) {
|
|
843
|
+
full = formatReusedRow(hit, identity);
|
|
844
|
+
cached = true;
|
|
845
|
+
await noteReuse("test", full, identity);
|
|
846
|
+
}
|
|
847
|
+
}
|
|
848
|
+
if (!full) {
|
|
849
|
+
const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
|
|
850
|
+
full = fullUsesManifest
|
|
851
|
+
? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir))
|
|
852
|
+
: (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"])))[0];
|
|
853
|
+
}
|
|
854
|
+
fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
|
|
855
|
+
const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
|
|
856
|
+
if (full)
|
|
857
|
+
classifySignalOnlyTest(full);
|
|
858
|
+
if (!cached && identity && full && !dirt && !isInfraResult(full)) {
|
|
859
|
+
verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
|
|
860
|
+
}
|
|
745
861
|
const merged = withTelemetry(dirt
|
|
746
862
|
? dirtyRefusal("test", dirt, ctx.commands.test)
|
|
747
863
|
: { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import type { BaselineFileDuration } from "./baseline.js";
|
|
2
|
+
export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
|
|
3
|
+
/** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
|
|
4
|
+
* --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
|
|
5
|
+
* through any symlink in it (e.g. macOS's `/var` -> `/private/var`, under which every OS temp dir —
|
|
6
|
+
* and so every test fixture worktree — lives); `cwd` as tickmarkr holds it may not be. Resolving cwd
|
|
7
|
+
* before computing the relative path is what makes the two actually comparable. Manifests built from
|
|
8
|
+
* a selected-test screen are already repo-relative and pass through unchanged. */
|
|
9
|
+
export declare function toManifestPath(file: string, cwd: string): string;
|
|
10
|
+
export interface TestReportCompletion {
|
|
11
|
+
at: number;
|
|
12
|
+
status: "passed" | "failed";
|
|
13
|
+
/** Failure fingerprints for this file; absent/empty on a passed file. */
|
|
14
|
+
failures?: string[];
|
|
15
|
+
}
|
|
16
|
+
/** The runner's own machine report — requested/started/completed are the runner's claims about ITSELF. */
|
|
17
|
+
export interface TestReport {
|
|
18
|
+
nonce: string;
|
|
19
|
+
requested: string[];
|
|
20
|
+
started: Record<string, number>;
|
|
21
|
+
completed: Record<string, TestReportCompletion>;
|
|
22
|
+
/** Files the reporter observed complete MORE than once — `completed`'s object keys cannot show
|
|
23
|
+
* this themselves (a second write silently overwrites the first), so the reporter records the
|
|
24
|
+
* evidence separately before it is lost. */
|
|
25
|
+
duplicateCompletions?: string[];
|
|
26
|
+
/** Written last, once, when the runner reaches its own terminal state. Its absence means the run
|
|
27
|
+
* never certified completion — killed, crashed, or still in flight — and is never a verdict. */
|
|
28
|
+
certificate?: {
|
|
29
|
+
at: number;
|
|
30
|
+
exitCode: number;
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
/** Reads and structurally validates the report; a missing or malformed file is `undefined` — never a partial parse. */
|
|
34
|
+
export declare function readTestReport(path: string): TestReport | undefined;
|
|
35
|
+
export type ManifestVerdictKind = "pass" | "infra" | "work" | "fail-closed";
|
|
36
|
+
export interface ManifestVerdict {
|
|
37
|
+
kind: ManifestVerdictKind;
|
|
38
|
+
pass: boolean;
|
|
39
|
+
details: string;
|
|
40
|
+
meta: Record<string, unknown>;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* The independent validator: given the manifest THIS invocation was asked to prove, its bound nonce
|
|
44
|
+
* and the independently observed process exit code, decide the
|
|
45
|
+
* verdict from the report alone. `killedFile` short-circuits every report-shaped check — a job this
|
|
46
|
+
* module killed for a per-file hang never reaches its report.
|
|
47
|
+
*/
|
|
48
|
+
export declare function verifyManifestReport(opts: {
|
|
49
|
+
manifest: readonly string[];
|
|
50
|
+
nonce: string;
|
|
51
|
+
exitCode: number | undefined;
|
|
52
|
+
report: TestReport | undefined;
|
|
53
|
+
killedFile?: string;
|
|
54
|
+
hangBudgetMs?: number;
|
|
55
|
+
}): ManifestVerdict;
|
|
56
|
+
/** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
|
|
57
|
+
export declare const FILE_HANG_SLACK = 3;
|
|
58
|
+
export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
|
|
59
|
+
export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
|
|
60
|
+
export interface ManifestRunResult {
|
|
61
|
+
exitCode: number | undefined;
|
|
62
|
+
stdout: string;
|
|
63
|
+
stderr: string;
|
|
64
|
+
report: TestReport | undefined;
|
|
65
|
+
killedFile?: string;
|
|
66
|
+
hangBudgetMs?: number;
|
|
67
|
+
/** The child's own pid (its process GROUP id too, since it is spawned detached) — for a caller
|
|
68
|
+
* that wants to prove the group is really gone after a hang kill (`process.kill(-pid, 0)` throws). */
|
|
69
|
+
pid?: number;
|
|
70
|
+
}
|
|
71
|
+
/** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
|
|
72
|
+
* timeout kills the detached process group, including descendants holding the output pipes. */
|
|
73
|
+
export declare function runManifestedTest(cmd: string, cwd: string, opts: {
|
|
74
|
+
manifest: readonly string[];
|
|
75
|
+
nonce: string;
|
|
76
|
+
reportPath: string;
|
|
77
|
+
env?: NodeJS.ProcessEnv;
|
|
78
|
+
baselineDurations?: readonly BaselineFileDuration[] | null;
|
|
79
|
+
longestFile?: BaselineFileDuration | null;
|
|
80
|
+
pollMs?: number;
|
|
81
|
+
overallCeilingMs?: number;
|
|
82
|
+
}): Promise<ManifestRunResult>;
|
|
83
|
+
export interface ManifestGateOutcome {
|
|
84
|
+
pass: boolean;
|
|
85
|
+
kind: ManifestVerdictKind;
|
|
86
|
+
details: string;
|
|
87
|
+
classification?: "infra" | "regression";
|
|
88
|
+
meta: Record<string, unknown>;
|
|
89
|
+
exitCode: number;
|
|
90
|
+
reportPath: string;
|
|
91
|
+
}
|
|
92
|
+
/** One configured runner execution, and its own collection under the same arguments and environment.
|
|
93
|
+
* The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
|
|
94
|
+
export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
|
|
95
|
+
baselineDurations?: readonly BaselineFileDuration[] | null;
|
|
96
|
+
longestFile?: BaselineFileDuration | null;
|
|
97
|
+
overallCeilingMs?: number;
|
|
98
|
+
artifactDir?: string;
|
|
99
|
+
}): Promise<ManifestGateOutcome>;
|