tickmarkr 2.6.2 → 2.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -2
- package/dist/adapters/codex.d.ts +2 -1
- package/dist/adapters/codex.js +26 -5
- package/dist/adapters/model-lints.d.ts +1 -0
- package/dist/adapters/model-lints.js +48 -1
- package/dist/adapters/model-windows.d.ts +2 -1
- package/dist/adapters/model-windows.js +18 -7
- package/dist/adapters/types.d.ts +6 -0
- package/dist/cli/commands/approve.d.ts +6 -3
- package/dist/cli/commands/approve.js +43 -6
- package/dist/cli/commands/doctor.d.ts +8 -2
- package/dist/cli/commands/doctor.js +6 -1
- package/dist/cli/commands/fleet.js +264 -59
- package/dist/cli/commands/init.js +5 -3
- package/dist/cli/commands/plan.js +13 -8
- package/dist/cli/commands/report.d.ts +40 -2
- package/dist/cli/commands/report.js +278 -11
- package/dist/cli/commands/resume.js +2 -4
- package/dist/cli/commands/run.d.ts +11 -0
- package/dist/cli/commands/run.js +33 -5
- package/dist/cli/commands/status.js +49 -4
- package/dist/cli/commands/verify.js +332 -121
- package/dist/compile/native.js +137 -0
- package/dist/config/config.d.ts +40 -9
- package/dist/config/config.js +133 -15
- package/dist/config/fleet-overlay.d.ts +13 -3
- package/dist/config/fleet-overlay.js +12 -8
- package/dist/drivers/herdr.d.ts +12 -0
- package/dist/drivers/herdr.js +51 -0
- package/dist/drivers/orca.d.ts +9 -1
- package/dist/drivers/orca.js +29 -7
- package/dist/drivers/types.d.ts +2 -0
- package/dist/drivers/types.js +2 -2
- package/dist/gates/acceptance.d.ts +7 -0
- package/dist/gates/acceptance.js +27 -5
- package/dist/gates/baseline.d.ts +20 -1
- package/dist/gates/baseline.js +100 -20
- package/dist/gates/cache.d.ts +8 -0
- package/dist/gates/cache.js +12 -2
- package/dist/gates/llm.d.ts +6 -0
- package/dist/gates/llm.js +27 -8
- package/dist/gates/review.d.ts +6 -1
- package/dist/gates/review.js +122 -32
- package/dist/gates/run-gates.d.ts +54 -3
- package/dist/gates/run-gates.js +331 -45
- package/dist/gates/test-manifest.d.ts +42 -0
- package/dist/gates/test-manifest.js +69 -10
- package/dist/route/router.d.ts +23 -1
- package/dist/route/router.js +54 -16
- package/dist/run/consult.d.ts +3 -1
- package/dist/run/consult.js +4 -2
- package/dist/run/daemon.d.ts +2 -1
- package/dist/run/daemon.js +349 -79
- package/dist/run/interactive-seed.d.ts +4 -0
- package/dist/run/interactive-seed.js +35 -9
- package/dist/run/journal.d.ts +123 -1
- package/dist/run/journal.js +480 -17
- package/dist/run/lease.d.ts +13 -0
- package/dist/run/lease.js +45 -0
- package/dist/run/protocol.d.ts +15 -0
- package/dist/run/protocol.js +11 -1
- package/dist/run/receipt-resolver.d.ts +22 -0
- package/dist/run/receipt-resolver.js +40 -1
- package/dist/run/repair-selection.d.ts +11 -1
- package/dist/run/repair-selection.js +17 -9
- package/dist/run/wall-budget.d.ts +48 -0
- package/dist/run/wall-budget.js +280 -0
- package/dist/tui/cockpit/live-store.d.ts +6 -0
- package/dist/tui/cockpit/live-store.js +36 -11
- package/dist/tui/cockpit/run-cockpit.js +2 -2
- package/dist/tui/cockpit/run-view.d.ts +2 -1
- package/dist/tui/cockpit/run-view.js +13 -9
- package/dist/tui/cockpit/setup-cockpit.d.ts +2 -0
- package/dist/tui/cockpit/setup-cockpit.js +4 -0
- package/package.json +2 -1
- package/schema/config.schema.json +8 -1
- package/skills/tickmarkr-auto/SKILL.md +2 -2
- package/skills/tickmarkr-loop/SKILL.md +17 -5
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +4 -1
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { randomUUID } from "node:crypto";
|
|
1
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
2
2
|
import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
|
|
3
3
|
import { loadavg, tmpdir } from "node:os";
|
|
4
4
|
import { join, posix } from "node:path";
|
|
@@ -10,14 +10,14 @@ import { acceptanceGate } from "./acceptance.js";
|
|
|
10
10
|
import { compareToBaseline, effectiveCeilingMs, waitForCalmWindow, calmWindowReady } from "./baseline.js";
|
|
11
11
|
import { evidenceGate } from "./evidence.js";
|
|
12
12
|
import { captureLlmOutput } from "./llm.js";
|
|
13
|
-
import { disallowedBy } from "../route/preference.js";
|
|
13
|
+
import { disallowedBy, observedSeat } from "../route/preference.js";
|
|
14
14
|
import { marginalCostRank } from "../route/router.js";
|
|
15
15
|
import { carriedAuthorVendors, gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
16
16
|
import { scopeGate } from "./scope.js";
|
|
17
|
-
import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
|
|
17
|
+
import { discoverTestManifest, evaluateManifestedTest, isVitestTestCommand, VITEST_CACHE_ENV, worktreeVitestCache } from "./test-manifest.js";
|
|
18
18
|
import { executionSignal } from "../run/execution-budget.js";
|
|
19
19
|
import { failureDisposition } from "../run/recovery.js";
|
|
20
|
-
import { dependencyLinkRefusal, preserveWorktree, producerFields, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
20
|
+
import { dependencyLinkRefusal, FORK_CAP_ENV, preserveWorktree, producerFields, ROUTING_ENV_SEAMS, shGit, resolvedCapacity, SUITE_PARENT_ENV, verificationProtocol } from "../run/git.js";
|
|
21
21
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
22
22
|
import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
23
23
|
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
@@ -215,6 +215,44 @@ export function testCommandForFiles(testCmd, files) {
|
|
|
215
215
|
const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
|
|
216
216
|
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
217
217
|
}
|
|
218
|
+
/** OBS-635: a screen costing at least this share of the full suite runs the full suite instead. */
|
|
219
|
+
export const SCREEN_PROMOTION_RATIO = 0.75;
|
|
220
|
+
/**
|
|
221
|
+
* The screen's share of the full suite's cost, from the per-file durations the harness measured at
|
|
222
|
+
* baseline capture — never a worker's timing. Undefined (unknown) unless every selected file has a
|
|
223
|
+
* measured duration and the measured total is positive; unknown keeps the conservative screen path.
|
|
224
|
+
*/
|
|
225
|
+
export function screenCostRatio(baseline, selected) {
|
|
226
|
+
const entry = baseline.commands.test;
|
|
227
|
+
const files = entry?.infra ? undefined : entry?.fileDurations;
|
|
228
|
+
if (!files?.length)
|
|
229
|
+
return undefined;
|
|
230
|
+
const cost = new Map(files.map((f) => [f.file, f.durationMs]));
|
|
231
|
+
const total = files.reduce((sum, f) => sum + f.durationMs, 0);
|
|
232
|
+
if (!(total > 0) || selected.some((file) => !cost.has(file)))
|
|
233
|
+
return undefined;
|
|
234
|
+
return selected.reduce((sum, file) => sum + cost.get(file), 0) / total;
|
|
235
|
+
}
|
|
236
|
+
/** OBS-635: the full manifest the runner lists NOW, under the environment evaluateManifestedTest's own
|
|
237
|
+
* discovery receives (test-manifest.ts manifestEnvironment), so it compares with the one a verdict
|
|
238
|
+
* certified. Undefined when the runner cannot list — an unlisted manifest certifies nothing. */
|
|
239
|
+
async function listFullManifest(cmd, worktree) {
|
|
240
|
+
const env = { ...process.env, PATH: `${join(worktree, "node_modules/.bin")}:${process.env.PATH ?? ""}`,
|
|
241
|
+
[VITEST_CACHE_ENV]: worktreeVitestCache(worktree),
|
|
242
|
+
[FORK_CAP_ENV]: String(resolvedCapacity().forkCap), [SUITE_PARENT_ENV]: String(process.pid) };
|
|
243
|
+
for (const key of [...ROUTING_ENV_SEAMS, "VITEST", "TEST", "VITEST_WORKER_ID", "VITEST_POOL_ID"])
|
|
244
|
+
delete env[key];
|
|
245
|
+
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-full-manifest-"));
|
|
246
|
+
try {
|
|
247
|
+
return (await discoverTestManifest(cmd, worktree, { dir, nonce: randomUUID(), env })).files;
|
|
248
|
+
}
|
|
249
|
+
catch {
|
|
250
|
+
return undefined;
|
|
251
|
+
}
|
|
252
|
+
finally {
|
|
253
|
+
rmSync(dir, { recursive: true, force: true });
|
|
254
|
+
}
|
|
255
|
+
}
|
|
218
256
|
/** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
|
|
219
257
|
async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry = {}, retried = false) {
|
|
220
258
|
const entry = baseline.commands.test;
|
|
@@ -249,6 +287,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
|
|
|
249
287
|
meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
|
|
250
288
|
};
|
|
251
289
|
}
|
|
290
|
+
/** A retry base no runner invocation parses. evaluateManifestedTest builds its stranded single-fork
|
|
291
|
+
* retry from the base it is handed, and one it cannot parse throws before any spawn — so this base
|
|
292
|
+
* disables that inner recovery: a worker-RPC-stranded re-observation comes back infra (the caller
|
|
293
|
+
* parks it as ambiguous) instead of launching a second execution. */
|
|
294
|
+
export const REOBSERVATION_RETRY_BASE = "tickmarkr-reobservation-refuses-stranded-retry";
|
|
295
|
+
/** OBS-1106 residual: ONE isolated re-observation of a timeout-shaped red's attributed failing files on
|
|
296
|
+
* the same checkout, narrowed exactly as a screen is. Never cached and never a verdict: the caller
|
|
297
|
+
* keeps the original red and reads this only to decide whether that red is chargeable. Exactly one
|
|
298
|
+
* execution — the bounded infra/host-starved retries and the stranded single-fork recovery are all
|
|
299
|
+
* refused, so a diagnostic never buys more. */
|
|
300
|
+
export async function reobserveTestFiles(worktree, testCmd, baseline, files, artifactDir) {
|
|
301
|
+
const cmd = testCommandForFiles(testCmd, files);
|
|
302
|
+
if (!isVitestTestCommand(testCmd, worktree))
|
|
303
|
+
return (await compareToBaseline(worktree, { test: cmd }, baseline, ["test"], { selected: files, authorizeRetry: () => false }))[0];
|
|
304
|
+
const r = await runVitestManifestGate(worktree, cmd, baseline, files, artifactDir, { retryBaseCommand: REOBSERVATION_RETRY_BASE });
|
|
305
|
+
// fail closed whatever a recovery did: a re-observation never reads a recovered verdict
|
|
306
|
+
return r.meta?.recovery === undefined ? r
|
|
307
|
+
: { ...r, pass: false, meta: { ...r.meta, classification: "infra", infra: true, retryable: false, recoveryRefused: true } };
|
|
308
|
+
}
|
|
252
309
|
const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
|
|
253
310
|
const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
|
|
254
311
|
/** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
|
|
@@ -262,12 +319,56 @@ function classifySignalOnlyTest(g) {
|
|
|
262
319
|
return;
|
|
263
320
|
g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
|
|
264
321
|
}
|
|
322
|
+
/** OBS-1151: a criterion's comparable subject — its canonical text, the task's declared bounds and the
|
|
323
|
+
* operator context. The cited files' blobs are compared separately, over the union of both citations. */
|
|
324
|
+
export function judgmentSubjectKey(task, criterion, operatorContext) {
|
|
325
|
+
return createHash("sha256").update(JSON.stringify([criterion, [...task.files].sort(), [...(task.outOfScope ?? [])].sort(), operatorContext ?? ""])).digest("hex");
|
|
326
|
+
}
|
|
327
|
+
async function blobAt(worktree, commit, path) {
|
|
328
|
+
const r = await shGit(`git rev-parse --verify --quiet ${shq(`${commit}:${path}`)}`, worktree);
|
|
329
|
+
return r.code === 0 && r.stdout.trim() ? r.stdout.trim() : undefined;
|
|
330
|
+
}
|
|
331
|
+
/** OBS-1151: the criteria whose fresh ruling reverses the newest prior ruling on the same subject key whose
|
|
332
|
+
* cited paths hold the identical blob at both commits (older comparable priors are still found behind a
|
|
333
|
+
* newer prior on different blobs). A citation-less side or an
|
|
334
|
+
* unreadable blob is unknown, never identical — that criterion's fresh ruling is simply fresh. */
|
|
335
|
+
export async function judgeContradictions(worktree, head, fresh, priors) {
|
|
336
|
+
const found = [];
|
|
337
|
+
for (const c of fresh) {
|
|
338
|
+
if (!c.paths.length)
|
|
339
|
+
continue;
|
|
340
|
+
// C-3 (D-669): the comparable prior is the NEWEST one on identical cited blobs, not the newest one
|
|
341
|
+
// carrying the key — PASS(A) → FAIL(B) → fresh FAIL(A) must still be adjudicated against PASS(A).
|
|
342
|
+
for (const prior of priors) {
|
|
343
|
+
const was = prior.criteria.find((q) => q.key === c.key);
|
|
344
|
+
if (!was || !was.paths.length)
|
|
345
|
+
continue;
|
|
346
|
+
const paths = [...new Set([...was.paths, ...c.paths])].sort();
|
|
347
|
+
let identical = true;
|
|
348
|
+
for (const path of paths) {
|
|
349
|
+
const [before, now] = await Promise.all([blobAt(worktree, prior.commit, path), blobAt(worktree, head, path)]);
|
|
350
|
+
if (!before || !now || before !== now) {
|
|
351
|
+
identical = false;
|
|
352
|
+
break;
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
if (!identical)
|
|
356
|
+
continue;
|
|
357
|
+
if (was.met !== c.met)
|
|
358
|
+
found.push({ id: c.id, met: c.met, priorMet: was.met, priorCommit: prior.commit, paths });
|
|
359
|
+
break;
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
return found;
|
|
363
|
+
}
|
|
265
364
|
export async function runGates(task, ctx) {
|
|
266
365
|
const results = [];
|
|
267
366
|
const evidence = {
|
|
268
367
|
artifactDir: ctx.artifactDir,
|
|
269
368
|
runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
|
|
270
369
|
taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
|
|
370
|
+
// OBS-1140: task and standalone gates honour the configured quota, not the built-in default.
|
|
371
|
+
quotaBytes: ctx.cfg.gates?.evidenceQuotaBytes,
|
|
271
372
|
...ctx.evidence,
|
|
272
373
|
};
|
|
273
374
|
// Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
|
|
@@ -305,6 +406,9 @@ export async function runGates(task, ctx) {
|
|
|
305
406
|
await receiptNotes;
|
|
306
407
|
};
|
|
307
408
|
let selectionDecision;
|
|
409
|
+
// OBS-635: the identity a full suite measured inside the battery, revalidated after semantics.
|
|
410
|
+
let fullInBattery = false;
|
|
411
|
+
let batteryFullIdentity;
|
|
308
412
|
let commits = [];
|
|
309
413
|
// Check before cache identity, npm policy probes, or any gate command.
|
|
310
414
|
const dependencyRefusal = dependencyLinkRefusal(ctx.worktree);
|
|
@@ -338,6 +442,29 @@ export async function runGates(task, ctx) {
|
|
|
338
442
|
payload: { gate, reason: "cached-red-discarded", ...(reason === "recheck" ? {} : { bypass: reason }) }, result: hit });
|
|
339
443
|
return true;
|
|
340
444
|
};
|
|
445
|
+
// OBS-1168(c): every judge/review seat this round opens, so a failed sibling can cancel the other.
|
|
446
|
+
// A cancelled round dispatches, re-routes and publishes nothing more; its closed seats' own errors
|
|
447
|
+
// are consequences of the cancel, never a second failure.
|
|
448
|
+
// A seat whose creation was still pending at the cancel is refused before dispatch: onSlot runs inside
|
|
449
|
+
// llm.ts's launch guard, so the throw closes (and awaits) that half-launched pane and never runs it.
|
|
450
|
+
const semanticSlots = new Set();
|
|
451
|
+
const closing = [];
|
|
452
|
+
const via = ctx.via && {
|
|
453
|
+
...ctx.via, onSlot: (slot) => {
|
|
454
|
+
semanticSlots.add(slot);
|
|
455
|
+
ctx.via.onSlot?.(slot);
|
|
456
|
+
if (cancelled)
|
|
457
|
+
throw new Error("semantic round cancelled before dispatch");
|
|
458
|
+
},
|
|
459
|
+
};
|
|
460
|
+
let cancelled = false;
|
|
461
|
+
const cancelSemantic = () => {
|
|
462
|
+
if (cancelled)
|
|
463
|
+
return;
|
|
464
|
+
cancelled = true;
|
|
465
|
+
for (const slot of semanticSlots)
|
|
466
|
+
closing.push(via.driver.close(slot).catch(() => { }));
|
|
467
|
+
};
|
|
341
468
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
342
469
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
343
470
|
const failed = () => results.some((r) => !r.pass);
|
|
@@ -639,6 +766,28 @@ export async function runGates(task, ctx) {
|
|
|
639
766
|
},
|
|
640
767
|
};
|
|
641
768
|
};
|
|
769
|
+
const fullTestIdentity = () => computeVerificationIdentity({
|
|
770
|
+
worktree: ctx.worktree,
|
|
771
|
+
gate: "test",
|
|
772
|
+
scope: ctx.verificationScope,
|
|
773
|
+
command: ctx.commands.test,
|
|
774
|
+
baseline: ctx.baseline,
|
|
775
|
+
selectedSet: undefined,
|
|
776
|
+
capacity: resolvedCapacity(),
|
|
777
|
+
});
|
|
778
|
+
// OBS-635: a full green answers only for the manifest it certified. The tree identity cannot see an
|
|
779
|
+
// ignored generated test the runner would collect, so every full-green reuse rediscovers the
|
|
780
|
+
// runner's listing and requires the verdict's to equal it; a runner without a listing is bound by
|
|
781
|
+
// its identity alone. Reads before the semantic gates share one listing; `listing` resets after them.
|
|
782
|
+
let listing;
|
|
783
|
+
const certifiesFullManifest = async (verdict) => {
|
|
784
|
+
if (!verdict.pass || !isVitestTestCommand(ctx.commands.test, ctx.worktree))
|
|
785
|
+
return true;
|
|
786
|
+
const certified = verdict.meta?.manifest;
|
|
787
|
+
const current = await (listing ??= listFullManifest(ctx.commands.test, ctx.worktree));
|
|
788
|
+
return Array.isArray(certified) && current !== undefined
|
|
789
|
+
&& certified.length === current.length && [...certified].sort().every((file, i) => file === current[i]);
|
|
790
|
+
};
|
|
642
791
|
// shell tools vs the shared baseline
|
|
643
792
|
const retryOptions = (identity) => ctx.authorizeInfraRetry
|
|
644
793
|
? { authorizeRetry: (cause) => ctx.authorizeInfraRetry(identity ? verificationIdentityKey(identity) : "", cause === "infra" ? "infrastructure" : "host-starved") }
|
|
@@ -668,7 +817,7 @@ export async function runGates(task, ctx) {
|
|
|
668
817
|
if (hit)
|
|
669
818
|
classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
|
|
670
819
|
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
671
|
-
&& !(await discardCachedRed(g, hit))) {
|
|
820
|
+
&& !(await discardCachedRed(g, hit)) && (g !== "test" || selected !== undefined || await certifiesFullManifest(hit))) {
|
|
672
821
|
r = formatReusedRow(hit, identity);
|
|
673
822
|
cached = true;
|
|
674
823
|
if (g === "build")
|
|
@@ -695,6 +844,10 @@ export async function runGates(task, ctx) {
|
|
|
695
844
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
696
845
|
if (g === "test" && selected)
|
|
697
846
|
selectedDurationMs = spans.get("test")?.durationMs ?? 0;
|
|
847
|
+
if (g === "test" && !selected) {
|
|
848
|
+
fullInBattery = true;
|
|
849
|
+
batteryFullIdentity = identity;
|
|
850
|
+
}
|
|
698
851
|
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
699
852
|
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
700
853
|
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
@@ -802,7 +955,9 @@ export async function runGates(task, ctx) {
|
|
|
802
955
|
// v1.87 T2: the judge is a configured seat like any other — check it against the operator's
|
|
803
956
|
// policy BEFORE spending a dispatch on it. disallowedBy carries the whole deny grammar (adapter,
|
|
804
957
|
// model, or adapter:model), so a model-scoped deny cannot slip past an adapter-id-only read.
|
|
805
|
-
|
|
958
|
+
// OBS-1186: under the exact cached identity of that channel, as compile, doctor and route read it.
|
|
959
|
+
const judgeSeat = observedSeat(ctx.health, ctx.cfg.judge.adapter, ctx.cfg.judge.model);
|
|
960
|
+
const judgeDenied = disallowedBy(judgeSeat, ctx.cfg.routing, "judge");
|
|
806
961
|
if (judgeDenied) {
|
|
807
962
|
return {
|
|
808
963
|
result: {
|
|
@@ -815,8 +970,8 @@ export async function runGates(task, ctx) {
|
|
|
815
970
|
};
|
|
816
971
|
}
|
|
817
972
|
const judgeAdapter = getAdapter(ctx.cfg.judge.adapter, ctx.adapters);
|
|
818
|
-
const jvia =
|
|
819
|
-
? { driver:
|
|
973
|
+
const jvia = via
|
|
974
|
+
? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", judgeAdapter.id), label: via.labelFor("judge") }
|
|
820
975
|
: undefined;
|
|
821
976
|
// v1.19 (T2): testCmd threads the detected test runner to the gate so named-test oracles run
|
|
822
977
|
// deterministically (filtered via -t) before any LLM judge dispatch.
|
|
@@ -850,6 +1005,56 @@ export async function runGates(task, ctx) {
|
|
|
850
1005
|
}
|
|
851
1006
|
return captured.value;
|
|
852
1007
|
};
|
|
1008
|
+
const judgePool = () => (ctx.judgeChannels ?? []).filter((c) => disallowedBy(c, ctx.cfg.routing, "judge") === null);
|
|
1009
|
+
const rankJudges = (pool) => [...pool]
|
|
1010
|
+
.sort((x, y) => TIER_RANK[y.tier] - TIER_RANK[x.tier] || marginalCostRank(x) - marginalCostRank(y));
|
|
1011
|
+
// OBS-1151 (+add.1): the fresh judgment always stands on its own reading — a prior PASS is never
|
|
1012
|
+
// reused. Only a criterion that REVERSES the newest prior ruling on the same subject key over
|
|
1013
|
+
// identical cited blobs needs a second, distinct judge; agreement stands (a sound FAIL included),
|
|
1014
|
+
// and a split, no distinct eligible seat or an unreadable adjudication parks for the operator.
|
|
1015
|
+
const adjudicate = async (fresh) => {
|
|
1016
|
+
const primary = String(fresh.meta?.judge ?? channelKey({ adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model }));
|
|
1017
|
+
const head = await shGit("git rev-parse HEAD", ctx.worktree);
|
|
1018
|
+
const commit = head.code === 0 ? head.stdout.trim() : "";
|
|
1019
|
+
const criteria = fresh.meta.judgment.map((c) => ({
|
|
1020
|
+
id: c.id, key: judgmentSubjectKey(task, c.criterion, ctx.operatorContext), met: c.met, paths: c.paths,
|
|
1021
|
+
}));
|
|
1022
|
+
const record = { commit, judge: primary, criteria };
|
|
1023
|
+
const stamped = { ...fresh, meta: { ...fresh.meta, judgment: record } };
|
|
1024
|
+
if (!commit || !ctx.priorJudgments?.length)
|
|
1025
|
+
return commit ? stamped : { ...fresh, meta: { ...fresh.meta, judgment: undefined } };
|
|
1026
|
+
const disputed = await judgeContradictions(ctx.worktree, commit, criteria, ctx.priorJudgments);
|
|
1027
|
+
if (!disputed.length)
|
|
1028
|
+
return stamped;
|
|
1029
|
+
await ctx.onGate?.({ phase: "note", gate: "acceptance", name: "judge-disagreement", payload: { primary, commit, disputed } });
|
|
1030
|
+
const park = (why, adjudicator) => ({
|
|
1031
|
+
gate: "acceptance", pass: false,
|
|
1032
|
+
details: `judge disagreement on ${disputed.map((d) => d.id).join(", ")}: ${primary} reverses an earlier ruling over identical cited blobs (${[...new Set(disputed.flatMap((d) => d.paths))].join(", ")}) — ${why}; parked for an operator ruling, no worker charge`,
|
|
1033
|
+
meta: { classification: "infra", infra: true, retryable: false, cause: "judge-disagreement", judge: primary,
|
|
1034
|
+
judgeDisagreement: { primary, ...(adjudicator ? { adjudicator } : {}), disputed, outcome: why } },
|
|
1035
|
+
});
|
|
1036
|
+
const primaryAdapter = primary.slice(0, primary.indexOf(":"));
|
|
1037
|
+
// One DISTINCT seat: never the primary channel, a different adapter when the pool has one.
|
|
1038
|
+
const pool = judgePool().filter((c) => channelKey(c) !== primary);
|
|
1039
|
+
const seat = rankJudges(pool.filter((c) => c.adapter !== primaryAdapter))[0] ?? rankJudges(pool)[0];
|
|
1040
|
+
if (!seat)
|
|
1041
|
+
return park("no distinct eligible judge is available");
|
|
1042
|
+
const adjudicator = channelKey(seat);
|
|
1043
|
+
const seatAdapter = getAdapter(seat.adapter, ctx.adapters);
|
|
1044
|
+
const seatVia = via
|
|
1045
|
+
? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", seatAdapter.id) + "-r2", label: via.labelFor("judge") }
|
|
1046
|
+
: undefined;
|
|
1047
|
+
const second = await invokeJudge(seatAdapter, seat.model, seatVia, configuredEffort(ctx.cfg, seat));
|
|
1048
|
+
const rulings = Array.isArray(second.meta?.judgment) ? second.meta.judgment : undefined;
|
|
1049
|
+
// A citation-less ruling cannot be compared, so it confirms nothing (invented evidence is already unparseable,
|
|
1050
|
+
// and an internally inconsistent verdict carries no judgment rows at all).
|
|
1051
|
+
if (second.meta?.unparseable === true || !rulings)
|
|
1052
|
+
return park("the adjudicating judge returned no readable verdict", adjudicator);
|
|
1053
|
+
const agreed = disputed.every((d) => rulings.some((r) => r.id === d.id && r.met === d.met && r.paths.length > 0));
|
|
1054
|
+
if (!agreed)
|
|
1055
|
+
return park("the adjudicating judge split from the fresh ruling", adjudicator);
|
|
1056
|
+
return { ...stamped, meta: { ...stamped.meta, adjudication: { primary, adjudicator, criteria: disputed.map((d) => d.id), agreed: true } } };
|
|
1057
|
+
};
|
|
853
1058
|
// OBS-1182: every judge seat launches at its OWN configured effort, never the worker's.
|
|
854
1059
|
let a = await invokeJudge(judgeAdapter, ctx.cfg.judge.model, jvia, configuredEffort(ctx.cfg, ctx.cfg.judge));
|
|
855
1060
|
// GATE-09: an unparseable judge verdict retries the JUDGE exactly once on a failover channel — never
|
|
@@ -865,7 +1070,7 @@ export async function runGates(task, ctx) {
|
|
|
865
1070
|
// If no other adapter is live, the exclusion degrades to a channel-level reroute within the same
|
|
866
1071
|
// adapter so a single-adapter fleet still retries (matching the daemon's unknown-excludeAdapter
|
|
867
1072
|
// degradation path).
|
|
868
|
-
if (a.meta?.unparseable === true && typeof a.meta.judge === "string") {
|
|
1073
|
+
if (!cancelled && a.meta?.unparseable === true && typeof a.meta.judge === "string") {
|
|
869
1074
|
const flakedKey = a.meta.judge;
|
|
870
1075
|
const flakedAdapter = flakedKey.slice(0, flakedKey.indexOf(":"));
|
|
871
1076
|
const pick = (pool) => pool
|
|
@@ -879,17 +1084,23 @@ export async function runGates(task, ctx) {
|
|
|
879
1084
|
const sameAdapter = pick(judgePool.filter((c) => c.adapter === flakedAdapter && channelKey(c) !== flakedKey));
|
|
880
1085
|
// Prefer a different adapter; if the fleet only has one adapter, retry on a different channel of
|
|
881
1086
|
// that adapter; if the fleet has only one channel, fall back to the original judge config.
|
|
882
|
-
const retry = crossAdapter ?? sameAdapter ??
|
|
1087
|
+
const retry = crossAdapter ?? sameAdapter ?? judgeSeat;
|
|
883
1088
|
const retryAdapter = getAdapter(retry.adapter, ctx.adapters);
|
|
884
|
-
const retryJvia =
|
|
1089
|
+
const retryJvia = via
|
|
885
1090
|
// unconditional -r1 suffix: under keepPanes:forever a same-channel retry cannot collide with the
|
|
886
1091
|
// still-open first pane (herdr agent_name_taken regression, research Pitfall 4)
|
|
887
|
-
? { driver:
|
|
1092
|
+
? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", retryAdapter.id) + "-r1", label: via.labelFor("judge") }
|
|
888
1093
|
: undefined;
|
|
889
1094
|
// the retry IS a second acceptanceGate call: one code path, one parser, zero new parse leniency.
|
|
890
1095
|
a = await invokeJudge(retryAdapter, retry.model, retryJvia, configuredEffort(ctx.cfg, retry));
|
|
891
1096
|
a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
|
|
892
1097
|
}
|
|
1098
|
+
// OBS-1168(b): the re-routed seat could not launch either — no seat produced a verdict, so this is
|
|
1099
|
+
// an infra park over whatever the deterministic gates proved, never a charge against the worker.
|
|
1100
|
+
if (a.meta?.cause === "seat-launch-failed")
|
|
1101
|
+
a = { ...a, meta: { ...a.meta, classification: "infra", infra: true, retryable: false } };
|
|
1102
|
+
else if (!cancelled && Array.isArray(a.meta?.judgment))
|
|
1103
|
+
a = await adjudicate(a);
|
|
893
1104
|
// No dispatch, no key: a deterministic-oracle round writes no `invocations` field rather than an
|
|
894
1105
|
// empty array a reader could mistake for "measured, and it cost nothing".
|
|
895
1106
|
return { result: invocationSpans.length ? { ...a, meta: { ...a.meta, invocations: invocationSpans } } : a, invocations };
|
|
@@ -908,6 +1119,8 @@ export async function runGates(task, ctx) {
|
|
|
908
1119
|
const captured = await captureLlmDispatches(ctx.adapters, run);
|
|
909
1120
|
invocations.push(...captured.invocations);
|
|
910
1121
|
const rv = captured.value;
|
|
1122
|
+
if (cancelled)
|
|
1123
|
+
return rv; // a cancelled seat's non-answer says nothing about the seat
|
|
911
1124
|
if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
|
|
912
1125
|
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
|
|
913
1126
|
if (typeof rv.meta.reviewer === "string") {
|
|
@@ -938,7 +1151,7 @@ export async function runGates(task, ctx) {
|
|
|
938
1151
|
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
939
1152
|
const carriedAuthors = ctx.carriedAuthors ?? [];
|
|
940
1153
|
let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
|
|
941
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg,
|
|
1154
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
|
|
942
1155
|
// OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
|
|
943
1156
|
// a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
|
|
944
1157
|
// verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -948,9 +1161,29 @@ export async function runGates(task, ctx) {
|
|
|
948
1161
|
let retryPrior = [...priorReviewers];
|
|
949
1162
|
const routes = [];
|
|
950
1163
|
let hop = 0;
|
|
951
|
-
|
|
1164
|
+
// OBS-1196: seats already re-asked after a malformed verdict — once per seat, so never unbounded.
|
|
1165
|
+
const reemitted = new Set();
|
|
1166
|
+
while (!cancelled && (rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
|
|
952
1167
|
hop++;
|
|
953
1168
|
const flaked = rv.meta.reviewer;
|
|
1169
|
+
const retryVia = via
|
|
1170
|
+
? { ...via, nameFor: (role, adapter) => via.nameFor(role, adapter) + `-r${hop}` }
|
|
1171
|
+
: undefined;
|
|
1172
|
+
// OBS-1196: a malformed (unparseable) verdict is a delivery defect of THIS seat, not a reason to drop it. Ask the
|
|
1173
|
+
// same seat once more on the same subject; reviewGate mints a fresh nonce, so only a new, whole,
|
|
1174
|
+
// nonce-bound verdict can answer — the malformed bytes are never salvaged into one.
|
|
1175
|
+
if (rv.meta.cause === "malformed-verdict" && rv.meta.closureInvalid !== true && !reemitted.has(flaked)) {
|
|
1176
|
+
reemitted.add(flaked);
|
|
1177
|
+
const others = ctx.channels.map(channelKey).filter((key) => key !== flaked);
|
|
1178
|
+
const again = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, others, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
|
|
1179
|
+
if (again.meta?.noEligibleReviewer !== true) {
|
|
1180
|
+
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-reemission", payload: { reviewer: flaked, cause: "malformed-verdict",
|
|
1181
|
+
delivered: again.meta?.unparseable !== true && again.meta?.noVerdict !== true }, result: again });
|
|
1182
|
+
routes.push(`review re-emission (same seat, fresh nonce): ${flaked} produced a malformed verdict; asked once more`);
|
|
1183
|
+
rv = { ...again, details: `${routes.join("\n")}\n${again.details}`, meta: { ...again.meta, reviewReemission: { reviewer: flaked } } };
|
|
1184
|
+
continue;
|
|
1185
|
+
}
|
|
1186
|
+
}
|
|
954
1187
|
const emptyOutput = rv.meta.cause === "empty-output";
|
|
955
1188
|
if (emptyOutput) {
|
|
956
1189
|
await ctx.onGate?.({
|
|
@@ -959,9 +1192,6 @@ export async function runGates(task, ctx) {
|
|
|
959
1192
|
result: { ...rv, meta: { ...rv.meta, skipped: true } },
|
|
960
1193
|
});
|
|
961
1194
|
}
|
|
962
|
-
const retryVia = ctx.via
|
|
963
|
-
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + `-r${hop}` }
|
|
964
|
-
: undefined;
|
|
965
1195
|
const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
|
|
966
1196
|
const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
|
|
967
1197
|
// RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
|
|
@@ -995,11 +1225,14 @@ export async function runGates(task, ctx) {
|
|
|
995
1225
|
break;
|
|
996
1226
|
}
|
|
997
1227
|
}
|
|
998
|
-
if (rv.meta?.noVerdict === true
|
|
1228
|
+
if (rv.meta?.noVerdict === true
|
|
1229
|
+
|| (rv.meta?.unparseable === true && rv.meta.closureInvalid !== true && typeof rv.meta.reviewer === "string")) {
|
|
999
1230
|
// Terminal: every eligible seat returned no verdict. The carried materials stay open — an infra
|
|
1000
1231
|
// row is not a passing review — and the sibling judge result is untouched beside it.
|
|
1232
|
+
// OBS-1196: an exhausted pool of UNDELIVERED (unparseable) verdicts is the same non-verdict: an infra
|
|
1233
|
+
// park, never a worker retry. A delivered verdict that breaks the closure protocol keeps its path.
|
|
1001
1234
|
const carried = (ctx.carriedFindings ?? []).filter((f) => f.class === "review:material").map((f) => f.fingerprint);
|
|
1002
|
-
rv = { ...rv, meta: { ...rv.meta, classification: "infra", infra: true, carriedFindings: carried } };
|
|
1235
|
+
rv = { ...rv, meta: { ...rv.meta, noVerdict: true, classification: "infra", infra: true, carriedFindings: carried } };
|
|
1003
1236
|
}
|
|
1004
1237
|
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
|
1005
1238
|
};
|
|
@@ -1058,11 +1291,27 @@ export async function runGates(task, ctx) {
|
|
|
1058
1291
|
else
|
|
1059
1292
|
selected = [...new Set([...selected, ...required])].sort();
|
|
1060
1293
|
}
|
|
1061
|
-
|
|
1294
|
+
// OBS-635: a screen buys nothing when the full suite that must follow it already has a qualified
|
|
1295
|
+
// green on this exact identity (read BEFORE any screen), or when the harness-measured screen costs
|
|
1296
|
+
// at least 75 % of it — then the full suite runs in the battery instead. Unknown timing keeps the
|
|
1297
|
+
// screen. A selected-only green never answers here: its identity names its selection.
|
|
1298
|
+
let promotion;
|
|
1299
|
+
if (selected) {
|
|
1300
|
+
const hit = verdictStore.get(await fullTestIdentity());
|
|
1301
|
+
const costRatio = screenCostRatio(ctx.baseline, selected);
|
|
1302
|
+
if (hit?.pass === true && !isInfraResult(hit) && await certifiesFullManifest(hit))
|
|
1303
|
+
promotion = { reason: "full-green-cache" };
|
|
1304
|
+
else if (costRatio !== undefined && costRatio >= SCREEN_PROMOTION_RATIO)
|
|
1305
|
+
promotion = { reason: "screen-cost-promoted", costRatio };
|
|
1306
|
+
if (promotion)
|
|
1307
|
+
selected = undefined;
|
|
1308
|
+
}
|
|
1309
|
+
if (ctx.selectionReason || promotion)
|
|
1062
1310
|
selectionDecision = {
|
|
1063
|
-
scope: selected ? "selected" : "full", reason: selected ? selectionReason
|
|
1064
|
-
: selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason,
|
|
1311
|
+
scope: selected ? "selected" : "full", reason: promotion?.reason ?? (selected ? selectionReason
|
|
1312
|
+
: selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason),
|
|
1065
1313
|
requiredFiles: [...(ctx.requiredRepairTests ?? [])],
|
|
1314
|
+
...(promotion?.costRatio !== undefined ? { costRatio: promotion.costRatio } : {}),
|
|
1066
1315
|
};
|
|
1067
1316
|
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
|
|
1068
1317
|
if (failed())
|
|
@@ -1086,28 +1335,73 @@ export async function runGates(task, ctx) {
|
|
|
1086
1335
|
// enough: an acceptance-first await withholds a completed review behind a slow/hung judge and a
|
|
1087
1336
|
// process death can lose that already-earned verdict. The returned result is still sorted into
|
|
1088
1337
|
// GATE_NAMES order by done(); the event stream truthfully records each independent completion.
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
if (
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
}
|
|
1099
|
-
|
|
1100
|
-
|
|
1338
|
+
// OBS-1168(c): a sibling that throws, or that ends seatless (no seat could launch, so the round can
|
|
1339
|
+
// only park infra), cancels the other — its seats are closed — and the round still AWAITS it, with
|
|
1340
|
+
// or without an execution policy, so no verdict of a settled round publishes after its engagement.
|
|
1341
|
+
const seatless = (r) => r.meta?.cause === "seat-launch-failed" && r.meta?.infra === true;
|
|
1342
|
+
let failure;
|
|
1343
|
+
const fail = (reason) => {
|
|
1344
|
+
if (!cancelled)
|
|
1345
|
+
failure ??= { reason };
|
|
1346
|
+
cancelSemantic();
|
|
1347
|
+
};
|
|
1348
|
+
const judged = judging?.then(async (outcome) => {
|
|
1349
|
+
if (cancelled)
|
|
1350
|
+
return;
|
|
1351
|
+
await withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result));
|
|
1352
|
+
if (seatless(outcome.result))
|
|
1353
|
+
cancelSemantic();
|
|
1354
|
+
}).catch(fail);
|
|
1355
|
+
const reviewed = reviewing?.then(async (outcome) => {
|
|
1356
|
+
if (cancelled)
|
|
1357
|
+
return;
|
|
1358
|
+
await record(outcome);
|
|
1359
|
+
if (seatless(outcome))
|
|
1360
|
+
cancelSemantic();
|
|
1361
|
+
}).catch(fail);
|
|
1362
|
+
// ponytail: a headless seat has no slot to close; it is awaited to its own timeout, never abandoned.
|
|
1363
|
+
await Promise.all([judged, reviewed]);
|
|
1364
|
+
await Promise.all(closing);
|
|
1365
|
+
if (failure)
|
|
1366
|
+
throw failure.reason;
|
|
1367
|
+
executionSignal()?.throwIfAborted();
|
|
1101
1368
|
if (failed())
|
|
1102
1369
|
return done();
|
|
1103
1370
|
}
|
|
1371
|
+
// OBS-635: the semantic gates ran oracles and vendor CLIs in this worktree after an in-battery full
|
|
1372
|
+
// suite spoke, and its green stands only for the identity it measured. Oracle dirt withdraws it; a
|
|
1373
|
+
// changed or unmeasurable identity — tree, command, baseline, environment, dependency resolution,
|
|
1374
|
+
// capacity, protocol, lifecycle, full manifest — buys a fresh merge-candidate suite below.
|
|
1375
|
+
let rerunFull = false;
|
|
1376
|
+
listing = undefined;
|
|
1377
|
+
if (fullInBattery && ctx.commands.test !== undefined && (enabled("acceptance") || enabled("review"))) {
|
|
1378
|
+
const dirt = await dirtyWorktree();
|
|
1379
|
+
if (dirt) {
|
|
1380
|
+
const refusal = withTelemetry(await dirtyRoundRefusal("test", dirt));
|
|
1381
|
+
results[results.findIndex((r) => r.gate === "test")] = refusal;
|
|
1382
|
+
await ctx.onGate?.({ phase: "end", gate: "test", result: refusal });
|
|
1383
|
+
return done();
|
|
1384
|
+
}
|
|
1385
|
+
const now = await fullTestIdentity();
|
|
1386
|
+
// D-598: an unmeasurable lifecycle on either side is not comparable to anything (VerdictStore R41
|
|
1387
|
+
// refuses it); two `unknown`s hashing equal is not an unchanged identity, so the suite reruns.
|
|
1388
|
+
const measurable = (id) => id?.envParts?.verification?.lifecycle !== "unknown";
|
|
1389
|
+
rerunFull = !now || !batteryFullIdentity || !measurable(now) || !measurable(batteryFullIdentity)
|
|
1390
|
+
|| verificationIdentityKey(now) !== verificationIdentityKey(batteryFullIdentity)
|
|
1391
|
+
|| !(await certifiesFullManifest(results.find((r) => r.gate === "test")));
|
|
1392
|
+
if (rerunFull) {
|
|
1393
|
+
// The in-battery row keeps its own interval; the replacement measures from zero.
|
|
1394
|
+
spans.delete("test");
|
|
1395
|
+
loadSamples.delete("test");
|
|
1396
|
+
}
|
|
1397
|
+
}
|
|
1104
1398
|
// The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
|
|
1105
1399
|
// the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
|
|
1106
1400
|
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict replaces the
|
|
1107
1401
|
// screen's entry in the returned record (one `test` entry), and `fullSuite` says which suite spoke
|
|
1108
1402
|
// while `selectedTests` keeps what the screen ran. In the stream, a held screen is superseded (one
|
|
1109
1403
|
// `test` end event); a published screen keeps its own earlier event and this is the second.
|
|
1110
|
-
if (selected) {
|
|
1404
|
+
if (selected || rerunFull) {
|
|
1111
1405
|
await emitStart("test");
|
|
1112
1406
|
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
1113
1407
|
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
@@ -1118,20 +1412,12 @@ export async function runGates(task, ctx) {
|
|
|
1118
1412
|
let cached = false;
|
|
1119
1413
|
let identity;
|
|
1120
1414
|
if (ctx.commands.test !== undefined) {
|
|
1121
|
-
identity = await
|
|
1122
|
-
worktree: ctx.worktree,
|
|
1123
|
-
gate: "test",
|
|
1124
|
-
scope: ctx.verificationScope,
|
|
1125
|
-
command: ctx.commands.test,
|
|
1126
|
-
baseline: ctx.baseline,
|
|
1127
|
-
selectedSet: undefined,
|
|
1128
|
-
capacity: resolvedCapacity(),
|
|
1129
|
-
});
|
|
1415
|
+
identity = await fullTestIdentity();
|
|
1130
1416
|
const hit = verdictStore.get(identity);
|
|
1131
1417
|
if (hit)
|
|
1132
1418
|
classifySignalOnlyTest(hit);
|
|
1133
1419
|
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
1134
|
-
&& !(await discardCachedRed("test", hit))) {
|
|
1420
|
+
&& !(await discardCachedRed("test", hit)) && await certifiesFullManifest(hit)) {
|
|
1135
1421
|
full = formatReusedRow(hit, identity);
|
|
1136
1422
|
cached = true;
|
|
1137
1423
|
await noteReuse("test", full, identity);
|
|
@@ -80,11 +80,49 @@ export declare function verifyManifestReport(opts: {
|
|
|
80
80
|
report: TestReport | undefined;
|
|
81
81
|
killedFile?: string;
|
|
82
82
|
hangBudgetMs?: number;
|
|
83
|
+
/** Active elapsed time (detected suspend subtracted) versus raw wall service at the kill. */
|
|
84
|
+
hangActiveMs?: number;
|
|
85
|
+
hangWallMs?: number;
|
|
83
86
|
}): ManifestVerdict;
|
|
84
87
|
/** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
|
|
85
88
|
export declare const FILE_HANG_SLACK = 3;
|
|
86
89
|
export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
|
|
87
90
|
export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
|
|
91
|
+
/**
|
|
92
|
+
* OBS-953 (+add): a per-file hang budget counted in raw wall time charged a lid-close to the file that
|
|
93
|
+
* was running. Each poll compares how far the wall clock and the monotonic clock advanced since the
|
|
94
|
+
* last one. Wall advancing more than monotonic by over CLOCK_JUMP_SLACK_MS is a DETECTED discontinuity
|
|
95
|
+
* (`host-suspend`): its offset is subtracted from the active elapsed time of every file started before
|
|
96
|
+
* it ended. A poll that arrives overdue while both clocks advanced together is recorded as `unknown`
|
|
97
|
+
* and subtracts nothing — an ambiguous gap never excuses an active hang. Whether the host really slept
|
|
98
|
+
* is not provable from these clocks (a manually set clock jumps the same way); only the offset is.
|
|
99
|
+
*/
|
|
100
|
+
export declare const CLOCK_JUMP_SLACK_MS = 1000;
|
|
101
|
+
export interface HostInterruption {
|
|
102
|
+
kind: "host-suspend" | "unknown";
|
|
103
|
+
/** Wall time of the previous poll and of the poll that observed the gap. */
|
|
104
|
+
from: number;
|
|
105
|
+
to: number;
|
|
106
|
+
wallMs: number;
|
|
107
|
+
monoMs: number;
|
|
108
|
+
/** The wall-over-monotonic offset of a detected suspend; 0 for an unknown gap. */
|
|
109
|
+
subtractedMs: number;
|
|
110
|
+
}
|
|
111
|
+
export interface HangClocks {
|
|
112
|
+
wall: () => number;
|
|
113
|
+
mono: () => number;
|
|
114
|
+
}
|
|
115
|
+
export declare const setHangClocksForTests: (clocks: HangClocks) => void;
|
|
116
|
+
export declare const resetHangClocksForTests: () => void;
|
|
117
|
+
/** A poll gap worth recording, or undefined for an on-time poll. */
|
|
118
|
+
export declare function classifyPollGap(prev: {
|
|
119
|
+
wall: number;
|
|
120
|
+
mono: number;
|
|
121
|
+
}, now: {
|
|
122
|
+
wall: number;
|
|
123
|
+
mono: number;
|
|
124
|
+
}, pollMs: number): HostInterruption | undefined;
|
|
125
|
+
export declare const runWithInterruptionSink: <T>(sink: (interruption: HostInterruption) => void, run: () => Promise<T>) => Promise<T>;
|
|
88
126
|
export interface ManifestRunResult {
|
|
89
127
|
evidenceReceipt: GateEvidenceReceipt;
|
|
90
128
|
evidenceReceipts: GateEvidenceReceipt[];
|
|
@@ -94,6 +132,10 @@ export interface ManifestRunResult {
|
|
|
94
132
|
report: TestReport | undefined;
|
|
95
133
|
killedFile?: string;
|
|
96
134
|
hangBudgetMs?: number;
|
|
135
|
+
/** A hang's active elapsed time (suspend subtracted) and its raw wall service, kept apart. */
|
|
136
|
+
hangActiveMs?: number;
|
|
137
|
+
hangWallMs?: number;
|
|
138
|
+
interruptions: HostInterruption[];
|
|
97
139
|
/** The child's own pid (its process GROUP id too, since it is spawned detached) — for a caller
|
|
98
140
|
* that wants to prove the group is really gone after a hang kill (`process.kill(-pid, 0)` throws). */
|
|
99
141
|
pid?: number;
|