tickmarkr 2.6.1 → 2.6.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -3
- package/dist/adapters/catalog-remote.js +89 -47
- package/dist/adapters/claude-code.js +9 -6
- package/dist/adapters/codex.js +7 -4
- package/dist/adapters/prompt.d.ts +1 -0
- package/dist/adapters/prompt.js +14 -6
- package/dist/adapters/registry.js +3 -3
- package/dist/adapters/types.d.ts +12 -4
- package/dist/adapters/types.js +6 -0
- package/dist/cli/commands/approve.d.ts +11 -4
- package/dist/cli/commands/approve.js +82 -27
- package/dist/cli/commands/compile.js +13 -3
- package/dist/cli/commands/doctor.d.ts +8 -2
- package/dist/cli/commands/doctor.js +11 -3
- package/dist/cli/commands/fleet.js +87 -11
- package/dist/cli/commands/plan.js +13 -8
- package/dist/cli/commands/report.d.ts +2 -1
- package/dist/cli/commands/report.js +74 -8
- package/dist/cli/commands/resume.js +4 -2
- package/dist/cli/commands/status.js +43 -20
- package/dist/cli/help.d.ts +2 -0
- package/dist/cli/help.js +9 -2
- package/dist/compile/native.js +7 -0
- package/dist/config/config.d.ts +35 -2
- package/dist/config/config.js +86 -10
- package/dist/config/fleet-overlay.d.ts +13 -2
- package/dist/config/fleet-overlay.js +60 -0
- package/dist/drivers/herdr.d.ts +12 -0
- package/dist/drivers/herdr.js +51 -0
- package/dist/drivers/orca.d.ts +35 -2
- package/dist/drivers/orca.js +222 -67
- package/dist/drivers/types.d.ts +2 -0
- package/dist/drivers/types.js +2 -2
- package/dist/eval/canary.d.ts +2 -1
- package/dist/eval/canary.js +2 -2
- package/dist/eval/dispatch.js +1 -0
- package/dist/gates/acceptance.d.ts +9 -1
- package/dist/gates/acceptance.js +31 -4
- package/dist/gates/baseline.d.ts +32 -2
- package/dist/gates/baseline.js +111 -24
- package/dist/gates/cache.d.ts +8 -0
- package/dist/gates/cache.js +12 -2
- package/dist/gates/llm.d.ts +11 -4
- package/dist/gates/llm.js +40 -21
- package/dist/gates/review.d.ts +14 -1
- package/dist/gates/review.js +160 -34
- package/dist/gates/run-gates.d.ts +56 -4
- package/dist/gates/run-gates.js +358 -58
- package/dist/gates/test-manifest.d.ts +45 -1
- package/dist/gates/test-manifest.js +78 -12
- package/dist/graph/schema.d.ts +2 -0
- package/dist/graph/schema.js +2 -0
- package/dist/plan/scope.js +2 -2
- package/dist/route/preference.d.ts +20 -2
- package/dist/route/preference.js +48 -13
- package/dist/route/router.d.ts +12 -1
- package/dist/route/router.js +56 -24
- package/dist/run/consult.d.ts +15 -1
- package/dist/run/consult.js +18 -7
- package/dist/run/daemon.d.ts +38 -2
- package/dist/run/daemon.js +895 -192
- package/dist/run/git.d.ts +8 -0
- package/dist/run/git.js +14 -0
- package/dist/run/interactive-seed.d.ts +4 -0
- package/dist/run/interactive-seed.js +35 -9
- package/dist/run/journal.d.ts +152 -3
- package/dist/run/journal.js +551 -50
- package/dist/run/lease.d.ts +13 -0
- package/dist/run/lease.js +45 -0
- package/dist/run/merge.d.ts +3 -1
- package/dist/run/merge.js +3 -2
- package/dist/run/operator-summary.d.ts +3 -0
- package/dist/run/operator-summary.js +3 -1
- package/dist/run/protocol.d.ts +46 -1
- package/dist/run/protocol.js +14 -2
- package/dist/run/receipt-resolver.d.ts +22 -0
- package/dist/run/receipt-resolver.js +40 -1
- package/dist/run/repair-selection.d.ts +11 -1
- package/dist/run/repair-selection.js +17 -9
- package/dist/run/supervision.d.ts +7 -1
- package/dist/run/supervision.js +5 -2
- package/dist/run/wall-budget.d.ts +48 -0
- package/dist/run/wall-budget.js +280 -0
- package/dist/tui/cockpit/board.js +3 -3
- package/dist/tui/cockpit/decision-actions.d.ts +8 -5
- package/dist/tui/cockpit/decision-actions.js +55 -32
- package/dist/tui/cockpit/derive.js +13 -2
- package/dist/tui/cockpit/live-runtime.d.ts +10 -0
- package/dist/tui/cockpit/live-runtime.js +50 -3
- package/dist/tui/cockpit/run-cockpit.d.ts +3 -0
- package/dist/tui/cockpit/run-cockpit.js +27 -2
- package/dist/tui/cockpit/run-view.d.ts +9 -2
- package/dist/tui/cockpit/run-view.js +66 -9
- package/dist/tui/cockpit/setup-cockpit.d.ts +6 -0
- package/dist/tui/cockpit/setup-cockpit.js +10 -3
- package/dist/tui/ink/fleet-app.d.ts +15 -3
- package/dist/tui/ink/fleet-app.js +91 -22
- package/package.json +3 -1
- package/schema/config.schema.json +825 -0
- package/skills/tickmarkr-loop/SKILL.md +15 -3
- package/skills/tickmarkr-overseer/SKILL.md +42 -0
- package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +91 -0
- package/skills/tickmarkr-overseer/scripts/context-statusline.sh +81 -0
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +36 -34
- package/skills/tickmarkr-overseer/scripts/watch-journal.sh +6 -4
package/dist/gates/run-gates.js
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
import { randomUUID } from "node:crypto";
|
|
1
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
2
2
|
import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
|
|
3
3
|
import { loadavg, tmpdir } from "node:os";
|
|
4
4
|
import { join, posix } from "node:path";
|
|
5
|
-
import { channelKey, shq } from "../adapters/types.js";
|
|
5
|
+
import { channelKey, configuredEffort, shq } from "../adapters/types.js";
|
|
6
6
|
import { TIER_RANK } from "../config/config.js";
|
|
7
7
|
import { getAdapter } from "../adapters/registry.js";
|
|
8
8
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
@@ -10,14 +10,14 @@ import { acceptanceGate } from "./acceptance.js";
|
|
|
10
10
|
import { compareToBaseline, effectiveCeilingMs, waitForCalmWindow, calmWindowReady } from "./baseline.js";
|
|
11
11
|
import { evidenceGate } from "./evidence.js";
|
|
12
12
|
import { captureLlmOutput } from "./llm.js";
|
|
13
|
-
import { disallowedBy } from "../route/preference.js";
|
|
13
|
+
import { disallowedBy, observedSeat } from "../route/preference.js";
|
|
14
14
|
import { marginalCostRank } from "../route/router.js";
|
|
15
15
|
import { carriedAuthorVendors, gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
|
|
16
16
|
import { scopeGate } from "./scope.js";
|
|
17
|
-
import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
|
|
17
|
+
import { discoverTestManifest, evaluateManifestedTest, isVitestTestCommand, VITEST_CACHE_ENV, worktreeVitestCache } from "./test-manifest.js";
|
|
18
18
|
import { executionSignal } from "../run/execution-budget.js";
|
|
19
19
|
import { failureDisposition } from "../run/recovery.js";
|
|
20
|
-
import { dependencyLinkRefusal, preserveWorktree, producerFields, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
|
|
20
|
+
import { dependencyLinkRefusal, FORK_CAP_ENV, preserveWorktree, producerFields, ROUTING_ENV_SEAMS, shGit, resolvedCapacity, SUITE_PARENT_ENV, verificationProtocol } from "../run/git.js";
|
|
21
21
|
import { withJudgeInvocationEvidence } from "../run/journal.js";
|
|
22
22
|
import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
|
|
23
23
|
const productionLoadProvider = () => loadavg()[0] ?? 0;
|
|
@@ -43,13 +43,14 @@ function instrumentLlmAdapter(adapter, clocks) {
|
|
|
43
43
|
return new Proxy(adapter, {
|
|
44
44
|
get(target, property) {
|
|
45
45
|
if (property === "headlessCommand") {
|
|
46
|
-
return (promptFile, model) => {
|
|
47
|
-
const command = target.headlessCommand(promptFile, model);
|
|
46
|
+
return (promptFile, model, effort) => {
|
|
47
|
+
const command = target.headlessCommand(promptFile, model, effort);
|
|
48
48
|
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-gate-invocation-"));
|
|
49
49
|
const startedAtPath = join(dir, "started-at");
|
|
50
50
|
const completedAtPath = join(dir, "completed-at");
|
|
51
51
|
clocks.push({
|
|
52
52
|
channel: channelKey({ adapter: target.id, model }),
|
|
53
|
+
...(effort ? { effort } : {}),
|
|
53
54
|
preparedAt: Date.now(),
|
|
54
55
|
startedAtPath,
|
|
55
56
|
completedAtPath,
|
|
@@ -85,7 +86,7 @@ function finishLlmDispatches(clocks) {
|
|
|
85
86
|
finally {
|
|
86
87
|
rmSync(clock.dir, { recursive: true, force: true });
|
|
87
88
|
}
|
|
88
|
-
return { channel: clock.channel, durationMs: completedAt - startedAt };
|
|
89
|
+
return { channel: clock.channel, ...(clock.effort ? { effort: clock.effort } : {}), durationMs: completedAt - startedAt };
|
|
89
90
|
});
|
|
90
91
|
}
|
|
91
92
|
async function captureLlmDispatches(adapters, run) {
|
|
@@ -214,6 +215,44 @@ export function testCommandForFiles(testCmd, files) {
|
|
|
214
215
|
const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
|
|
215
216
|
return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
|
|
216
217
|
}
|
|
218
|
+
/** OBS-635: a screen costing at least this share of the full suite runs the full suite instead. */
|
|
219
|
+
export const SCREEN_PROMOTION_RATIO = 0.75;
|
|
220
|
+
/**
|
|
221
|
+
* The screen's share of the full suite's cost, from the per-file durations the harness measured at
|
|
222
|
+
* baseline capture — never a worker's timing. Undefined (unknown) unless every selected file has a
|
|
223
|
+
* measured duration and the measured total is positive; unknown keeps the conservative screen path.
|
|
224
|
+
*/
|
|
225
|
+
export function screenCostRatio(baseline, selected) {
|
|
226
|
+
const entry = baseline.commands.test;
|
|
227
|
+
const files = entry?.infra ? undefined : entry?.fileDurations;
|
|
228
|
+
if (!files?.length)
|
|
229
|
+
return undefined;
|
|
230
|
+
const cost = new Map(files.map((f) => [f.file, f.durationMs]));
|
|
231
|
+
const total = files.reduce((sum, f) => sum + f.durationMs, 0);
|
|
232
|
+
if (!(total > 0) || selected.some((file) => !cost.has(file)))
|
|
233
|
+
return undefined;
|
|
234
|
+
return selected.reduce((sum, file) => sum + cost.get(file), 0) / total;
|
|
235
|
+
}
|
|
236
|
+
/** OBS-635: the full manifest the runner lists NOW, under the environment evaluateManifestedTest's own
|
|
237
|
+
* discovery receives (test-manifest.ts manifestEnvironment), so it compares with the one a verdict
|
|
238
|
+
* certified. Undefined when the runner cannot list — an unlisted manifest certifies nothing. */
|
|
239
|
+
async function listFullManifest(cmd, worktree) {
|
|
240
|
+
const env = { ...process.env, PATH: `${join(worktree, "node_modules/.bin")}:${process.env.PATH ?? ""}`,
|
|
241
|
+
[VITEST_CACHE_ENV]: worktreeVitestCache(worktree),
|
|
242
|
+
[FORK_CAP_ENV]: String(resolvedCapacity().forkCap), [SUITE_PARENT_ENV]: String(process.pid) };
|
|
243
|
+
for (const key of [...ROUTING_ENV_SEAMS, "VITEST", "TEST", "VITEST_WORKER_ID", "VITEST_POOL_ID"])
|
|
244
|
+
delete env[key];
|
|
245
|
+
const dir = mkdtempSync(join(tmpdir(), "tickmarkr-full-manifest-"));
|
|
246
|
+
try {
|
|
247
|
+
return (await discoverTestManifest(cmd, worktree, { dir, nonce: randomUUID(), env })).files;
|
|
248
|
+
}
|
|
249
|
+
catch {
|
|
250
|
+
return undefined;
|
|
251
|
+
}
|
|
252
|
+
finally {
|
|
253
|
+
rmSync(dir, { recursive: true, force: true });
|
|
254
|
+
}
|
|
255
|
+
}
|
|
217
256
|
/** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
|
|
218
257
|
async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry = {}, retried = false) {
|
|
219
258
|
const entry = baseline.commands.test;
|
|
@@ -248,6 +287,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
|
|
|
248
287
|
meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
|
|
249
288
|
};
|
|
250
289
|
}
|
|
290
|
+
/** A retry base no runner invocation parses. evaluateManifestedTest builds its stranded single-fork
|
|
291
|
+
* retry from the base it is handed, and one it cannot parse throws before any spawn — so this base
|
|
292
|
+
* disables that inner recovery: a worker-RPC-stranded re-observation comes back infra (the caller
|
|
293
|
+
* parks it as ambiguous) instead of launching a second execution. */
|
|
294
|
+
export const REOBSERVATION_RETRY_BASE = "tickmarkr-reobservation-refuses-stranded-retry";
|
|
295
|
+
/** OBS-1106 residual: ONE isolated re-observation of a timeout-shaped red's attributed failing files on
|
|
296
|
+
* the same checkout, narrowed exactly as a screen is. Never cached and never a verdict: the caller
|
|
297
|
+
* keeps the original red and reads this only to decide whether that red is chargeable. Exactly one
|
|
298
|
+
* execution — the bounded infra/host-starved retries and the stranded single-fork recovery are all
|
|
299
|
+
* refused, so a diagnostic never buys more. */
|
|
300
|
+
export async function reobserveTestFiles(worktree, testCmd, baseline, files, artifactDir) {
|
|
301
|
+
const cmd = testCommandForFiles(testCmd, files);
|
|
302
|
+
if (!isVitestTestCommand(testCmd, worktree))
|
|
303
|
+
return (await compareToBaseline(worktree, { test: cmd }, baseline, ["test"], { selected: files, authorizeRetry: () => false }))[0];
|
|
304
|
+
const r = await runVitestManifestGate(worktree, cmd, baseline, files, artifactDir, { retryBaseCommand: REOBSERVATION_RETRY_BASE });
|
|
305
|
+
// fail closed whatever a recovery did: a re-observation never reads a recovered verdict
|
|
306
|
+
return r.meta?.recovery === undefined ? r
|
|
307
|
+
: { ...r, pass: false, meta: { ...r.meta, classification: "infra", infra: true, retryable: false, recoveryRefused: true } };
|
|
308
|
+
}
|
|
251
309
|
const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
|
|
252
310
|
const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
|
|
253
311
|
/** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
|
|
@@ -261,12 +319,56 @@ function classifySignalOnlyTest(g) {
|
|
|
261
319
|
return;
|
|
262
320
|
g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
|
|
263
321
|
}
|
|
322
|
+
/** OBS-1151: a criterion's comparable subject — its canonical text, the task's declared bounds and the
|
|
323
|
+
* operator context. The cited files' blobs are compared separately, over the union of both citations. */
|
|
324
|
+
export function judgmentSubjectKey(task, criterion, operatorContext) {
|
|
325
|
+
return createHash("sha256").update(JSON.stringify([criterion, [...task.files].sort(), [...(task.outOfScope ?? [])].sort(), operatorContext ?? ""])).digest("hex");
|
|
326
|
+
}
|
|
327
|
+
async function blobAt(worktree, commit, path) {
|
|
328
|
+
const r = await shGit(`git rev-parse --verify --quiet ${shq(`${commit}:${path}`)}`, worktree);
|
|
329
|
+
return r.code === 0 && r.stdout.trim() ? r.stdout.trim() : undefined;
|
|
330
|
+
}
|
|
331
|
+
/** OBS-1151: the criteria whose fresh ruling reverses the newest prior ruling on the same subject key whose
|
|
332
|
+
* cited paths hold the identical blob at both commits (older comparable priors are still found behind a
|
|
333
|
+
* newer prior on different blobs). A citation-less side or an
|
|
334
|
+
* unreadable blob is unknown, never identical — that criterion's fresh ruling is simply fresh. */
|
|
335
|
+
export async function judgeContradictions(worktree, head, fresh, priors) {
|
|
336
|
+
const found = [];
|
|
337
|
+
for (const c of fresh) {
|
|
338
|
+
if (!c.paths.length)
|
|
339
|
+
continue;
|
|
340
|
+
// C-3 (D-669): the comparable prior is the NEWEST one on identical cited blobs, not the newest one
|
|
341
|
+
// carrying the key — PASS(A) → FAIL(B) → fresh FAIL(A) must still be adjudicated against PASS(A).
|
|
342
|
+
for (const prior of priors) {
|
|
343
|
+
const was = prior.criteria.find((q) => q.key === c.key);
|
|
344
|
+
if (!was || !was.paths.length)
|
|
345
|
+
continue;
|
|
346
|
+
const paths = [...new Set([...was.paths, ...c.paths])].sort();
|
|
347
|
+
let identical = true;
|
|
348
|
+
for (const path of paths) {
|
|
349
|
+
const [before, now] = await Promise.all([blobAt(worktree, prior.commit, path), blobAt(worktree, head, path)]);
|
|
350
|
+
if (!before || !now || before !== now) {
|
|
351
|
+
identical = false;
|
|
352
|
+
break;
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
if (!identical)
|
|
356
|
+
continue;
|
|
357
|
+
if (was.met !== c.met)
|
|
358
|
+
found.push({ id: c.id, met: c.met, priorMet: was.met, priorCommit: prior.commit, paths });
|
|
359
|
+
break;
|
|
360
|
+
}
|
|
361
|
+
}
|
|
362
|
+
return found;
|
|
363
|
+
}
|
|
264
364
|
export async function runGates(task, ctx) {
|
|
265
365
|
const results = [];
|
|
266
366
|
const evidence = {
|
|
267
367
|
artifactDir: ctx.artifactDir,
|
|
268
368
|
runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
|
|
269
369
|
taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
|
|
370
|
+
// OBS-1140: task and standalone gates honour the configured quota, not the built-in default.
|
|
371
|
+
quotaBytes: ctx.cfg.gates?.evidenceQuotaBytes,
|
|
270
372
|
...ctx.evidence,
|
|
271
373
|
};
|
|
272
374
|
// Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
|
|
@@ -304,6 +406,9 @@ export async function runGates(task, ctx) {
|
|
|
304
406
|
await receiptNotes;
|
|
305
407
|
};
|
|
306
408
|
let selectionDecision;
|
|
409
|
+
// OBS-635: the identity a full suite measured inside the battery, revalidated after semantics.
|
|
410
|
+
let fullInBattery = false;
|
|
411
|
+
let batteryFullIdentity;
|
|
307
412
|
let commits = [];
|
|
308
413
|
// Check before cache identity, npm policy probes, or any gate command.
|
|
309
414
|
const dependencyRefusal = dependencyLinkRefusal(ctx.worktree);
|
|
@@ -337,14 +442,41 @@ export async function runGates(task, ctx) {
|
|
|
337
442
|
payload: { gate, reason: "cached-red-discarded", ...(reason === "recheck" ? {} : { bypass: reason }) }, result: hit });
|
|
338
443
|
return true;
|
|
339
444
|
};
|
|
445
|
+
// OBS-1168(c): every judge/review seat this round opens, so a failed sibling can cancel the other.
|
|
446
|
+
// A cancelled round dispatches, re-routes and publishes nothing more; its closed seats' own errors
|
|
447
|
+
// are consequences of the cancel, never a second failure.
|
|
448
|
+
// A seat whose creation was still pending at the cancel is refused before dispatch: onSlot runs inside
|
|
449
|
+
// llm.ts's launch guard, so the throw closes (and awaits) that half-launched pane and never runs it.
|
|
450
|
+
const semanticSlots = new Set();
|
|
451
|
+
const closing = [];
|
|
452
|
+
const via = ctx.via && {
|
|
453
|
+
...ctx.via, onSlot: (slot) => {
|
|
454
|
+
semanticSlots.add(slot);
|
|
455
|
+
ctx.via.onSlot?.(slot);
|
|
456
|
+
if (cancelled)
|
|
457
|
+
throw new Error("semantic round cancelled before dispatch");
|
|
458
|
+
},
|
|
459
|
+
};
|
|
460
|
+
let cancelled = false;
|
|
461
|
+
const cancelSemantic = () => {
|
|
462
|
+
if (cancelled)
|
|
463
|
+
return;
|
|
464
|
+
cancelled = true;
|
|
465
|
+
for (const slot of semanticSlots)
|
|
466
|
+
closing.push(via.driver.close(slot).catch(() => { }));
|
|
467
|
+
};
|
|
340
468
|
const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
|
|
341
469
|
const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
|
|
342
470
|
const failed = () => results.some((r) => !r.pass);
|
|
343
471
|
// T4 (OBS-265): a GREEN selected-test run is a screen, not the round's verdict — the merge-candidate
|
|
344
|
-
// round re-runs the full suite on the same commit and THAT is what the round reports.
|
|
345
|
-
//
|
|
472
|
+
// round re-runs the full suite on the same commit and THAT is what the round reports. With no
|
|
473
|
+
// semantic gate to act on it, the screen is held so its full suite speaks for it in one row.
|
|
346
474
|
// (A RED screen IS the verdict: the round ends there, so it is recorded immediately.)
|
|
347
475
|
let heldTest;
|
|
476
|
+
// OBS-1176: when acceptance/review WILL act on a green screen, the screen is published before they
|
|
477
|
+
// start, as its own selected row. The full suite afterwards is a second invocation on its own row —
|
|
478
|
+
// it carries only its own receipts and interval, so it neither erases nor re-counts the screen.
|
|
479
|
+
const publishScreen = enabled("acceptance") || enabled("review");
|
|
348
480
|
// v2.0 T2 (OBS-554): this round's per-gate measurement. Every interval a gate actually spends
|
|
349
481
|
// executing is added HERE, at the call site that runs it, so a gate that runs twice (the test
|
|
350
482
|
// gate's screen and its full suite) sums to its own cost and never to the span between them.
|
|
@@ -634,6 +766,28 @@ export async function runGates(task, ctx) {
|
|
|
634
766
|
},
|
|
635
767
|
};
|
|
636
768
|
};
|
|
769
|
+
const fullTestIdentity = () => computeVerificationIdentity({
|
|
770
|
+
worktree: ctx.worktree,
|
|
771
|
+
gate: "test",
|
|
772
|
+
scope: ctx.verificationScope,
|
|
773
|
+
command: ctx.commands.test,
|
|
774
|
+
baseline: ctx.baseline,
|
|
775
|
+
selectedSet: undefined,
|
|
776
|
+
capacity: resolvedCapacity(),
|
|
777
|
+
});
|
|
778
|
+
// OBS-635: a full green answers only for the manifest it certified. The tree identity cannot see an
|
|
779
|
+
// ignored generated test the runner would collect, so every full-green reuse rediscovers the
|
|
780
|
+
// runner's listing and requires the verdict's to equal it; a runner without a listing is bound by
|
|
781
|
+
// its identity alone. Reads before the semantic gates share one listing; `listing` resets after them.
|
|
782
|
+
let listing;
|
|
783
|
+
const certifiesFullManifest = async (verdict) => {
|
|
784
|
+
if (!verdict.pass || !isVitestTestCommand(ctx.commands.test, ctx.worktree))
|
|
785
|
+
return true;
|
|
786
|
+
const certified = verdict.meta?.manifest;
|
|
787
|
+
const current = await (listing ??= listFullManifest(ctx.commands.test, ctx.worktree));
|
|
788
|
+
return Array.isArray(certified) && current !== undefined
|
|
789
|
+
&& certified.length === current.length && [...certified].sort().every((file, i) => file === current[i]);
|
|
790
|
+
};
|
|
637
791
|
// shell tools vs the shared baseline
|
|
638
792
|
const retryOptions = (identity) => ctx.authorizeInfraRetry
|
|
639
793
|
? { authorizeRetry: (cause) => ctx.authorizeInfraRetry(identity ? verificationIdentityKey(identity) : "", cause === "infra" ? "infrastructure" : "host-starved") }
|
|
@@ -663,7 +817,7 @@ export async function runGates(task, ctx) {
|
|
|
663
817
|
if (hit)
|
|
664
818
|
classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
|
|
665
819
|
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
666
|
-
&& !(await discardCachedRed(g, hit))) {
|
|
820
|
+
&& !(await discardCachedRed(g, hit)) && (g !== "test" || selected !== undefined || await certifiesFullManifest(hit))) {
|
|
667
821
|
r = formatReusedRow(hit, identity);
|
|
668
822
|
cached = true;
|
|
669
823
|
if (g === "build")
|
|
@@ -690,6 +844,10 @@ export async function runGates(task, ctx) {
|
|
|
690
844
|
// the screen's interval IS the test gate's first interval, so the split needs no second clock
|
|
691
845
|
if (g === "test" && selected)
|
|
692
846
|
selectedDurationMs = spans.get("test")?.durationMs ?? 0;
|
|
847
|
+
if (g === "test" && !selected) {
|
|
848
|
+
fullInBattery = true;
|
|
849
|
+
batteryFullIdentity = identity;
|
|
850
|
+
}
|
|
693
851
|
// The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
|
|
694
852
|
// tracked file makes it dirty again, and every gate after it — including the next shell gate,
|
|
695
853
|
// which would then run against bytes HEAD does not hold — inherits that. So re-check after each
|
|
@@ -711,6 +869,13 @@ export async function runGates(task, ctx) {
|
|
|
711
869
|
const screened = { ...r, meta: { ...r.meta, selectedTests: selected } };
|
|
712
870
|
if (!screened.pass)
|
|
713
871
|
await record(screened);
|
|
872
|
+
else if (publishScreen) {
|
|
873
|
+
await record(screened);
|
|
874
|
+
// The screen's interval now lives on its own row; the full suite measures from zero.
|
|
875
|
+
spans.delete("test");
|
|
876
|
+
loadSamples.delete("test");
|
|
877
|
+
selectedDurationMs = undefined;
|
|
878
|
+
}
|
|
714
879
|
else {
|
|
715
880
|
heldTest = withTelemetry(screened);
|
|
716
881
|
results.push(heldTest);
|
|
@@ -790,7 +955,9 @@ export async function runGates(task, ctx) {
|
|
|
790
955
|
// v1.87 T2: the judge is a configured seat like any other — check it against the operator's
|
|
791
956
|
// policy BEFORE spending a dispatch on it. disallowedBy carries the whole deny grammar (adapter,
|
|
792
957
|
// model, or adapter:model), so a model-scoped deny cannot slip past an adapter-id-only read.
|
|
793
|
-
|
|
958
|
+
// OBS-1186: under the exact cached identity of that channel, as compile, doctor and route read it.
|
|
959
|
+
const judgeSeat = observedSeat(ctx.health, ctx.cfg.judge.adapter, ctx.cfg.judge.model);
|
|
960
|
+
const judgeDenied = disallowedBy(judgeSeat, ctx.cfg.routing, "judge");
|
|
794
961
|
if (judgeDenied) {
|
|
795
962
|
return {
|
|
796
963
|
result: {
|
|
@@ -803,8 +970,8 @@ export async function runGates(task, ctx) {
|
|
|
803
970
|
};
|
|
804
971
|
}
|
|
805
972
|
const judgeAdapter = getAdapter(ctx.cfg.judge.adapter, ctx.adapters);
|
|
806
|
-
const jvia =
|
|
807
|
-
? { driver:
|
|
973
|
+
const jvia = via
|
|
974
|
+
? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", judgeAdapter.id), label: via.labelFor("judge") }
|
|
808
975
|
: undefined;
|
|
809
976
|
// v1.19 (T2): testCmd threads the detected test runner to the gate so named-test oracles run
|
|
810
977
|
// deterministically (filtered via -t) before any LLM judge dispatch.
|
|
@@ -815,8 +982,8 @@ export async function runGates(task, ctx) {
|
|
|
815
982
|
// Separate from `invocations` above deliberately: that array is transcript evidence and records
|
|
816
983
|
// one entry per CAPTURED OUTPUT, so a dispatch that produced none contributes nothing to it.
|
|
817
984
|
const invocationSpans = [];
|
|
818
|
-
const invokeJudge = async (adapter, model, via) => {
|
|
819
|
-
const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
|
|
985
|
+
const invokeJudge = async (adapter, model, via, effort) => {
|
|
986
|
+
const captured = await captureLlmDispatches([adapter], ([instrumented]) => acceptanceGate(task, ctx.worktree, ctx.baseRef, { adapter: instrumented, model, effort }, via, { testCmd: ctx.commands.test, diffCap: ctx.cfg.gates.diffCap }));
|
|
820
987
|
// The instrumented adapter is reached only by runLlm. Deterministic oracles and diff-cap exits
|
|
821
988
|
// never call headlessCommand, so they produce no clock and cannot manufacture an invocation.
|
|
822
989
|
invocationSpans.push(...captured.invocations);
|
|
@@ -838,7 +1005,58 @@ export async function runGates(task, ctx) {
|
|
|
838
1005
|
}
|
|
839
1006
|
return captured.value;
|
|
840
1007
|
};
|
|
841
|
-
|
|
1008
|
+
const judgePool = () => (ctx.judgeChannels ?? []).filter((c) => disallowedBy(c, ctx.cfg.routing, "judge") === null);
|
|
1009
|
+
const rankJudges = (pool) => [...pool]
|
|
1010
|
+
.sort((x, y) => TIER_RANK[y.tier] - TIER_RANK[x.tier] || marginalCostRank(x) - marginalCostRank(y));
|
|
1011
|
+
// OBS-1151 (+add.1): the fresh judgment always stands on its own reading — a prior PASS is never
|
|
1012
|
+
// reused. Only a criterion that REVERSES the newest prior ruling on the same subject key over
|
|
1013
|
+
// identical cited blobs needs a second, distinct judge; agreement stands (a sound FAIL included),
|
|
1014
|
+
// and a split, no distinct eligible seat or an unreadable adjudication parks for the operator.
|
|
1015
|
+
const adjudicate = async (fresh) => {
|
|
1016
|
+
const primary = String(fresh.meta?.judge ?? channelKey({ adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model }));
|
|
1017
|
+
const head = await shGit("git rev-parse HEAD", ctx.worktree);
|
|
1018
|
+
const commit = head.code === 0 ? head.stdout.trim() : "";
|
|
1019
|
+
const criteria = fresh.meta.judgment.map((c) => ({
|
|
1020
|
+
id: c.id, key: judgmentSubjectKey(task, c.criterion, ctx.operatorContext), met: c.met, paths: c.paths,
|
|
1021
|
+
}));
|
|
1022
|
+
const record = { commit, judge: primary, criteria };
|
|
1023
|
+
const stamped = { ...fresh, meta: { ...fresh.meta, judgment: record } };
|
|
1024
|
+
if (!commit || !ctx.priorJudgments?.length)
|
|
1025
|
+
return commit ? stamped : { ...fresh, meta: { ...fresh.meta, judgment: undefined } };
|
|
1026
|
+
const disputed = await judgeContradictions(ctx.worktree, commit, criteria, ctx.priorJudgments);
|
|
1027
|
+
if (!disputed.length)
|
|
1028
|
+
return stamped;
|
|
1029
|
+
await ctx.onGate?.({ phase: "note", gate: "acceptance", name: "judge-disagreement", payload: { primary, commit, disputed } });
|
|
1030
|
+
const park = (why, adjudicator) => ({
|
|
1031
|
+
gate: "acceptance", pass: false,
|
|
1032
|
+
details: `judge disagreement on ${disputed.map((d) => d.id).join(", ")}: ${primary} reverses an earlier ruling over identical cited blobs (${[...new Set(disputed.flatMap((d) => d.paths))].join(", ")}) — ${why}; parked for an operator ruling, no worker charge`,
|
|
1033
|
+
meta: { classification: "infra", infra: true, retryable: false, cause: "judge-disagreement", judge: primary,
|
|
1034
|
+
judgeDisagreement: { primary, ...(adjudicator ? { adjudicator } : {}), disputed, outcome: why } },
|
|
1035
|
+
});
|
|
1036
|
+
const primaryAdapter = primary.slice(0, primary.indexOf(":"));
|
|
1037
|
+
// One DISTINCT seat: never the primary channel, a different adapter when the pool has one.
|
|
1038
|
+
const pool = judgePool().filter((c) => channelKey(c) !== primary);
|
|
1039
|
+
const seat = rankJudges(pool.filter((c) => c.adapter !== primaryAdapter))[0] ?? rankJudges(pool)[0];
|
|
1040
|
+
if (!seat)
|
|
1041
|
+
return park("no distinct eligible judge is available");
|
|
1042
|
+
const adjudicator = channelKey(seat);
|
|
1043
|
+
const seatAdapter = getAdapter(seat.adapter, ctx.adapters);
|
|
1044
|
+
const seatVia = via
|
|
1045
|
+
? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", seatAdapter.id) + "-r2", label: via.labelFor("judge") }
|
|
1046
|
+
: undefined;
|
|
1047
|
+
const second = await invokeJudge(seatAdapter, seat.model, seatVia, configuredEffort(ctx.cfg, seat));
|
|
1048
|
+
const rulings = Array.isArray(second.meta?.judgment) ? second.meta.judgment : undefined;
|
|
1049
|
+
// A citation-less ruling cannot be compared, so it confirms nothing (invented evidence is already unparseable,
|
|
1050
|
+
// and an internally inconsistent verdict carries no judgment rows at all).
|
|
1051
|
+
if (second.meta?.unparseable === true || !rulings)
|
|
1052
|
+
return park("the adjudicating judge returned no readable verdict", adjudicator);
|
|
1053
|
+
const agreed = disputed.every((d) => rulings.some((r) => r.id === d.id && r.met === d.met && r.paths.length > 0));
|
|
1054
|
+
if (!agreed)
|
|
1055
|
+
return park("the adjudicating judge split from the fresh ruling", adjudicator);
|
|
1056
|
+
return { ...stamped, meta: { ...stamped.meta, adjudication: { primary, adjudicator, criteria: disputed.map((d) => d.id), agreed: true } } };
|
|
1057
|
+
};
|
|
1058
|
+
// OBS-1182: every judge seat launches at its OWN configured effort, never the worker's.
|
|
1059
|
+
let a = await invokeJudge(judgeAdapter, ctx.cfg.judge.model, jvia, configuredEffort(ctx.cfg, ctx.cfg.judge));
|
|
842
1060
|
// GATE-09: an unparseable judge verdict retries the JUDGE exactly once on a failover channel — never
|
|
843
1061
|
// the worker (run-20260711-185020 P43-03 L70-72 billed a judge flake as a worker attempt). The flaked
|
|
844
1062
|
// first verdict NEVER enters results (no false gate-result journal event, no operator notify, no stale
|
|
@@ -852,7 +1070,7 @@ export async function runGates(task, ctx) {
|
|
|
852
1070
|
// If no other adapter is live, the exclusion degrades to a channel-level reroute within the same
|
|
853
1071
|
// adapter so a single-adapter fleet still retries (matching the daemon's unknown-excludeAdapter
|
|
854
1072
|
// degradation path).
|
|
855
|
-
if (a.meta?.unparseable === true && typeof a.meta.judge === "string") {
|
|
1073
|
+
if (!cancelled && a.meta?.unparseable === true && typeof a.meta.judge === "string") {
|
|
856
1074
|
const flakedKey = a.meta.judge;
|
|
857
1075
|
const flakedAdapter = flakedKey.slice(0, flakedKey.indexOf(":"));
|
|
858
1076
|
const pick = (pool) => pool
|
|
@@ -866,17 +1084,23 @@ export async function runGates(task, ctx) {
|
|
|
866
1084
|
const sameAdapter = pick(judgePool.filter((c) => c.adapter === flakedAdapter && channelKey(c) !== flakedKey));
|
|
867
1085
|
// Prefer a different adapter; if the fleet only has one adapter, retry on a different channel of
|
|
868
1086
|
// that adapter; if the fleet has only one channel, fall back to the original judge config.
|
|
869
|
-
const retry = crossAdapter ?? sameAdapter ??
|
|
1087
|
+
const retry = crossAdapter ?? sameAdapter ?? judgeSeat;
|
|
870
1088
|
const retryAdapter = getAdapter(retry.adapter, ctx.adapters);
|
|
871
|
-
const retryJvia =
|
|
1089
|
+
const retryJvia = via
|
|
872
1090
|
// unconditional -r1 suffix: under keepPanes:forever a same-channel retry cannot collide with the
|
|
873
1091
|
// still-open first pane (herdr agent_name_taken regression, research Pitfall 4)
|
|
874
|
-
? { driver:
|
|
1092
|
+
? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", retryAdapter.id) + "-r1", label: via.labelFor("judge") }
|
|
875
1093
|
: undefined;
|
|
876
1094
|
// the retry IS a second acceptanceGate call: one code path, one parser, zero new parse leniency.
|
|
877
|
-
a = await invokeJudge(retryAdapter, retry.model, retryJvia);
|
|
1095
|
+
a = await invokeJudge(retryAdapter, retry.model, retryJvia, configuredEffort(ctx.cfg, retry));
|
|
878
1096
|
a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
|
|
879
1097
|
}
|
|
1098
|
+
// OBS-1168(b): the re-routed seat could not launch either — no seat produced a verdict, so this is
|
|
1099
|
+
// an infra park over whatever the deterministic gates proved, never a charge against the worker.
|
|
1100
|
+
if (a.meta?.cause === "seat-launch-failed")
|
|
1101
|
+
a = { ...a, meta: { ...a.meta, classification: "infra", infra: true, retryable: false } };
|
|
1102
|
+
else if (!cancelled && Array.isArray(a.meta?.judgment))
|
|
1103
|
+
a = await adjudicate(a);
|
|
880
1104
|
// No dispatch, no key: a deterministic-oracle round writes no `invocations` field rather than an
|
|
881
1105
|
// empty array a reader could mistake for "measured, and it cost nothing".
|
|
882
1106
|
return { result: invocationSpans.length ? { ...a, meta: { ...a.meta, invocations: invocationSpans } } : a, invocations };
|
|
@@ -895,6 +1119,8 @@ export async function runGates(task, ctx) {
|
|
|
895
1119
|
const captured = await captureLlmDispatches(ctx.adapters, run);
|
|
896
1120
|
invocations.push(...captured.invocations);
|
|
897
1121
|
const rv = captured.value;
|
|
1122
|
+
if (cancelled)
|
|
1123
|
+
return rv; // a cancelled seat's non-answer says nothing about the seat
|
|
898
1124
|
if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
|
|
899
1125
|
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
|
|
900
1126
|
if (typeof rv.meta.reviewer === "string") {
|
|
@@ -925,7 +1151,7 @@ export async function runGates(task, ctx) {
|
|
|
925
1151
|
const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
|
|
926
1152
|
const carriedAuthors = ctx.carriedAuthors ?? [];
|
|
927
1153
|
let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
|
|
928
|
-
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg,
|
|
1154
|
+
let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
|
|
929
1155
|
// OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
|
|
930
1156
|
// a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
|
|
931
1157
|
// verdict never enters results; an exhausted pool preserves its cause.
|
|
@@ -935,9 +1161,29 @@ export async function runGates(task, ctx) {
|
|
|
935
1161
|
let retryPrior = [...priorReviewers];
|
|
936
1162
|
const routes = [];
|
|
937
1163
|
let hop = 0;
|
|
938
|
-
|
|
1164
|
+
// OBS-1196: seats already re-asked after a malformed verdict — once per seat, so never unbounded.
|
|
1165
|
+
const reemitted = new Set();
|
|
1166
|
+
while (!cancelled && (rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
|
|
939
1167
|
hop++;
|
|
940
1168
|
const flaked = rv.meta.reviewer;
|
|
1169
|
+
const retryVia = via
|
|
1170
|
+
? { ...via, nameFor: (role, adapter) => via.nameFor(role, adapter) + `-r${hop}` }
|
|
1171
|
+
: undefined;
|
|
1172
|
+
// OBS-1196: a malformed (unparseable) verdict is a delivery defect of THIS seat, not a reason to drop it. Ask the
|
|
1173
|
+
// same seat once more on the same subject; reviewGate mints a fresh nonce, so only a new, whole,
|
|
1174
|
+
// nonce-bound verdict can answer — the malformed bytes are never salvaged into one.
|
|
1175
|
+
if (rv.meta.cause === "malformed-verdict" && rv.meta.closureInvalid !== true && !reemitted.has(flaked)) {
|
|
1176
|
+
reemitted.add(flaked);
|
|
1177
|
+
const others = ctx.channels.map(channelKey).filter((key) => key !== flaked);
|
|
1178
|
+
const again = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, others, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
|
|
1179
|
+
if (again.meta?.noEligibleReviewer !== true) {
|
|
1180
|
+
await ctx.onGate?.({ phase: "note", gate: "review", name: "review-reemission", payload: { reviewer: flaked, cause: "malformed-verdict",
|
|
1181
|
+
delivered: again.meta?.unparseable !== true && again.meta?.noVerdict !== true }, result: again });
|
|
1182
|
+
routes.push(`review re-emission (same seat, fresh nonce): ${flaked} produced a malformed verdict; asked once more`);
|
|
1183
|
+
rv = { ...again, details: `${routes.join("\n")}\n${again.details}`, meta: { ...again.meta, reviewReemission: { reviewer: flaked } } };
|
|
1184
|
+
continue;
|
|
1185
|
+
}
|
|
1186
|
+
}
|
|
941
1187
|
const emptyOutput = rv.meta.cause === "empty-output";
|
|
942
1188
|
if (emptyOutput) {
|
|
943
1189
|
await ctx.onGate?.({
|
|
@@ -946,9 +1192,6 @@ export async function runGates(task, ctx) {
|
|
|
946
1192
|
result: { ...rv, meta: { ...rv.meta, skipped: true } },
|
|
947
1193
|
});
|
|
948
1194
|
}
|
|
949
|
-
const retryVia = ctx.via
|
|
950
|
-
? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + `-r${hop}` }
|
|
951
|
-
: undefined;
|
|
952
1195
|
const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
|
|
953
1196
|
const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
|
|
954
1197
|
// RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
|
|
@@ -982,11 +1225,14 @@ export async function runGates(task, ctx) {
|
|
|
982
1225
|
break;
|
|
983
1226
|
}
|
|
984
1227
|
}
|
|
985
|
-
if (rv.meta?.noVerdict === true
|
|
1228
|
+
if (rv.meta?.noVerdict === true
|
|
1229
|
+
|| (rv.meta?.unparseable === true && rv.meta.closureInvalid !== true && typeof rv.meta.reviewer === "string")) {
|
|
986
1230
|
// Terminal: every eligible seat returned no verdict. The carried materials stay open — an infra
|
|
987
1231
|
// row is not a passing review — and the sibling judge result is untouched beside it.
|
|
1232
|
+
// OBS-1196: an exhausted pool of UNDELIVERED (unparseable) verdicts is the same non-verdict: an infra
|
|
1233
|
+
// park, never a worker retry. A delivered verdict that breaks the closure protocol keeps its path.
|
|
988
1234
|
const carried = (ctx.carriedFindings ?? []).filter((f) => f.class === "review:material").map((f) => f.fingerprint);
|
|
989
|
-
rv = { ...rv, meta: { ...rv.meta, classification: "infra", infra: true, carriedFindings: carried } };
|
|
1235
|
+
rv = { ...rv, meta: { ...rv.meta, noVerdict: true, classification: "infra", infra: true, carriedFindings: carried } };
|
|
990
1236
|
}
|
|
991
1237
|
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
|
992
1238
|
};
|
|
@@ -1045,11 +1291,27 @@ export async function runGates(task, ctx) {
|
|
|
1045
1291
|
else
|
|
1046
1292
|
selected = [...new Set([...selected, ...required])].sort();
|
|
1047
1293
|
}
|
|
1048
|
-
|
|
1294
|
+
// OBS-635: a screen buys nothing when the full suite that must follow it already has a qualified
|
|
1295
|
+
// green on this exact identity (read BEFORE any screen), or when the harness-measured screen costs
|
|
1296
|
+
// at least 75 % of it — then the full suite runs in the battery instead. Unknown timing keeps the
|
|
1297
|
+
// screen. A selected-only green never answers here: its identity names its selection.
|
|
1298
|
+
let promotion;
|
|
1299
|
+
if (selected) {
|
|
1300
|
+
const hit = verdictStore.get(await fullTestIdentity());
|
|
1301
|
+
const costRatio = screenCostRatio(ctx.baseline, selected);
|
|
1302
|
+
if (hit?.pass === true && !isInfraResult(hit) && await certifiesFullManifest(hit))
|
|
1303
|
+
promotion = { reason: "full-green-cache" };
|
|
1304
|
+
else if (costRatio !== undefined && costRatio >= SCREEN_PROMOTION_RATIO)
|
|
1305
|
+
promotion = { reason: "screen-cost-promoted", costRatio };
|
|
1306
|
+
if (promotion)
|
|
1307
|
+
selected = undefined;
|
|
1308
|
+
}
|
|
1309
|
+
if (ctx.selectionReason || promotion)
|
|
1049
1310
|
selectionDecision = {
|
|
1050
|
-
scope: selected ? "selected" : "full", reason: selected ? selectionReason
|
|
1051
|
-
: selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason,
|
|
1311
|
+
scope: selected ? "selected" : "full", reason: promotion?.reason ?? (selected ? selectionReason
|
|
1312
|
+
: selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason),
|
|
1052
1313
|
requiredFiles: [...(ctx.requiredRepairTests ?? [])],
|
|
1314
|
+
...(promotion?.costRatio !== undefined ? { costRatio: promotion.costRatio } : {}),
|
|
1053
1315
|
};
|
|
1054
1316
|
await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
|
|
1055
1317
|
if (failed())
|
|
@@ -1073,27 +1335,73 @@ export async function runGates(task, ctx) {
|
|
|
1073
1335
|
// enough: an acceptance-first await withholds a completed review behind a slow/hung judge and a
|
|
1074
1336
|
// process death can lose that already-earned verdict. The returned result is still sorted into
|
|
1075
1337
|
// GATE_NAMES order by done(); the event stream truthfully records each independent completion.
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
if (
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
}
|
|
1086
|
-
|
|
1087
|
-
|
|
1338
|
+
// OBS-1168(c): a sibling that throws, or that ends seatless (no seat could launch, so the round can
|
|
1339
|
+
// only park infra), cancels the other — its seats are closed — and the round still AWAITS it, with
|
|
1340
|
+
// or without an execution policy, so no verdict of a settled round publishes after its engagement.
|
|
1341
|
+
const seatless = (r) => r.meta?.cause === "seat-launch-failed" && r.meta?.infra === true;
|
|
1342
|
+
let failure;
|
|
1343
|
+
const fail = (reason) => {
|
|
1344
|
+
if (!cancelled)
|
|
1345
|
+
failure ??= { reason };
|
|
1346
|
+
cancelSemantic();
|
|
1347
|
+
};
|
|
1348
|
+
const judged = judging?.then(async (outcome) => {
|
|
1349
|
+
if (cancelled)
|
|
1350
|
+
return;
|
|
1351
|
+
await withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result));
|
|
1352
|
+
if (seatless(outcome.result))
|
|
1353
|
+
cancelSemantic();
|
|
1354
|
+
}).catch(fail);
|
|
1355
|
+
const reviewed = reviewing?.then(async (outcome) => {
|
|
1356
|
+
if (cancelled)
|
|
1357
|
+
return;
|
|
1358
|
+
await record(outcome);
|
|
1359
|
+
if (seatless(outcome))
|
|
1360
|
+
cancelSemantic();
|
|
1361
|
+
}).catch(fail);
|
|
1362
|
+
// ponytail: a headless seat has no slot to close; it is awaited to its own timeout, never abandoned.
|
|
1363
|
+
await Promise.all([judged, reviewed]);
|
|
1364
|
+
await Promise.all(closing);
|
|
1365
|
+
if (failure)
|
|
1366
|
+
throw failure.reason;
|
|
1367
|
+
executionSignal()?.throwIfAborted();
|
|
1088
1368
|
if (failed())
|
|
1089
1369
|
return done();
|
|
1090
1370
|
}
|
|
1371
|
+
// OBS-635: the semantic gates ran oracles and vendor CLIs in this worktree after an in-battery full
|
|
1372
|
+
// suite spoke, and its green stands only for the identity it measured. Oracle dirt withdraws it; a
|
|
1373
|
+
// changed or unmeasurable identity — tree, command, baseline, environment, dependency resolution,
|
|
1374
|
+
// capacity, protocol, lifecycle, full manifest — buys a fresh merge-candidate suite below.
|
|
1375
|
+
let rerunFull = false;
|
|
1376
|
+
listing = undefined;
|
|
1377
|
+
if (fullInBattery && ctx.commands.test !== undefined && (enabled("acceptance") || enabled("review"))) {
|
|
1378
|
+
const dirt = await dirtyWorktree();
|
|
1379
|
+
if (dirt) {
|
|
1380
|
+
const refusal = withTelemetry(await dirtyRoundRefusal("test", dirt));
|
|
1381
|
+
results[results.findIndex((r) => r.gate === "test")] = refusal;
|
|
1382
|
+
await ctx.onGate?.({ phase: "end", gate: "test", result: refusal });
|
|
1383
|
+
return done();
|
|
1384
|
+
}
|
|
1385
|
+
const now = await fullTestIdentity();
|
|
1386
|
+
// D-598: an unmeasurable lifecycle on either side is not comparable to anything (VerdictStore R41
|
|
1387
|
+
// refuses it); two `unknown`s hashing equal is not an unchanged identity, so the suite reruns.
|
|
1388
|
+
const measurable = (id) => id?.envParts?.verification?.lifecycle !== "unknown";
|
|
1389
|
+
rerunFull = !now || !batteryFullIdentity || !measurable(now) || !measurable(batteryFullIdentity)
|
|
1390
|
+
|| verificationIdentityKey(now) !== verificationIdentityKey(batteryFullIdentity)
|
|
1391
|
+
|| !(await certifiesFullManifest(results.find((r) => r.gate === "test")));
|
|
1392
|
+
if (rerunFull) {
|
|
1393
|
+
// The in-battery row keeps its own interval; the replacement measures from zero.
|
|
1394
|
+
spans.delete("test");
|
|
1395
|
+
loadSamples.delete("test");
|
|
1396
|
+
}
|
|
1397
|
+
}
|
|
1091
1398
|
// The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
|
|
1092
1399
|
// the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
|
|
1093
|
-
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict
|
|
1094
|
-
//
|
|
1095
|
-
//
|
|
1096
|
-
|
|
1400
|
+
// on a subset (spec: "nothing merges without a complete green suite"). Its verdict replaces the
|
|
1401
|
+
// screen's entry in the returned record (one `test` entry), and `fullSuite` says which suite spoke
|
|
1402
|
+
// while `selectedTests` keeps what the screen ran. In the stream, a held screen is superseded (one
|
|
1403
|
+
// `test` end event); a published screen keeps its own earlier event and this is the second.
|
|
1404
|
+
if (selected || rerunFull) {
|
|
1097
1405
|
await emitStart("test");
|
|
1098
1406
|
// This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
|
|
1099
1407
|
// may have run one before it, and every gate between the battery and here reads commits only, so
|
|
@@ -1104,20 +1412,12 @@ export async function runGates(task, ctx) {
|
|
|
1104
1412
|
let cached = false;
|
|
1105
1413
|
let identity;
|
|
1106
1414
|
if (ctx.commands.test !== undefined) {
|
|
1107
|
-
identity = await
|
|
1108
|
-
worktree: ctx.worktree,
|
|
1109
|
-
gate: "test",
|
|
1110
|
-
scope: ctx.verificationScope,
|
|
1111
|
-
command: ctx.commands.test,
|
|
1112
|
-
baseline: ctx.baseline,
|
|
1113
|
-
selectedSet: undefined,
|
|
1114
|
-
capacity: resolvedCapacity(),
|
|
1115
|
-
});
|
|
1415
|
+
identity = await fullTestIdentity();
|
|
1116
1416
|
const hit = verdictStore.get(identity);
|
|
1117
1417
|
if (hit)
|
|
1118
1418
|
classifySignalOnlyTest(hit);
|
|
1119
1419
|
if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
|
|
1120
|
-
&& !(await discardCachedRed("test", hit))) {
|
|
1420
|
+
&& !(await discardCachedRed("test", hit)) && await certifiesFullManifest(hit)) {
|
|
1121
1421
|
full = formatReusedRow(hit, identity);
|
|
1122
1422
|
cached = true;
|
|
1123
1423
|
await noteReuse("test", full, identity);
|