tickmarkr 2.6.2 → 2.6.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/README.md +2 -0
  2. package/dist/cli/commands/approve.d.ts +6 -3
  3. package/dist/cli/commands/approve.js +16 -4
  4. package/dist/cli/commands/doctor.d.ts +6 -2
  5. package/dist/cli/commands/fleet.js +46 -8
  6. package/dist/cli/commands/plan.js +13 -8
  7. package/dist/cli/commands/report.d.ts +2 -1
  8. package/dist/cli/commands/report.js +56 -6
  9. package/dist/cli/commands/status.js +19 -1
  10. package/dist/compile/native.js +7 -0
  11. package/dist/config/config.d.ts +17 -4
  12. package/dist/config/config.js +43 -6
  13. package/dist/config/fleet-overlay.d.ts +13 -3
  14. package/dist/config/fleet-overlay.js +12 -8
  15. package/dist/drivers/herdr.d.ts +12 -0
  16. package/dist/drivers/herdr.js +51 -0
  17. package/dist/drivers/orca.d.ts +9 -1
  18. package/dist/drivers/orca.js +29 -7
  19. package/dist/drivers/types.d.ts +2 -0
  20. package/dist/drivers/types.js +2 -2
  21. package/dist/gates/acceptance.d.ts +7 -0
  22. package/dist/gates/acceptance.js +27 -5
  23. package/dist/gates/baseline.d.ts +20 -1
  24. package/dist/gates/baseline.js +100 -20
  25. package/dist/gates/cache.d.ts +8 -0
  26. package/dist/gates/cache.js +12 -2
  27. package/dist/gates/llm.d.ts +6 -0
  28. package/dist/gates/llm.js +27 -8
  29. package/dist/gates/review.d.ts +6 -1
  30. package/dist/gates/review.js +122 -32
  31. package/dist/gates/run-gates.d.ts +54 -3
  32. package/dist/gates/run-gates.js +331 -45
  33. package/dist/gates/test-manifest.d.ts +42 -0
  34. package/dist/gates/test-manifest.js +69 -10
  35. package/dist/route/router.d.ts +12 -1
  36. package/dist/route/router.js +26 -9
  37. package/dist/run/consult.d.ts +3 -1
  38. package/dist/run/consult.js +4 -2
  39. package/dist/run/daemon.d.ts +2 -1
  40. package/dist/run/daemon.js +342 -77
  41. package/dist/run/interactive-seed.d.ts +4 -0
  42. package/dist/run/interactive-seed.js +35 -9
  43. package/dist/run/journal.d.ts +26 -0
  44. package/dist/run/journal.js +143 -15
  45. package/dist/run/lease.d.ts +13 -0
  46. package/dist/run/lease.js +45 -0
  47. package/dist/run/protocol.d.ts +15 -0
  48. package/dist/run/protocol.js +11 -1
  49. package/dist/run/receipt-resolver.d.ts +22 -0
  50. package/dist/run/receipt-resolver.js +40 -1
  51. package/dist/run/repair-selection.d.ts +11 -1
  52. package/dist/run/repair-selection.js +17 -9
  53. package/dist/run/wall-budget.d.ts +48 -0
  54. package/dist/run/wall-budget.js +280 -0
  55. package/dist/tui/cockpit/run-cockpit.js +2 -2
  56. package/dist/tui/cockpit/run-view.d.ts +2 -1
  57. package/dist/tui/cockpit/run-view.js +13 -9
  58. package/dist/tui/cockpit/setup-cockpit.d.ts +2 -0
  59. package/dist/tui/cockpit/setup-cockpit.js +4 -0
  60. package/package.json +2 -1
  61. package/schema/config.schema.json +8 -1
  62. package/skills/tickmarkr-loop/SKILL.md +7 -1
  63. package/skills/tickmarkr-overseer/scripts/classify-vitest-log.sh +4 -1
@@ -1,4 +1,4 @@
1
- import { randomUUID } from "node:crypto";
1
+ import { createHash, randomUUID } from "node:crypto";
2
2
  import { existsSync, mkdtempSync, readFileSync, rmSync, statSync } from "node:fs";
3
3
  import { loadavg, tmpdir } from "node:os";
4
4
  import { join, posix } from "node:path";
@@ -10,14 +10,14 @@ import { acceptanceGate } from "./acceptance.js";
10
10
  import { compareToBaseline, effectiveCeilingMs, waitForCalmWindow, calmWindowReady } from "./baseline.js";
11
11
  import { evidenceGate } from "./evidence.js";
12
12
  import { captureLlmOutput } from "./llm.js";
13
- import { disallowedBy } from "../route/preference.js";
13
+ import { disallowedBy, observedSeat } from "../route/preference.js";
14
14
  import { marginalCostRank } from "../route/router.js";
15
15
  import { carriedAuthorVendors, gateReviewerFloor, pickReviewer, reviewGate } from "./review.js";
16
16
  import { scopeGate } from "./scope.js";
17
- import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
17
+ import { discoverTestManifest, evaluateManifestedTest, isVitestTestCommand, VITEST_CACHE_ENV, worktreeVitestCache } from "./test-manifest.js";
18
18
  import { executionSignal } from "../run/execution-budget.js";
19
19
  import { failureDisposition } from "../run/recovery.js";
20
- import { dependencyLinkRefusal, preserveWorktree, producerFields, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
20
+ import { dependencyLinkRefusal, FORK_CAP_ENV, preserveWorktree, producerFields, ROUTING_ENV_SEAMS, shGit, resolvedCapacity, SUITE_PARENT_ENV, verificationProtocol } from "../run/git.js";
21
21
  import { withJudgeInvocationEvidence } from "../run/journal.js";
22
22
  import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
23
23
  const productionLoadProvider = () => loadavg()[0] ?? 0;
@@ -215,6 +215,44 @@ export function testCommandForFiles(testCmd, files) {
215
215
  const fwd = wrapped && !/\s--\s/.test(testCmd) ? " --" : "";
216
216
  return `${testCmd}${fwd} ${files.map(shq).join(" ")}`;
217
217
  }
218
+ /** OBS-635: a screen costing at least this share of the full suite runs the full suite instead. */
219
+ export const SCREEN_PROMOTION_RATIO = 0.75;
220
+ /**
221
+ * The screen's share of the full suite's cost, from the per-file durations the harness measured at
222
+ * baseline capture — never a worker's timing. Undefined (unknown) unless every selected file has a
223
+ * measured duration and the measured total is positive; unknown keeps the conservative screen path.
224
+ */
225
+ export function screenCostRatio(baseline, selected) {
226
+ const entry = baseline.commands.test;
227
+ const files = entry?.infra ? undefined : entry?.fileDurations;
228
+ if (!files?.length)
229
+ return undefined;
230
+ const cost = new Map(files.map((f) => [f.file, f.durationMs]));
231
+ const total = files.reduce((sum, f) => sum + f.durationMs, 0);
232
+ if (!(total > 0) || selected.some((file) => !cost.has(file)))
233
+ return undefined;
234
+ return selected.reduce((sum, file) => sum + cost.get(file), 0) / total;
235
+ }
236
+ /** OBS-635: the full manifest the runner lists NOW, under the environment evaluateManifestedTest's own
237
+ * discovery receives (test-manifest.ts manifestEnvironment), so it compares with the one a verdict
238
+ * certified. Undefined when the runner cannot list — an unlisted manifest certifies nothing. */
239
+ async function listFullManifest(cmd, worktree) {
240
+ const env = { ...process.env, PATH: `${join(worktree, "node_modules/.bin")}:${process.env.PATH ?? ""}`,
241
+ [VITEST_CACHE_ENV]: worktreeVitestCache(worktree),
242
+ [FORK_CAP_ENV]: String(resolvedCapacity().forkCap), [SUITE_PARENT_ENV]: String(process.pid) };
243
+ for (const key of [...ROUTING_ENV_SEAMS, "VITEST", "TEST", "VITEST_WORKER_ID", "VITEST_POOL_ID"])
244
+ delete env[key];
245
+ const dir = mkdtempSync(join(tmpdir(), "tickmarkr-full-manifest-"));
246
+ try {
247
+ return (await discoverTestManifest(cmd, worktree, { dir, nonce: randomUUID(), env })).files;
248
+ }
249
+ catch {
250
+ return undefined;
251
+ }
252
+ finally {
253
+ rmSync(dir, { recursive: true, force: true });
254
+ }
255
+ }
218
256
  /** The manifest-report path for a detected vitest test command — never the stdout-count/file-count path. */
219
257
  async function runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry = {}, retried = false) {
220
258
  const entry = baseline.commands.test;
@@ -249,6 +287,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
249
287
  meta: { ...outcome.meta, reportPath, ...(selected ? { selectedTests: [...selected] } : {}) },
250
288
  };
251
289
  }
290
+ /** A retry base no runner invocation parses. evaluateManifestedTest builds its stranded single-fork
291
+ * retry from the base it is handed, and one it cannot parse throws before any spawn — so this base
292
+ * disables that inner recovery: a worker-RPC-stranded re-observation comes back infra (the caller
293
+ * parks it as ambiguous) instead of launching a second execution. */
294
+ export const REOBSERVATION_RETRY_BASE = "tickmarkr-reobservation-refuses-stranded-retry";
295
+ /** OBS-1106 residual: ONE isolated re-observation of a timeout-shaped red's attributed failing files on
296
+ * the same checkout, narrowed exactly as a screen is. Never cached and never a verdict: the caller
297
+ * keeps the original red and reads this only to decide whether that red is chargeable. Exactly one
298
+ * execution — the bounded infra/host-starved retries and the stranded single-fork recovery are all
299
+ * refused, so a diagnostic never buys more. */
300
+ export async function reobserveTestFiles(worktree, testCmd, baseline, files, artifactDir) {
301
+ const cmd = testCommandForFiles(testCmd, files);
302
+ if (!isVitestTestCommand(testCmd, worktree))
303
+ return (await compareToBaseline(worktree, { test: cmd }, baseline, ["test"], { selected: files, authorizeRetry: () => false }))[0];
304
+ const r = await runVitestManifestGate(worktree, cmd, baseline, files, artifactDir, { retryBaseCommand: REOBSERVATION_RETRY_BASE });
305
+ // fail closed whatever a recovery did: a re-observation never reads a recovered verdict
306
+ return r.meta?.recovery === undefined ? r
307
+ : { ...r, pass: false, meta: { ...r.meta, classification: "infra", infra: true, retryable: false, recoveryRefused: true } };
308
+ }
252
309
  const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
253
310
  const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
254
311
  /** D1: apply the daemon's signal-only rider before either battery cache read or write. Its onGate
@@ -262,12 +319,56 @@ function classifySignalOnlyTest(g) {
262
319
  return;
263
320
  g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
264
321
  }
322
+ /** OBS-1151: a criterion's comparable subject — its canonical text, the task's declared bounds and the
323
+ * operator context. The cited files' blobs are compared separately, over the union of both citations. */
324
+ export function judgmentSubjectKey(task, criterion, operatorContext) {
325
+ return createHash("sha256").update(JSON.stringify([criterion, [...task.files].sort(), [...(task.outOfScope ?? [])].sort(), operatorContext ?? ""])).digest("hex");
326
+ }
327
+ async function blobAt(worktree, commit, path) {
328
+ const r = await shGit(`git rev-parse --verify --quiet ${shq(`${commit}:${path}`)}`, worktree);
329
+ return r.code === 0 && r.stdout.trim() ? r.stdout.trim() : undefined;
330
+ }
331
+ /** OBS-1151: the criteria whose fresh ruling reverses the newest prior ruling on the same subject key whose
332
+ * cited paths hold the identical blob at both commits (older comparable priors are still found behind a
333
+ * newer prior on different blobs). A citation-less side or an
334
+ * unreadable blob is unknown, never identical — that criterion's fresh ruling is simply fresh. */
335
+ export async function judgeContradictions(worktree, head, fresh, priors) {
336
+ const found = [];
337
+ for (const c of fresh) {
338
+ if (!c.paths.length)
339
+ continue;
340
+ // C-3 (D-669): the comparable prior is the NEWEST one on identical cited blobs, not the newest one
341
+ // carrying the key — PASS(A) → FAIL(B) → fresh FAIL(A) must still be adjudicated against PASS(A).
342
+ for (const prior of priors) {
343
+ const was = prior.criteria.find((q) => q.key === c.key);
344
+ if (!was || !was.paths.length)
345
+ continue;
346
+ const paths = [...new Set([...was.paths, ...c.paths])].sort();
347
+ let identical = true;
348
+ for (const path of paths) {
349
+ const [before, now] = await Promise.all([blobAt(worktree, prior.commit, path), blobAt(worktree, head, path)]);
350
+ if (!before || !now || before !== now) {
351
+ identical = false;
352
+ break;
353
+ }
354
+ }
355
+ if (!identical)
356
+ continue;
357
+ if (was.met !== c.met)
358
+ found.push({ id: c.id, met: c.met, priorMet: was.met, priorCommit: prior.commit, paths });
359
+ break;
360
+ }
361
+ }
362
+ return found;
363
+ }
265
364
  export async function runGates(task, ctx) {
266
365
  const results = [];
267
366
  const evidence = {
268
367
  artifactDir: ctx.artifactDir,
269
368
  runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
270
369
  taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
370
+ // OBS-1140: task and standalone gates honour the configured quota, not the built-in default.
371
+ quotaBytes: ctx.cfg.gates?.evidenceQuotaBytes,
271
372
  ...ctx.evidence,
272
373
  };
273
374
  // Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
@@ -305,6 +406,9 @@ export async function runGates(task, ctx) {
305
406
  await receiptNotes;
306
407
  };
307
408
  let selectionDecision;
409
+ // OBS-635: the identity a full suite measured inside the battery, revalidated after semantics.
410
+ let fullInBattery = false;
411
+ let batteryFullIdentity;
308
412
  let commits = [];
309
413
  // Check before cache identity, npm policy probes, or any gate command.
310
414
  const dependencyRefusal = dependencyLinkRefusal(ctx.worktree);
@@ -338,6 +442,29 @@ export async function runGates(task, ctx) {
338
442
  payload: { gate, reason: "cached-red-discarded", ...(reason === "recheck" ? {} : { bypass: reason }) }, result: hit });
339
443
  return true;
340
444
  };
445
+ // OBS-1168(c): every judge/review seat this round opens, so a failed sibling can cancel the other.
446
+ // A cancelled round dispatches, re-routes and publishes nothing more; its closed seats' own errors
447
+ // are consequences of the cancel, never a second failure.
448
+ // A seat whose creation was still pending at the cancel is refused before dispatch: onSlot runs inside
449
+ // llm.ts's launch guard, so the throw closes (and awaits) that half-launched pane and never runs it.
450
+ const semanticSlots = new Set();
451
+ const closing = [];
452
+ const via = ctx.via && {
453
+ ...ctx.via, onSlot: (slot) => {
454
+ semanticSlots.add(slot);
455
+ ctx.via.onSlot?.(slot);
456
+ if (cancelled)
457
+ throw new Error("semantic round cancelled before dispatch");
458
+ },
459
+ };
460
+ let cancelled = false;
461
+ const cancelSemantic = () => {
462
+ if (cancelled)
463
+ return;
464
+ cancelled = true;
465
+ for (const slot of semanticSlots)
466
+ closing.push(via.driver.close(slot).catch(() => { }));
467
+ };
341
468
  const shapeGates = ctx.cfg.gates.byShape?.[task.shape];
342
469
  const enabled = (g) => task.gates.includes(g) && (g !== "acceptance" && g !== "review" || shapeGates?.[g] !== false);
343
470
  const failed = () => results.some((r) => !r.pass);
@@ -639,6 +766,28 @@ export async function runGates(task, ctx) {
639
766
  },
640
767
  };
641
768
  };
769
+ const fullTestIdentity = () => computeVerificationIdentity({
770
+ worktree: ctx.worktree,
771
+ gate: "test",
772
+ scope: ctx.verificationScope,
773
+ command: ctx.commands.test,
774
+ baseline: ctx.baseline,
775
+ selectedSet: undefined,
776
+ capacity: resolvedCapacity(),
777
+ });
778
+ // OBS-635: a full green answers only for the manifest it certified. The tree identity cannot see an
779
+ // ignored generated test the runner would collect, so every full-green reuse rediscovers the
780
+ // runner's listing and requires the verdict's to equal it; a runner without a listing is bound by
781
+ // its identity alone. Reads before the semantic gates share one listing; `listing` resets after them.
782
+ let listing;
783
+ const certifiesFullManifest = async (verdict) => {
784
+ if (!verdict.pass || !isVitestTestCommand(ctx.commands.test, ctx.worktree))
785
+ return true;
786
+ const certified = verdict.meta?.manifest;
787
+ const current = await (listing ??= listFullManifest(ctx.commands.test, ctx.worktree));
788
+ return Array.isArray(certified) && current !== undefined
789
+ && certified.length === current.length && [...certified].sort().every((file, i) => file === current[i]);
790
+ };
642
791
  // shell tools vs the shared baseline
643
792
  const retryOptions = (identity) => ctx.authorizeInfraRetry
644
793
  ? { authorizeRetry: (cause) => ctx.authorizeInfraRetry(identity ? verificationIdentityKey(identity) : "", cause === "infra" ? "infrastructure" : "host-starved") }
@@ -668,7 +817,7 @@ export async function runGates(task, ctx) {
668
817
  if (hit)
669
818
  classifySignalOnlyTest(hit); // Older entries predate classification at the write seam.
670
819
  if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
671
- && !(await discardCachedRed(g, hit))) {
820
+ && !(await discardCachedRed(g, hit)) && (g !== "test" || selected !== undefined || await certifiesFullManifest(hit))) {
672
821
  r = formatReusedRow(hit, identity);
673
822
  cached = true;
674
823
  if (g === "build")
@@ -695,6 +844,10 @@ export async function runGates(task, ctx) {
695
844
  // the screen's interval IS the test gate's first interval, so the split needs no second clock
696
845
  if (g === "test" && selected)
697
846
  selectedDurationMs = spans.get("test")?.durationMs ?? 0;
847
+ if (g === "test" && !selected) {
848
+ fullInBattery = true;
849
+ batteryFullIdentity = identity;
850
+ }
698
851
  // The pre-battery check proves the tree clean ONCE; a command that exits 0 having rewritten a
699
852
  // tracked file makes it dirty again, and every gate after it — including the next shell gate,
700
853
  // which would then run against bytes HEAD does not hold — inherits that. So re-check after each
@@ -802,7 +955,9 @@ export async function runGates(task, ctx) {
802
955
  // v1.87 T2: the judge is a configured seat like any other — check it against the operator's
803
956
  // policy BEFORE spending a dispatch on it. disallowedBy carries the whole deny grammar (adapter,
804
957
  // model, or adapter:model), so a model-scoped deny cannot slip past an adapter-id-only read.
805
- const judgeDenied = disallowedBy({ adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model }, ctx.cfg.routing, "judge");
958
+ // OBS-1186: under the exact cached identity of that channel, as compile, doctor and route read it.
959
+ const judgeSeat = observedSeat(ctx.health, ctx.cfg.judge.adapter, ctx.cfg.judge.model);
960
+ const judgeDenied = disallowedBy(judgeSeat, ctx.cfg.routing, "judge");
806
961
  if (judgeDenied) {
807
962
  return {
808
963
  result: {
@@ -815,8 +970,8 @@ export async function runGates(task, ctx) {
815
970
  };
816
971
  }
817
972
  const judgeAdapter = getAdapter(ctx.cfg.judge.adapter, ctx.adapters);
818
- const jvia = ctx.via
819
- ? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", judgeAdapter.id), label: ctx.via.labelFor("judge") }
973
+ const jvia = via
974
+ ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", judgeAdapter.id), label: via.labelFor("judge") }
820
975
  : undefined;
821
976
  // v1.19 (T2): testCmd threads the detected test runner to the gate so named-test oracles run
822
977
  // deterministically (filtered via -t) before any LLM judge dispatch.
@@ -850,6 +1005,56 @@ export async function runGates(task, ctx) {
850
1005
  }
851
1006
  return captured.value;
852
1007
  };
1008
+ const judgePool = () => (ctx.judgeChannels ?? []).filter((c) => disallowedBy(c, ctx.cfg.routing, "judge") === null);
1009
+ const rankJudges = (pool) => [...pool]
1010
+ .sort((x, y) => TIER_RANK[y.tier] - TIER_RANK[x.tier] || marginalCostRank(x) - marginalCostRank(y));
1011
+ // OBS-1151 (+add.1): the fresh judgment always stands on its own reading — a prior PASS is never
1012
+ // reused. Only a criterion that REVERSES the newest prior ruling on the same subject key over
1013
+ // identical cited blobs needs a second, distinct judge; agreement stands (a sound FAIL included),
1014
+ // and a split, no distinct eligible seat or an unreadable adjudication parks for the operator.
1015
+ const adjudicate = async (fresh) => {
1016
+ const primary = String(fresh.meta?.judge ?? channelKey({ adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model }));
1017
+ const head = await shGit("git rev-parse HEAD", ctx.worktree);
1018
+ const commit = head.code === 0 ? head.stdout.trim() : "";
1019
+ const criteria = fresh.meta.judgment.map((c) => ({
1020
+ id: c.id, key: judgmentSubjectKey(task, c.criterion, ctx.operatorContext), met: c.met, paths: c.paths,
1021
+ }));
1022
+ const record = { commit, judge: primary, criteria };
1023
+ const stamped = { ...fresh, meta: { ...fresh.meta, judgment: record } };
1024
+ if (!commit || !ctx.priorJudgments?.length)
1025
+ return commit ? stamped : { ...fresh, meta: { ...fresh.meta, judgment: undefined } };
1026
+ const disputed = await judgeContradictions(ctx.worktree, commit, criteria, ctx.priorJudgments);
1027
+ if (!disputed.length)
1028
+ return stamped;
1029
+ await ctx.onGate?.({ phase: "note", gate: "acceptance", name: "judge-disagreement", payload: { primary, commit, disputed } });
1030
+ const park = (why, adjudicator) => ({
1031
+ gate: "acceptance", pass: false,
1032
+ details: `judge disagreement on ${disputed.map((d) => d.id).join(", ")}: ${primary} reverses an earlier ruling over identical cited blobs (${[...new Set(disputed.flatMap((d) => d.paths))].join(", ")}) — ${why}; parked for an operator ruling, no worker charge`,
1033
+ meta: { classification: "infra", infra: true, retryable: false, cause: "judge-disagreement", judge: primary,
1034
+ judgeDisagreement: { primary, ...(adjudicator ? { adjudicator } : {}), disputed, outcome: why } },
1035
+ });
1036
+ const primaryAdapter = primary.slice(0, primary.indexOf(":"));
1037
+ // One DISTINCT seat: never the primary channel, a different adapter when the pool has one.
1038
+ const pool = judgePool().filter((c) => channelKey(c) !== primary);
1039
+ const seat = rankJudges(pool.filter((c) => c.adapter !== primaryAdapter))[0] ?? rankJudges(pool)[0];
1040
+ if (!seat)
1041
+ return park("no distinct eligible judge is available");
1042
+ const adjudicator = channelKey(seat);
1043
+ const seatAdapter = getAdapter(seat.adapter, ctx.adapters);
1044
+ const seatVia = via
1045
+ ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", seatAdapter.id) + "-r2", label: via.labelFor("judge") }
1046
+ : undefined;
1047
+ const second = await invokeJudge(seatAdapter, seat.model, seatVia, configuredEffort(ctx.cfg, seat));
1048
+ const rulings = Array.isArray(second.meta?.judgment) ? second.meta.judgment : undefined;
1049
+ // A citation-less ruling cannot be compared, so it confirms nothing (invented evidence is already unparseable,
1050
+ // and an internally inconsistent verdict carries no judgment rows at all).
1051
+ if (second.meta?.unparseable === true || !rulings)
1052
+ return park("the adjudicating judge returned no readable verdict", adjudicator);
1053
+ const agreed = disputed.every((d) => rulings.some((r) => r.id === d.id && r.met === d.met && r.paths.length > 0));
1054
+ if (!agreed)
1055
+ return park("the adjudicating judge split from the fresh ruling", adjudicator);
1056
+ return { ...stamped, meta: { ...stamped.meta, adjudication: { primary, adjudicator, criteria: disputed.map((d) => d.id), agreed: true } } };
1057
+ };
853
1058
  // OBS-1182: every judge seat launches at its OWN configured effort, never the worker's.
854
1059
  let a = await invokeJudge(judgeAdapter, ctx.cfg.judge.model, jvia, configuredEffort(ctx.cfg, ctx.cfg.judge));
855
1060
  // GATE-09: an unparseable judge verdict retries the JUDGE exactly once on a failover channel — never
@@ -865,7 +1070,7 @@ export async function runGates(task, ctx) {
865
1070
  // If no other adapter is live, the exclusion degrades to a channel-level reroute within the same
866
1071
  // adapter so a single-adapter fleet still retries (matching the daemon's unknown-excludeAdapter
867
1072
  // degradation path).
868
- if (a.meta?.unparseable === true && typeof a.meta.judge === "string") {
1073
+ if (!cancelled && a.meta?.unparseable === true && typeof a.meta.judge === "string") {
869
1074
  const flakedKey = a.meta.judge;
870
1075
  const flakedAdapter = flakedKey.slice(0, flakedKey.indexOf(":"));
871
1076
  const pick = (pool) => pool
@@ -879,17 +1084,23 @@ export async function runGates(task, ctx) {
879
1084
  const sameAdapter = pick(judgePool.filter((c) => c.adapter === flakedAdapter && channelKey(c) !== flakedKey));
880
1085
  // Prefer a different adapter; if the fleet only has one adapter, retry on a different channel of
881
1086
  // that adapter; if the fleet has only one channel, fall back to the original judge config.
882
- const retry = crossAdapter ?? sameAdapter ?? { adapter: ctx.cfg.judge.adapter, model: ctx.cfg.judge.model };
1087
+ const retry = crossAdapter ?? sameAdapter ?? judgeSeat;
883
1088
  const retryAdapter = getAdapter(retry.adapter, ctx.adapters);
884
- const retryJvia = ctx.via
1089
+ const retryJvia = via
885
1090
  // unconditional -r1 suffix: under keepPanes:forever a same-channel retry cannot collide with the
886
1091
  // still-open first pane (herdr agent_name_taken regression, research Pitfall 4)
887
- ? { driver: ctx.via.driver, keep: ctx.via.keep, onSlot: ctx.via.onSlot, name: ctx.via.nameFor("judge", retryAdapter.id) + "-r1", label: ctx.via.labelFor("judge") }
1092
+ ? { driver: via.driver, keep: via.keep, onSlot: via.onSlot, name: via.nameFor("judge", retryAdapter.id) + "-r1", label: via.labelFor("judge") }
888
1093
  : undefined;
889
1094
  // the retry IS a second acceptanceGate call: one code path, one parser, zero new parse leniency.
890
1095
  a = await invokeJudge(retryAdapter, retry.model, retryJvia, configuredEffort(ctx.cfg, retry));
891
1096
  a = { ...a, meta: { ...a.meta, judgeRetry: { flaked: flakedKey, retried: channelKey({ adapter: retry.adapter, model: retry.model }) } } };
892
1097
  }
1098
+ // OBS-1168(b): the re-routed seat could not launch either — no seat produced a verdict, so this is
1099
+ // an infra park over whatever the deterministic gates proved, never a charge against the worker.
1100
+ if (a.meta?.cause === "seat-launch-failed")
1101
+ a = { ...a, meta: { ...a.meta, classification: "infra", infra: true, retryable: false } };
1102
+ else if (!cancelled && Array.isArray(a.meta?.judgment))
1103
+ a = await adjudicate(a);
893
1104
  // No dispatch, no key: a deterministic-oracle round writes no `invocations` field rather than an
894
1105
  // empty array a reader could mistake for "measured, and it cost nothing".
895
1106
  return { result: invocationSpans.length ? { ...a, meta: { ...a.meta, invocations: invocationSpans } } : a, invocations };
@@ -908,6 +1119,8 @@ export async function runGates(task, ctx) {
908
1119
  const captured = await captureLlmDispatches(ctx.adapters, run);
909
1120
  invocations.push(...captured.invocations);
910
1121
  const rv = captured.value;
1122
+ if (cancelled)
1123
+ return rv; // a cancelled seat's non-answer says nothing about the seat
911
1124
  if (rv.meta?.noVerdict === true || rv.meta?.unparseable === true) {
912
1125
  await ctx.onGate?.({ phase: "note", gate: "review", name: "review-no-verdict", payload: { ...rv.meta }, result: rv });
913
1126
  if (typeof rv.meta.reviewer === "string") {
@@ -938,7 +1151,7 @@ export async function runGates(task, ctx) {
938
1151
  const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
939
1152
  const carriedAuthors = ctx.carriedAuthors ?? [];
940
1153
  let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
941
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
1154
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
942
1155
  // OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
943
1156
  // a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
944
1157
  // verdict never enters results; an exhausted pool preserves its cause.
@@ -948,9 +1161,29 @@ export async function runGates(task, ctx) {
948
1161
  let retryPrior = [...priorReviewers];
949
1162
  const routes = [];
950
1163
  let hop = 0;
951
- while ((rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
1164
+ // OBS-1196: seats already re-asked after a malformed verdict — once per seat, so never unbounded.
1165
+ const reemitted = new Set();
1166
+ while (!cancelled && (rv.meta?.unparseable === true || rv.meta?.noVerdict === true) && typeof rv.meta.reviewer === "string") {
952
1167
  hop++;
953
1168
  const flaked = rv.meta.reviewer;
1169
+ const retryVia = via
1170
+ ? { ...via, nameFor: (role, adapter) => via.nameFor(role, adapter) + `-r${hop}` }
1171
+ : undefined;
1172
+ // OBS-1196: a malformed (unparseable) verdict is a delivery defect of THIS seat, not a reason to drop it. Ask the
1173
+ // same seat once more on the same subject; reviewGate mints a fresh nonce, so only a new, whole,
1174
+ // nonce-bound verdict can answer — the malformed bytes are never salvaged into one.
1175
+ if (rv.meta.cause === "malformed-verdict" && rv.meta.closureInvalid !== true && !reemitted.has(flaked)) {
1176
+ reemitted.add(flaked);
1177
+ const others = ctx.channels.map(channelKey).filter((key) => key !== flaked);
1178
+ const again = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, others, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
1179
+ if (again.meta?.noEligibleReviewer !== true) {
1180
+ await ctx.onGate?.({ phase: "note", gate: "review", name: "review-reemission", payload: { reviewer: flaked, cause: "malformed-verdict",
1181
+ delivered: again.meta?.unparseable !== true && again.meta?.noVerdict !== true }, result: again });
1182
+ routes.push(`review re-emission (same seat, fresh nonce): ${flaked} produced a malformed verdict; asked once more`);
1183
+ rv = { ...again, details: `${routes.join("\n")}\n${again.details}`, meta: { ...again.meta, reviewReemission: { reviewer: flaked } } };
1184
+ continue;
1185
+ }
1186
+ }
954
1187
  const emptyOutput = rv.meta.cause === "empty-output";
955
1188
  if (emptyOutput) {
956
1189
  await ctx.onGate?.({
@@ -959,9 +1192,6 @@ export async function runGates(task, ctx) {
959
1192
  result: { ...rv, meta: { ...rv.meta, skipped: true } },
960
1193
  });
961
1194
  }
962
- const retryVia = ctx.via
963
- ? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + `-r${hop}` }
964
- : undefined;
965
1195
  const flakedAdapter = flaked.slice(0, flaked.indexOf(":"));
966
1196
  const adapterExclusions = ctx.channels.filter((c) => c.adapter === flakedAdapter).map(channelKey);
967
1197
  // RF-1: the retry filters by the floor reviewGate resolves — author tier, task floor, review.floor
@@ -995,11 +1225,14 @@ export async function runGates(task, ctx) {
995
1225
  break;
996
1226
  }
997
1227
  }
998
- if (rv.meta?.noVerdict === true) {
1228
+ if (rv.meta?.noVerdict === true
1229
+ || (rv.meta?.unparseable === true && rv.meta.closureInvalid !== true && typeof rv.meta.reviewer === "string")) {
999
1230
  // Terminal: every eligible seat returned no verdict. The carried materials stay open — an infra
1000
1231
  // row is not a passing review — and the sibling judge result is untouched beside it.
1232
+ // OBS-1196: an exhausted pool of UNDELIVERED (unparseable) verdicts is the same non-verdict: an infra
1233
+ // park, never a worker retry. A delivered verdict that breaks the closure protocol keeps its path.
1001
1234
  const carried = (ctx.carriedFindings ?? []).filter((f) => f.class === "review:material").map((f) => f.fingerprint);
1002
- rv = { ...rv, meta: { ...rv.meta, classification: "infra", infra: true, carriedFindings: carried } };
1235
+ rv = { ...rv, meta: { ...rv.meta, noVerdict: true, classification: "infra", infra: true, carriedFindings: carried } };
1003
1236
  }
1004
1237
  return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
1005
1238
  };
@@ -1058,11 +1291,27 @@ export async function runGates(task, ctx) {
1058
1291
  else
1059
1292
  selected = [...new Set([...selected, ...required])].sort();
1060
1293
  }
1061
- if (ctx.selectionReason)
1294
+ // OBS-635: a screen buys nothing when the full suite that must follow it already has a qualified
1295
+ // green on this exact identity (read BEFORE any screen), or when the harness-measured screen costs
1296
+ // at least 75 % of it — then the full suite runs in the battery instead. Unknown timing keeps the
1297
+ // screen. A selected-only green never answers here: its identity names its selection.
1298
+ let promotion;
1299
+ if (selected) {
1300
+ const hit = verdictStore.get(await fullTestIdentity());
1301
+ const costRatio = screenCostRatio(ctx.baseline, selected);
1302
+ if (hit?.pass === true && !isInfraResult(hit) && await certifiesFullManifest(hit))
1303
+ promotion = { reason: "full-green-cache" };
1304
+ else if (costRatio !== undefined && costRatio >= SCREEN_PROMOTION_RATIO)
1305
+ promotion = { reason: "screen-cost-promoted", costRatio };
1306
+ if (promotion)
1307
+ selected = undefined;
1308
+ }
1309
+ if (ctx.selectionReason || promotion)
1062
1310
  selectionDecision = {
1063
- scope: selected ? "selected" : "full", reason: selected ? selectionReason
1064
- : selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason,
1311
+ scope: selected ? "selected" : "full", reason: promotion?.reason ?? (selected ? selectionReason
1312
+ : selectionReason === "known-failing-files" ? "unsupported-selection-full-suite" : selectionReason),
1065
1313
  requiredFiles: [...(ctx.requiredRepairTests ?? [])],
1314
+ ...(promotion?.costRatio !== undefined ? { costRatio: promotion.costRatio } : {}),
1066
1315
  };
1067
1316
  await runBattery(selected ? { ...ctx.commands, test: testCommandForFiles(ctx.commands.test, selected) } : ctx.commands, selected, enabled("test") ? ["test"] : []);
1068
1317
  if (failed())
@@ -1086,28 +1335,73 @@ export async function runGates(task, ctx) {
1086
1335
  // enough: an acceptance-first await withholds a completed review behind a slow/hung judge and a
1087
1336
  // process death can lose that already-earned verdict. The returned result is still sorted into
1088
1337
  // GATE_NAMES order by done(); the event stream truthfully records each independent completion.
1089
- const judged = judging?.then((outcome) => withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result)));
1090
- const reviewed = reviewing?.then((outcome) => record(outcome));
1091
- if (executionSignal()) {
1092
- // A cancelled sibling still owns a process until it unwinds; do not settle the task early.
1093
- const settled = await Promise.allSettled([judged, reviewed]);
1094
- const rejected = settled.find((result) => result.status === "rejected");
1095
- if (rejected?.status === "rejected")
1096
- throw rejected.reason;
1097
- executionSignal()?.throwIfAborted();
1098
- }
1099
- else
1100
- await Promise.all([judged, reviewed]);
1338
+ // OBS-1168(c): a sibling that throws, or that ends seatless (no seat could launch, so the round can
1339
+ // only park infra), cancels the other — its seats are closed — and the round still AWAITS it, with
1340
+ // or without an execution policy, so no verdict of a settled round publishes after its engagement.
1341
+ const seatless = (r) => r.meta?.cause === "seat-launch-failed" && r.meta?.infra === true;
1342
+ let failure;
1343
+ const fail = (reason) => {
1344
+ if (!cancelled)
1345
+ failure ??= { reason };
1346
+ cancelSemantic();
1347
+ };
1348
+ const judged = judging?.then(async (outcome) => {
1349
+ if (cancelled)
1350
+ return;
1351
+ await withJudgeInvocationEvidence(outcome.invocations, () => record(outcome.result));
1352
+ if (seatless(outcome.result))
1353
+ cancelSemantic();
1354
+ }).catch(fail);
1355
+ const reviewed = reviewing?.then(async (outcome) => {
1356
+ if (cancelled)
1357
+ return;
1358
+ await record(outcome);
1359
+ if (seatless(outcome))
1360
+ cancelSemantic();
1361
+ }).catch(fail);
1362
+ // ponytail: a headless seat has no slot to close; it is awaited to its own timeout, never abandoned.
1363
+ await Promise.all([judged, reviewed]);
1364
+ await Promise.all(closing);
1365
+ if (failure)
1366
+ throw failure.reason;
1367
+ executionSignal()?.throwIfAborted();
1101
1368
  if (failed())
1102
1369
  return done();
1103
1370
  }
1371
+ // OBS-635: the semantic gates ran oracles and vendor CLIs in this worktree after an in-battery full
1372
+ // suite spoke, and its green stands only for the identity it measured. Oracle dirt withdraws it; a
1373
+ // changed or unmeasurable identity — tree, command, baseline, environment, dependency resolution,
1374
+ // capacity, protocol, lifecycle, full manifest — buys a fresh merge-candidate suite below.
1375
+ let rerunFull = false;
1376
+ listing = undefined;
1377
+ if (fullInBattery && ctx.commands.test !== undefined && (enabled("acceptance") || enabled("review"))) {
1378
+ const dirt = await dirtyWorktree();
1379
+ if (dirt) {
1380
+ const refusal = withTelemetry(await dirtyRoundRefusal("test", dirt));
1381
+ results[results.findIndex((r) => r.gate === "test")] = refusal;
1382
+ await ctx.onGate?.({ phase: "end", gate: "test", result: refusal });
1383
+ return done();
1384
+ }
1385
+ const now = await fullTestIdentity();
1386
+ // D-598: an unmeasurable lifecycle on either side is not comparable to anything (VerdictStore R41
1387
+ // refuses it); two `unknown`s hashing equal is not an unchanged identity, so the suite reruns.
1388
+ const measurable = (id) => id?.envParts?.verification?.lifecycle !== "unknown";
1389
+ rerunFull = !now || !batteryFullIdentity || !measurable(now) || !measurable(batteryFullIdentity)
1390
+ || verificationIdentityKey(now) !== verificationIdentityKey(batteryFullIdentity)
1391
+ || !(await certifiesFullManifest(results.find((r) => r.gate === "test")));
1392
+ if (rerunFull) {
1393
+ // The in-battery row keeps its own interval; the replacement measures from zero.
1394
+ spans.delete("test");
1395
+ loadSamples.delete("test");
1396
+ }
1397
+ }
1104
1398
  // The merge-candidate round: every other gate is green, so THIS round is the one that can merge —
1105
1399
  // the full suite runs on the exact gated commit before the pipeline reports green. Nothing merges
1106
1400
  // on a subset (spec: "nothing merges without a complete green suite"). Its verdict replaces the
1107
1401
  // screen's entry in the returned record (one `test` entry), and `fullSuite` says which suite spoke
1108
1402
  // while `selectedTests` keeps what the screen ran. In the stream, a held screen is superseded (one
1109
1403
  // `test` end event); a published screen keeps its own earlier event and this is the second.
1110
- if (selected) {
1404
+ if (selected || rerunFull) {
1111
1405
  await emitStart("test");
1112
1406
  // This is the last shell command a round can run — the judge's named-test oracle (acceptance.ts)
1113
1407
  // may have run one before it, and every gate between the battery and here reads commits only, so
@@ -1118,20 +1412,12 @@ export async function runGates(task, ctx) {
1118
1412
  let cached = false;
1119
1413
  let identity;
1120
1414
  if (ctx.commands.test !== undefined) {
1121
- identity = await computeVerificationIdentity({
1122
- worktree: ctx.worktree,
1123
- gate: "test",
1124
- scope: ctx.verificationScope,
1125
- command: ctx.commands.test,
1126
- baseline: ctx.baseline,
1127
- selectedSet: undefined,
1128
- capacity: resolvedCapacity(),
1129
- });
1415
+ identity = await fullTestIdentity();
1130
1416
  const hit = verdictStore.get(identity);
1131
1417
  if (hit)
1132
1418
  classifySignalOnlyTest(hit);
1133
1419
  if (hit && identity && !isInfraResult(hit) && (hit.pass || (ctx.verificationScope ?? "battery") === "battery")
1134
- && !(await discardCachedRed("test", hit))) {
1420
+ && !(await discardCachedRed("test", hit)) && await certifiesFullManifest(hit)) {
1135
1421
  full = formatReusedRow(hit, identity);
1136
1422
  cached = true;
1137
1423
  await noteReuse("test", full, identity);
@@ -80,11 +80,49 @@ export declare function verifyManifestReport(opts: {
80
80
  report: TestReport | undefined;
81
81
  killedFile?: string;
82
82
  hangBudgetMs?: number;
83
+ /** Active elapsed time (detected suspend subtracted) versus raw wall service at the kill. */
84
+ hangActiveMs?: number;
85
+ hangWallMs?: number;
83
86
  }): ManifestVerdict;
84
87
  /** How much longer than its baseline measurement one file may legitimately run before it is a hang. */
85
88
  export declare const FILE_HANG_SLACK = 3;
86
89
  export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
87
90
  export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
91
+ /**
92
+ * OBS-953 (+add): a per-file hang budget counted in raw wall time charged a lid-close to the file that
93
+ * was running. Each poll compares how far the wall clock and the monotonic clock advanced since the
94
+ * last one. Wall advancing more than monotonic by over CLOCK_JUMP_SLACK_MS is a DETECTED discontinuity
95
+ * (`host-suspend`): its offset is subtracted from the active elapsed time of every file started before
96
+ * it ended. A poll that arrives overdue while both clocks advanced together is recorded as `unknown`
97
+ * and subtracts nothing — an ambiguous gap never excuses an active hang. Whether the host really slept
98
+ * is not provable from these clocks (a manually set clock jumps the same way); only the offset is.
99
+ */
100
+ export declare const CLOCK_JUMP_SLACK_MS = 1000;
101
+ export interface HostInterruption {
102
+ kind: "host-suspend" | "unknown";
103
+ /** Wall time of the previous poll and of the poll that observed the gap. */
104
+ from: number;
105
+ to: number;
106
+ wallMs: number;
107
+ monoMs: number;
108
+ /** The wall-over-monotonic offset of a detected suspend; 0 for an unknown gap. */
109
+ subtractedMs: number;
110
+ }
111
+ export interface HangClocks {
112
+ wall: () => number;
113
+ mono: () => number;
114
+ }
115
+ export declare const setHangClocksForTests: (clocks: HangClocks) => void;
116
+ export declare const resetHangClocksForTests: () => void;
117
+ /** A poll gap worth recording, or undefined for an on-time poll. */
118
+ export declare function classifyPollGap(prev: {
119
+ wall: number;
120
+ mono: number;
121
+ }, now: {
122
+ wall: number;
123
+ mono: number;
124
+ }, pollMs: number): HostInterruption | undefined;
125
+ export declare const runWithInterruptionSink: <T>(sink: (interruption: HostInterruption) => void, run: () => Promise<T>) => Promise<T>;
88
126
  export interface ManifestRunResult {
89
127
  evidenceReceipt: GateEvidenceReceipt;
90
128
  evidenceReceipts: GateEvidenceReceipt[];
@@ -94,6 +132,10 @@ export interface ManifestRunResult {
94
132
  report: TestReport | undefined;
95
133
  killedFile?: string;
96
134
  hangBudgetMs?: number;
135
+ /** A hang's active elapsed time (suspend subtracted) and its raw wall service, kept apart. */
136
+ hangActiveMs?: number;
137
+ hangWallMs?: number;
138
+ interruptions: HostInterruption[];
97
139
  /** The child's own pid (its process GROUP id too, since it is spawned detached) — for a caller
98
140
  * that wants to prove the group is really gone after a hang kill (`process.kill(-pid, 0)` throws). */
99
141
  pid?: number;