tickmarkr 2.5.5 → 2.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +3 -1
  2. package/dist/cli/commands/approve.js +64 -3
  3. package/dist/cli/commands/doctor.d.ts +10 -0
  4. package/dist/cli/commands/fleet.js +4 -0
  5. package/dist/cli/commands/plan.js +20 -4
  6. package/dist/cli/commands/verify.js +102 -81
  7. package/dist/config/config.d.ts +11 -0
  8. package/dist/config/config.js +21 -12
  9. package/dist/config/fleet-overlay.js +55 -17
  10. package/dist/drivers/index.js +2 -1
  11. package/dist/drivers/orca.d.ts +21 -1
  12. package/dist/drivers/orca.js +209 -27
  13. package/dist/gates/baseline.d.ts +14 -3
  14. package/dist/gates/baseline.js +61 -14
  15. package/dist/gates/cache.d.ts +14 -13
  16. package/dist/gates/cache.js +17 -5
  17. package/dist/gates/llm.d.ts +3 -0
  18. package/dist/gates/llm.js +11 -0
  19. package/dist/gates/review.d.ts +28 -3
  20. package/dist/gates/review.js +106 -14
  21. package/dist/gates/run-gates.d.ts +6 -1
  22. package/dist/gates/run-gates.js +299 -101
  23. package/dist/gates/test-manifest.d.ts +30 -1
  24. package/dist/gates/test-manifest.js +113 -39
  25. package/dist/gates/test-reporter.js +14 -6
  26. package/dist/graph/graph.d.ts +2 -0
  27. package/dist/graph/graph.js +45 -1
  28. package/dist/run/consult.js +5 -4
  29. package/dist/run/daemon.d.ts +2 -0
  30. package/dist/run/daemon.js +480 -99
  31. package/dist/run/execution-budget.d.ts +25 -0
  32. package/dist/run/execution-budget.js +142 -0
  33. package/dist/run/git.d.ts +40 -0
  34. package/dist/run/git.js +89 -4
  35. package/dist/run/journal.d.ts +3 -1
  36. package/dist/run/journal.js +38 -16
  37. package/dist/run/lease.d.ts +44 -0
  38. package/dist/run/lease.js +226 -3
  39. package/dist/run/recovery.d.ts +8 -0
  40. package/dist/run/recovery.js +25 -0
  41. package/dist/run/repair-selection.d.ts +12 -0
  42. package/dist/run/repair-selection.js +56 -0
  43. package/dist/run/stall.d.ts +6 -1
  44. package/dist/run/stall.js +60 -3
  45. package/dist/tui/ink/fleet-app.d.ts +10 -2
  46. package/dist/tui/ink/fleet-app.js +33 -15
  47. package/package.json +2 -2
  48. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
@@ -2,6 +2,8 @@ import { existsSync, readFileSync } from "node:fs";
2
2
  import { availableParallelism, loadavg } from "node:os";
3
3
  import { join } from "node:path";
4
4
  import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
5
+ import { executionSignal } from "../run/execution-budget.js";
6
+ import { isVitestTestCommand, manifestFileCount } from "./test-manifest.js";
5
7
  // incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
6
8
  // codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
7
9
  // a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
@@ -464,6 +466,27 @@ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) =
464
466
  ceilingMs: effectiveCeilingMs({ durationMs }),
465
467
  ...(invalidatingLines.length ? { invalidatingLines } : {}),
466
468
  });
469
+ // OBS-1044: a vitest command's count is the manifest discovery seam's (the same listing the gate
470
+ // asks the report to certify), never a sum of every summary line in stdout — tests that spawn
471
+ // nested runners echo their own summaries through the filter and inflated a baseline by two files,
472
+ // so any diff that changed the nested echo read as a deficit. The listing is a collection, not a
473
+ // run: the suite still runs exactly once here. Any other command keeps the stdout sum.
474
+ const captureFileCount = async (cwd, cmd, raw) => {
475
+ if (!isVitestTestCommand(cmd, cwd))
476
+ return { fileCount: runnerFileCount(raw) };
477
+ const fileCount = await manifestFileCount(cmd, cwd);
478
+ return fileCount === null ? { fileCount } : { fileCount, fileCountSource: "manifest" };
479
+ };
480
+ /** OBS-1044: a cached vitest entry whose count is a stdout sum may be inflated; only a manifest-derived
481
+ * count is safe to apply as a deficit floor. A null count compares nothing and is safe as-is. */
482
+ export function staleFileCountCommands(baseline, commands, cwd) {
483
+ return Object.entries(commands)
484
+ .filter(([name, cmd]) => {
485
+ const entry = baseline.commands[name];
486
+ return entry?.fileCount != null && entry.fileCountSource !== "manifest" && isVitestTestCommand(cmd, cwd);
487
+ })
488
+ .map(([name]) => name);
489
+ }
467
490
  export async function captureBaseline(cwd, commands) {
468
491
  const base = { commands: {} };
469
492
  for (const [name, cmd] of Object.entries(commands)) {
@@ -495,7 +518,7 @@ export async function captureBaseline(cwd, commands) {
495
518
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
496
519
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
497
520
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
498
- base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), fileCount: runnerFileCount(raw) };
521
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "ceiling-kill"), ...await captureFileCount(cwd, cmd, raw) };
499
522
  continue;
500
523
  }
501
524
  // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
@@ -509,7 +532,7 @@ export async function captureBaseline(cwd, commands) {
509
532
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
510
533
  + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
511
534
  + `First invalidating line: ${invalidatingLines[0]}`);
512
- base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), fileCount: runnerFileCount(raw) };
535
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines), ...await captureFileCount(cwd, cmd, raw) };
513
536
  continue;
514
537
  }
515
538
  // OBS-966: a worker RPC timeout is infra even beside an all-green summary.
@@ -517,7 +540,7 @@ export async function captureBaseline(cwd, commands) {
517
540
  const runnerVerdict = classifyRunnerOutput(raw, r.code);
518
541
  if (runnerVerdict === "infra") {
519
542
  console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence — it recorded NO verdict; nothing is forgiven for this command`);
520
- base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), fileCount: runnerFileCount(raw) };
543
+ base.commands[name] = { ...invalidCaptureEntry(durationMs, "infra", withoutVitestEchoBlocks(combinedOutput).filter((line) => isInfraLine(line.replace(ANSI_RE, "")))), ...await captureFileCount(cwd, cmd, raw) };
521
544
  continue;
522
545
  }
523
546
  base.commands[name] = {
@@ -529,7 +552,7 @@ export async function captureBaseline(cwd, commands) {
529
552
  missingCommand: missingConfiguredCommand(cmd, r),
530
553
  durationMs,
531
554
  ...fileTiming(raw, durationMs),
532
- fileCount: runnerFileCount(raw),
555
+ ...await captureFileCount(cwd, cmd, raw),
533
556
  ceilingMs: effectiveCeilingMs({ durationMs }),
534
557
  // T7: the world this measurement was taken in, so a later reader can ask whether its own world
535
558
  // is the same one. Recorded from THIS command's own shell result, never re-derived here.
@@ -615,7 +638,10 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
615
638
  // T7: every verdict below carries the capacity ITS OWN command ran under, taken off the shell
616
639
  // result rather than re-derived after the fact. The skip row above ran no command and therefore
617
640
  // states no capacity — a row that never divided the machine must not claim that it did.
641
+ let recoveryBlocked;
618
642
  const record = (g) => {
643
+ if (recoveryBlocked)
644
+ g = { ...g, meta: { ...g.meta, recoveryBlocked } };
619
645
  const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
620
646
  const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
621
647
  const withRerun = rerunOf ? {
@@ -697,19 +723,37 @@ export async function compareToBaseline(cwd, commands, baseline, enabled, opts =
697
723
  // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
698
724
  const classification = classifyFreshRunnerOutput(entry, raw, r.code);
699
725
  if (name === "test" && classification === "infra" && !rerunOf && !opts.infraRerun) {
700
- const waitedMs = await waitForCalmWindow();
701
- const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
702
- results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, infraRerun: provenance }));
703
- continue;
726
+ const waitedMs = await waitForCalmWindow(executionSignal());
727
+ if (opts.authorizeRetry && !calmWindowReady()) {
728
+ recoveryBlocked = "calm window unavailable within the existing wait ceiling";
729
+ }
730
+ else if (opts.authorizeRetry && !opts.authorizeRetry("infra")) {
731
+ recoveryBlocked = "infrastructure retry allowance exhausted or subject unavailable";
732
+ }
733
+ else {
734
+ const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
735
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, infraRerun: provenance }));
736
+ continue;
737
+ }
704
738
  }
705
739
  // OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
706
740
  // own baseline measurement. The first read buys one calm rerun here, never a worker repair.
707
741
  if (name === "test" && classification !== "infra" && failing.length && !rerunOf && !opts.infraRerun
708
742
  && hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
709
- const waitedMs = await waitForCalmWindow();
710
- const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
711
- results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, rerunOf: provenance }));
712
- continue;
743
+ const waitedMs = await waitForCalmWindow(executionSignal());
744
+ if (opts.authorizeRetry && !calmWindowReady()) {
745
+ recoveryBlocked = "calm window unavailable within the existing wait ceiling";
746
+ }
747
+ else if (opts.authorizeRetry && !opts.authorizeRetry("host-starved")) {
748
+ recoveryBlocked = "verification retry allowance exhausted or subject unavailable";
749
+ }
750
+ else {
751
+ // Compatibility with OBS-896, not an infrastructure reclassification: a persistent
752
+ // timeout remains the ordinary conservative verdict after the single bounded remeasure.
753
+ const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
754
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { ...opts, rerunOf: provenance }));
755
+ continue;
756
+ }
713
757
  }
714
758
  if (classification === "infra") {
715
759
  const evidence = failing.length
@@ -827,11 +871,14 @@ const DEFAULT_CALM = {
827
871
  let calm = DEFAULT_CALM;
828
872
  export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
829
873
  export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
830
- export async function waitForCalmWindow() {
874
+ export function calmWindowReady() { return calm.loadProvider() <= calm.calmLoad(); }
875
+ export async function waitForCalmWindow(signal) {
831
876
  const started = Date.now();
832
877
  while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
833
- await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
878
+ signal?.throwIfAborted();
879
+ await new Promise((resolve) => setTimeout(resolve, signal ? Math.min(calm.pollMs, 100) : calm.pollMs));
834
880
  }
881
+ signal?.throwIfAborted();
835
882
  return Date.now() - started;
836
883
  }
837
884
  /** Summary totals survive digit-normalized fingerprints and exclude test-owned echoed output. */
@@ -1,6 +1,6 @@
1
1
  import type { Baseline } from "./baseline.js";
2
2
  import type { GateResult } from "./types.js";
3
- import { type RunCapacity } from "../run/git.js";
3
+ import { type RunCapacity, type VerificationProtocol } from "../run/git.js";
4
4
  export declare const DEFAULT_VERDICT_CACHE_BOUND = 128;
5
5
  export declare function setVerdictCacheBoundForTests(bound: number | undefined): void;
6
6
  export declare function resetVerdictCacheBoundForTests(): void;
@@ -16,15 +16,19 @@ export interface GateEnvironmentInput {
16
16
  capacity?: RunCapacity;
17
17
  selectedSet?: readonly string[];
18
18
  scope?: VerificationScope;
19
+ /** R41: the verification protocol + runner lifecycle policy; defaults to this process's. */
20
+ verification?: VerificationProtocol;
21
+ }
22
+ export interface EnvironmentParts {
23
+ nodeRuntime: string;
24
+ lockfile: string;
25
+ capacity: RunCapacity;
26
+ selectedSet?: readonly string[];
27
+ verification: VerificationProtocol;
19
28
  }
20
29
  export declare function environmentFingerprint(env: GateEnvironmentInput): {
21
30
  fingerprint: string;
22
- parts: {
23
- nodeRuntime: string;
24
- lockfile: string;
25
- capacity: RunCapacity;
26
- selectedSet?: readonly string[];
27
- };
31
+ parts: EnvironmentParts;
28
32
  };
29
33
  export type VerificationScope = "battery" | "tip" | "standalone";
30
34
  export interface VerificationIdentity {
@@ -38,12 +42,7 @@ export interface VerificationIdentity {
38
42
  command: string;
39
43
  baseline: string;
40
44
  environment: string;
41
- envParts?: {
42
- nodeRuntime: string;
43
- lockfile: string;
44
- capacity: RunCapacity;
45
- selectedSet?: readonly string[];
46
- };
45
+ envParts?: EnvironmentParts;
47
46
  }
48
47
  export declare function computeVerificationIdentity(params: {
49
48
  worktree: string;
@@ -56,6 +55,7 @@ export declare function computeVerificationIdentity(params: {
56
55
  tree?: string;
57
56
  lockfile?: string;
58
57
  nodeRuntime?: string;
58
+ verification?: VerificationProtocol;
59
59
  }): Promise<VerificationIdentity | undefined>;
60
60
  export declare function verificationIdentityKey(id: VerificationIdentity): string;
61
61
  export declare function formatReusedDetails(originalDetails: string, id: VerificationIdentity): string;
@@ -89,6 +89,7 @@ export declare class VerdictStore {
89
89
  readonly dir: string;
90
90
  constructor(dir: string);
91
91
  private initSequenceFromDisk;
92
+ private static unknownPolicy;
92
93
  get(id?: VerificationIdentity): CachedVerdict | undefined;
93
94
  set(id: VerificationIdentity | undefined, verdict: GateResult | CachedVerdict): boolean;
94
95
  size(): number;
@@ -3,7 +3,7 @@ import { execSync } from "node:child_process";
3
3
  import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, rmSync, unlinkSync, writeFileSync } from "node:fs";
4
4
  import { dirname, join, resolve } from "node:path";
5
5
  import { tmpdir } from "node:os";
6
- import { describeCapacity, resolvedCapacity, shGit } from "../run/git.js";
6
+ import { describeCapacity, resolvedCapacity, shGit, verificationProtocol } from "../run/git.js";
7
7
  import { shq } from "../adapters/types.js";
8
8
  export const DEFAULT_VERDICT_CACHE_BOUND = 128;
9
9
  let testCacheBound;
@@ -83,17 +83,23 @@ export function environmentFingerprint(env) {
83
83
  const cap = env.capacity ?? resolvedCapacity();
84
84
  const capacity = { forkCap: cap.forkCap, cores: cap.cores };
85
85
  const selectedSet = env.selectedSet ? [...env.selectedSet].sort() : undefined;
86
+ const verification = env.verification ?? verificationProtocol(process.env, env.worktree ?? process.cwd());
87
+ // R41: the protocol and the EFFECTIVE lifecycle are IN the hashed payload, so every entry written
88
+ // before this stamp — green or red — keys differently and is never answered; no store surgery is
89
+ // needed. `source` is provenance (kept in parts, printed on the row) and never enters the key: an
90
+ // explicit `false` and an npmrc `false` are the same policy for the child that ran.
86
91
  const payload = canonicalJson({
87
92
  nodeRuntime,
88
93
  lockfile,
89
94
  capacity,
90
95
  selectedSet: selectedSet ?? null,
91
96
  scope: env.scope ?? "battery",
97
+ verification: { protocol: verification.protocol, lifecycle: verification.lifecycle },
92
98
  });
93
99
  const fingerprint = createHash("sha256").update(payload).digest("hex").slice(0, 16);
94
100
  return {
95
101
  fingerprint,
96
- parts: { nodeRuntime, lockfile, capacity, selectedSet },
102
+ parts: { nodeRuntime, lockfile, capacity, selectedSet, verification },
97
103
  };
98
104
  }
99
105
  export async function computeVerificationIdentity(params) {
@@ -108,6 +114,7 @@ export async function computeVerificationIdentity(params) {
108
114
  lockfile: params.lockfile,
109
115
  nodeRuntime: params.nodeRuntime,
110
116
  scope: params.scope,
117
+ verification: params.verification,
111
118
  });
112
119
  return {
113
120
  gate: params.gate,
@@ -130,7 +137,7 @@ export function verificationIdentityKey(id) {
130
137
  export function formatReusedDetails(originalDetails, id) {
131
138
  const unadorned = originalDetails.replace(/^reused verdict \(identity: [^)]+\):\s*/, "");
132
139
  const envDesc = id.envParts
133
- ? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}]`
140
+ ? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}, protocol=${id.envParts.verification.protocol}, lifecycle=${id.envParts.verification.lifecycle} (${id.envParts.verification.source})]`
134
141
  : "";
135
142
  const gate = id.gate ?? "gate";
136
143
  const prefix = `reused ${id.scope === "tip" ? "tip " : ""}verdict (identity: gate=${gate} tree=${id.tree}${id.worktree ? ` worktree=${id.worktree}` : ""} command=${id.command} baseline=${id.baseline} env=${id.environment}${envDesc})`;
@@ -258,8 +265,13 @@ export class VerdictStore {
258
265
  // ignore
259
266
  }
260
267
  }
268
+ // R41: an identity whose lifecycle policy could not be measured is never answered and never
269
+ // stored — an unknown policy is not comparable to anything, so the battery runs the command.
270
+ static unknownPolicy(id) {
271
+ return id.envParts?.verification?.lifecycle === "unknown";
272
+ }
261
273
  get(id) {
262
- if (!id || !id.tree)
274
+ if (!id || !id.tree || VerdictStore.unknownPolicy(id))
263
275
  return undefined;
264
276
  const key = verificationIdentityKey(id);
265
277
  const p = join(this.dir, `verdict-${key}.json`);
@@ -276,7 +288,7 @@ export class VerdictStore {
276
288
  return undefined;
277
289
  }
278
290
  set(id, verdict) {
279
- if (!id || !id.tree)
291
+ if (!id || !id.tree || VerdictStore.unknownPolicy(id))
280
292
  return false;
281
293
  if (isInfraResult(verdict))
282
294
  return false;
@@ -62,11 +62,14 @@ export interface LlmRunResult {
62
62
  exitCode?: number;
63
63
  timedOut: boolean;
64
64
  launchNeverStarted?: boolean;
65
+ /** OBS-1039: seat-authored bytes stayed under REVIEW_SILENT_BYTE_FLOOR at the first liveness beat. */
66
+ silentAtBeat?: boolean;
65
67
  seatAuthoredBytes?: number;
66
68
  }
67
69
  export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
68
70
  export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
69
71
  export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
72
+ export declare const REVIEW_SILENT_BYTE_FLOOR = 64;
70
73
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
71
74
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
72
75
  export declare function dewrapPaneVerdict(out: string, nonce: string): string;
package/dist/gates/llm.js CHANGED
@@ -300,6 +300,10 @@ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
300
300
  return trailer ? seat.slice(0, trailer.index) : seat;
301
301
  }
302
302
  export const REVIEW_FIRST_LIVENESS_MS = 30_000;
303
+ // OBS-1039: a seat that wrote ten bytes and went quiet escaped the zero-byte beat and sat to the
304
+ // ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
305
+ // re-routed then, not at the ceiling. Pane path only; a headless runner buffers and keeps its ceiling.
306
+ export const REVIEW_SILENT_BYTE_FLOOR = 64;
303
307
  async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
304
308
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
305
309
  try {
@@ -356,6 +360,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
356
360
  let out;
357
361
  let timedOut = false;
358
362
  let launchNeverStarted = false;
363
+ let silentAtBeat = false;
359
364
  let seatAuthoredBytes = 0;
360
365
  const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
361
366
  if (!gatePrompt) {
@@ -412,6 +417,11 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
412
417
  forceClose = true;
413
418
  break;
414
419
  }
420
+ if (seatAuthoredBytes < REVIEW_SILENT_BYTE_FLOOR) {
421
+ silentAtBeat = true;
422
+ forceClose = true;
423
+ break;
424
+ }
415
425
  }
416
426
  // Producing reviews own their full ceiling; inactivity is not a review verdict.
417
427
  if (reviewing)
@@ -458,6 +468,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
458
468
  ...(Number.isFinite(exitCode) ? { exitCode } : {}),
459
469
  timedOut,
460
470
  launchNeverStarted,
471
+ silentAtBeat,
461
472
  seatAuthoredBytes,
462
473
  };
463
474
  }
@@ -64,6 +64,13 @@ export declare function matchClosureId(candidate: unknown, fingerprints: Iterabl
64
64
  * all route through matchClosureId.
65
65
  */
66
66
  export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
67
+ /**
68
+ * OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
69
+ * it answered about the materials and missed the id (a retyped, truncated or paraphrased fingerprint).
70
+ * That is a no-verdict about the carried work (re-route), not a parse defect. A verdict that omits a
71
+ * list, carries a non-string or duplicates an id stays malformed: its shape, not its ids, is wrong.
72
+ */
73
+ export declare function isReviewClosureMismatch(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
67
74
  export type ReviewerFloorCause = "author-tier" | "task-floor" | "config" | "prior-reviewer";
68
75
  /**
69
76
  * RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
@@ -101,12 +108,30 @@ export declare function pickReviewer(author: Assignment, channels: BillingChanne
101
108
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
102
109
  floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
103
110
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
104
- onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>): BillingChannel | null;
105
- export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent";
111
+ onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>): BillingChannel | null;
112
+ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch";
106
113
  /**
107
114
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
108
115
  * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
109
116
  * judgement rather than a guarantee made by this renderer.
110
117
  */
111
118
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
112
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[]): Promise<GateResult>;
119
+ /**
120
+ * OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
121
+ * never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
122
+ * its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
123
+ */
124
+ export declare function carriedAuthorVendors(channels: BillingChannel[], carriedAuthors?: readonly string[]): Set<string>;
125
+ /**
126
+ * OBS-1020: the compiled goal is the contract. After `resume --graph-changed` the worktree's copy of
127
+ * the spec is the pre-change text on the integration branch, so a reviewer that reads it grades a
128
+ * superseded contract. The daemon's repository root is where specs and planning records are current.
129
+ */
130
+ export declare function renderGoalSection(goal: string, repoRoot?: string): string;
131
+ /**
132
+ * OBS-1013 add.3: each carried id is printed ONCE, verbatim, inside a fenced block the reviewer can
133
+ * copy; the notes follow in the same order. A reviewer that retyped a 600-byte id from prose lost
134
+ * closure on a typo and that read as malformed — the block is what a closure list is copied from.
135
+ */
136
+ export declare function renderPriorMaterials(priorMaterials: readonly StructuredFinding[]): string;
137
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[], carriedAuthors?: readonly string[]): Promise<GateResult>;
@@ -1,5 +1,5 @@
1
1
  import { existsSync, writeFileSync } from "node:fs";
2
- import { join } from "node:path";
2
+ import { dirname, join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
5
5
  import { filesGlob } from "../graph/files-glob.js";
@@ -10,6 +10,7 @@ import { structuredFindings } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { marginalCostRank } from "../route/router.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
+ import { resolveStateDir } from "./cache.js";
13
14
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
15
  import { classifyVerdictCause } from "./verdict-cause.js";
15
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
@@ -130,8 +131,14 @@ export function checkDiffCap(gate, measured, cap, prefix = "") {
130
131
  gate,
131
132
  pass: false,
132
133
  details: prefix + `diff exceeds verifiable cap (${measured} > ${cap}) — ${DIFF_CAP_REMEDY}`,
133
- // daemon/run-gates: park('human') immediately — the diff cannot shrink by retrying (OBS-48).
134
- meta: { park: "human" },
134
+ meta: {
135
+ park: "diff-cap",
136
+ parkKind: "diff-cap",
137
+ measuredBytes: measured,
138
+ permittedBytes: cap,
139
+ measured,
140
+ permitted: cap,
141
+ },
135
142
  };
136
143
  }
137
144
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
@@ -147,12 +154,19 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
147
154
  pass: false,
148
155
  details: prefix
149
156
  + `captured artifact diff exceeds verifiable capture cap (${measured.captureBytes} > ${captureCap}) — ${DIFF_CAP_REMEDY}`,
150
- meta: { park: "human" },
157
+ meta: {
158
+ park: "diff-cap",
159
+ parkKind: "diff-cap",
160
+ measuredBytes: measured.captureBytes,
161
+ permittedBytes: captureCap,
162
+ measured: measured.captureBytes,
163
+ permitted: captureCap,
164
+ },
151
165
  };
152
166
  }
153
167
  export function isDiffCapPark(result) {
154
168
  return result.pass === false
155
- && result.meta?.park === "human"
169
+ && result.meta?.parkKind === "diff-cap"
156
170
  && /diff exceeds verifiable (?:capture )?cap/i.test(result.details);
157
171
  }
158
172
  // ponytail: single policy hook for callers after runGates — skips the escalation ladder on diff-cap trips.
@@ -193,6 +207,21 @@ export function isReviewClosureInvalid(v, priorIds) {
193
207
  || new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
194
208
  || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
195
209
  }
210
+ /**
211
+ * OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
212
+ * it answered about the materials and missed the id (a retyped, truncated or paraphrased fingerprint).
213
+ * That is a no-verdict about the carried work (re-route), not a parse defect. A verdict that omits a
214
+ * list, carries a non-string or duplicates an id stays malformed: its shape, not its ids, is wrong.
215
+ */
216
+ export function isReviewClosureMismatch(v, priorIds) {
217
+ if (!v || !Array.isArray(v.resolved) || !Array.isArray(v.reraised))
218
+ return false;
219
+ const ids = [...v.resolved, ...v.reraised];
220
+ if (!ids.every((id) => typeof id === "string"))
221
+ return false;
222
+ const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
223
+ return ids.some((id) => matchClosureId(id, priors) === undefined);
224
+ }
196
225
  // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
197
226
  // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
198
227
  // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
@@ -247,7 +276,9 @@ export function pickReviewer(author, channels, exclude = [], // v1.1 failover: r
247
276
  prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
248
277
  floor, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
249
278
  history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
250
- onSeat, demoted = new Set()) {
279
+ onSeat, demoted = new Set(),
280
+ // OBS-1033: vendors that authored a carried commit inside the accumulated diff — excluded for the round.
281
+ excludeVendors = new Set()) {
251
282
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
252
283
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
253
284
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -267,6 +298,7 @@ onSeat, demoted = new Set()) {
267
298
  && modelProvider(c.model, c.vendor) !== authorProvider
268
299
  && modelId(c.model) !== modelId(author.model)
269
300
  && !exclude.includes(channelKey(c))
301
+ && !excludeVendors.has(c.vendor)
270
302
  && TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
271
303
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
272
304
  const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
@@ -289,13 +321,66 @@ export function renderDeclaredWriteScope(files) {
289
321
  The task DECLARED these write-scope patterns:
290
322
  ${files.map((path) => `- ${path}`).join("\n")}`;
291
323
  }
324
+ /**
325
+ * OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
326
+ * never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
327
+ * its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
328
+ */
329
+ export function carriedAuthorVendors(channels, carriedAuthors = []) {
330
+ const vendors = new Set();
331
+ for (const key of carriedAuthors) {
332
+ const adapter = key.split(":")[0];
333
+ const exact = channels.filter((c) => channelKey(c) === key);
334
+ for (const c of exact.length ? exact : channels.filter((c) => c.adapter === adapter))
335
+ vendors.add(c.vendor);
336
+ }
337
+ return vendors;
338
+ }
339
+ /**
340
+ * OBS-1020: the compiled goal is the contract. After `resume --graph-changed` the worktree's copy of
341
+ * the spec is the pre-change text on the integration branch, so a reviewer that reads it grades a
342
+ * superseded contract. The daemon's repository root is where specs and planning records are current.
343
+ */
344
+ export function renderGoalSection(goal, repoRoot) {
345
+ return `## Goal (authoritative — compiled from the sealed graph; the worktree's spec file may be stale after resume --graph-changed)
346
+ ${goal}
347
+ ${repoRoot ? `Specs and planning records are read in the daemon's repository root ${repoRoot} (its specs/ and .planning/), never this worktree's copies.` : ""}`;
348
+ }
349
+ /**
350
+ * The daemon's repository root: the parent of the state dir the run's artifacts live under. Named
351
+ * only when that state dir exists — a guessed one would send the reviewer to a path that holds nothing.
352
+ */
353
+ function daemonRepoRoot(worktree, artifactDir) {
354
+ try {
355
+ const stateDir = resolveStateDir(worktree, artifactDir);
356
+ return stateDir.endsWith("/.tickmarkr") && existsSync(stateDir) ? dirname(stateDir) : undefined;
357
+ }
358
+ catch {
359
+ return undefined;
360
+ }
361
+ }
362
+ /**
363
+ * OBS-1013 add.3: each carried id is printed ONCE, verbatim, inside a fenced block the reviewer can
364
+ * copy; the notes follow in the same order. A reviewer that retyped a 600-byte id from prose lost
365
+ * closure on a typo and that read as malformed — the block is what a closure list is copied from.
366
+ */
367
+ export function renderPriorMaterials(priorMaterials) {
368
+ return `## Prior materials this attempt must close
369
+ Copy each fingerprint below EXACTLY (they appear once, in this block) into resolved or reraised:
370
+ \`\`\`text
371
+ ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}`).join("\n")}
372
+ \`\`\`
373
+ ${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
374
+ }
292
375
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
293
376
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
294
377
  // direct tests) skips persistence and changes nothing else.
295
378
  artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
296
379
  // RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
297
380
  // never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
298
- priorReviewers = []) {
381
+ priorReviewers = [],
382
+ // OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
383
+ carriedAuthors = []) {
299
384
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
300
385
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
301
386
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -370,7 +455,7 @@ priorReviewers = []) {
370
455
  const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
371
456
  const floorMeta = { reviewerFloor, reviewerFloorCause };
372
457
  let rotationSeat;
373
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
458
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers, carriedAuthorVendors(channels, carriedAuthors));
374
459
  if (!reviewer) {
375
460
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
376
461
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
@@ -390,6 +475,7 @@ priorReviewers = []) {
390
475
  if (capFail)
391
476
  return capFail;
392
477
  const nonce = generateVerdictNonce();
478
+ const repoRoot = daemonRepoRoot(worktree, artifactDir);
393
479
  const prompt = `TICKMARKR-REVIEW
394
480
  You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
395
481
  Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
@@ -397,13 +483,13 @@ Look for correctness bugs, security issues, and acceptance-criteria gaps. Approv
397
483
  ${COMPLETION_FAKING_CHECKLIST}
398
484
 
399
485
  ## Task ${task.id}: ${task.title} (complexity ${task.complexity})
486
+ ${renderGoalSection(task.goal, repoRoot)}
400
487
  ## Acceptance criteria
401
488
  ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
402
489
 
403
490
  ${renderDeclaredWriteScope(task.files)}
404
491
 
405
- ${priorMaterials.length ? `## Prior materials this attempt must close
406
- ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}\n${finding.note}`).join("\n\n")}
492
+ ${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
407
493
 
408
494
  ` : ""}## Diff
409
495
  \`\`\`diff
@@ -483,17 +569,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
483
569
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
484
570
  const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
485
571
  const closureInvalid = isReviewClosureInvalid(v, priorIds);
572
+ const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
486
573
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
487
574
  if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
488
575
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
489
576
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
490
577
  const bytes = llm.seatAuthoredBytes ?? Buffer.byteLength(raw.trim(), "utf8");
491
- const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
492
- : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
493
- : classifyVerdictCause(raw, nonce, "approve", llm);
578
+ const cause = closureMismatch ? "closure-mismatch" : closureInvalid ? "malformed-verdict"
579
+ : llm.launchNeverStarted ? "launch-never-started"
580
+ : llm.silentAtBeat ? "silent"
581
+ : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
582
+ : classifyVerdictCause(raw, nonce, "approve", llm);
494
583
  const failure = cause === "malformed-verdict"
495
584
  ? "review output unparseable"
496
- : "review dispatch failed — no structurally valid nonce-bound response";
585
+ : cause === "closure-mismatch"
586
+ ? "review verdict closes no carried fingerprint — closure ids match none of the carried materials"
587
+ : "review dispatch failed — no structurally valid nonce-bound response";
497
588
  return {
498
589
  gate: "review",
499
590
  pass: false,
@@ -508,6 +599,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
508
599
  provider,
509
600
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
510
601
  cause,
602
+ ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: [...priorIds] } : {}),
511
603
  bytes, seatAuthoredBytes: bytes,
512
604
  ...(saved ? { rawPath: saved } : {}),
513
605
  ...(savedBrief ? { briefPath: savedBrief } : {}),