tickmarkr 2.5.5 → 2.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +3 -1
  2. package/dist/adapters/prompt.js +21 -1
  3. package/dist/cli/commands/approve.js +69 -8
  4. package/dist/cli/commands/doctor.d.ts +10 -0
  5. package/dist/cli/commands/fleet.js +4 -0
  6. package/dist/cli/commands/plan.js +20 -4
  7. package/dist/cli/commands/status.js +95 -34
  8. package/dist/cli/commands/verify.js +108 -85
  9. package/dist/config/config.d.ts +11 -0
  10. package/dist/config/config.js +21 -12
  11. package/dist/config/fleet-overlay.js +55 -17
  12. package/dist/drivers/index.js +2 -1
  13. package/dist/drivers/orca.d.ts +21 -1
  14. package/dist/drivers/orca.js +209 -27
  15. package/dist/gates/baseline.d.ts +21 -5
  16. package/dist/gates/baseline.js +67 -17
  17. package/dist/gates/cache.d.ts +14 -13
  18. package/dist/gates/cache.js +17 -5
  19. package/dist/gates/llm.d.ts +3 -0
  20. package/dist/gates/llm.js +11 -0
  21. package/dist/gates/review.d.ts +28 -3
  22. package/dist/gates/review.js +118 -15
  23. package/dist/gates/run-gates.d.ts +10 -2
  24. package/dist/gates/run-gates.js +362 -108
  25. package/dist/gates/test-manifest.d.ts +33 -1
  26. package/dist/gates/test-manifest.js +132 -40
  27. package/dist/gates/test-reporter.js +20 -7
  28. package/dist/graph/graph.d.ts +4 -0
  29. package/dist/graph/graph.js +50 -1
  30. package/dist/run/activity.d.ts +28 -0
  31. package/dist/run/activity.js +194 -0
  32. package/dist/run/consult.js +5 -4
  33. package/dist/run/daemon.d.ts +11 -0
  34. package/dist/run/daemon.js +663 -128
  35. package/dist/run/execution-budget.d.ts +25 -0
  36. package/dist/run/execution-budget.js +142 -0
  37. package/dist/run/git.d.ts +46 -1
  38. package/dist/run/git.js +149 -12
  39. package/dist/run/journal.d.ts +23 -5
  40. package/dist/run/journal.js +98 -25
  41. package/dist/run/lease.d.ts +44 -0
  42. package/dist/run/lease.js +226 -3
  43. package/dist/run/operator-page-summary.d.ts +56 -0
  44. package/dist/run/operator-page-summary.js +68 -0
  45. package/dist/run/operator-summary.d.ts +69 -0
  46. package/dist/run/operator-summary.js +77 -0
  47. package/dist/run/protocol.d.ts +71 -0
  48. package/dist/run/protocol.js +32 -0
  49. package/dist/run/recovery.d.ts +8 -0
  50. package/dist/run/recovery.js +25 -0
  51. package/dist/run/repair-selection.d.ts +12 -0
  52. package/dist/run/repair-selection.js +56 -0
  53. package/dist/run/stall.d.ts +6 -1
  54. package/dist/run/stall.js +60 -3
  55. package/dist/tui/cockpit/board.d.ts +9 -0
  56. package/dist/tui/cockpit/board.js +10 -0
  57. package/dist/tui/cockpit/derive.d.ts +35 -0
  58. package/dist/tui/cockpit/derive.js +152 -10
  59. package/dist/tui/cockpit/evidence-view.d.ts +2 -0
  60. package/dist/tui/cockpit/evidence-view.js +42 -12
  61. package/dist/tui/cockpit/run-cockpit.d.ts +8 -1
  62. package/dist/tui/cockpit/run-cockpit.js +95 -1
  63. package/dist/tui/cockpit/run-view.d.ts +37 -0
  64. package/dist/tui/cockpit/run-view.js +189 -2
  65. package/dist/tui/ink/fleet-app.d.ts +10 -2
  66. package/dist/tui/ink/fleet-app.js +33 -15
  67. package/package.json +2 -2
  68. package/skills/tickmarkr-overseer/SKILL.md +55 -3
  69. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +29 -2
@@ -1,6 +1,6 @@
1
1
  import type { Baseline } from "./baseline.js";
2
2
  import type { GateResult } from "./types.js";
3
- import { type RunCapacity } from "../run/git.js";
3
+ import { type RunCapacity, type VerificationProtocol } from "../run/git.js";
4
4
  export declare const DEFAULT_VERDICT_CACHE_BOUND = 128;
5
5
  export declare function setVerdictCacheBoundForTests(bound: number | undefined): void;
6
6
  export declare function resetVerdictCacheBoundForTests(): void;
@@ -16,15 +16,19 @@ export interface GateEnvironmentInput {
16
16
  capacity?: RunCapacity;
17
17
  selectedSet?: readonly string[];
18
18
  scope?: VerificationScope;
19
+ /** R41: the verification protocol + runner lifecycle policy; defaults to this process's. */
20
+ verification?: VerificationProtocol;
21
+ }
22
+ export interface EnvironmentParts {
23
+ nodeRuntime: string;
24
+ lockfile: string;
25
+ capacity: RunCapacity;
26
+ selectedSet?: readonly string[];
27
+ verification: VerificationProtocol;
19
28
  }
20
29
  export declare function environmentFingerprint(env: GateEnvironmentInput): {
21
30
  fingerprint: string;
22
- parts: {
23
- nodeRuntime: string;
24
- lockfile: string;
25
- capacity: RunCapacity;
26
- selectedSet?: readonly string[];
27
- };
31
+ parts: EnvironmentParts;
28
32
  };
29
33
  export type VerificationScope = "battery" | "tip" | "standalone";
30
34
  export interface VerificationIdentity {
@@ -38,12 +42,7 @@ export interface VerificationIdentity {
38
42
  command: string;
39
43
  baseline: string;
40
44
  environment: string;
41
- envParts?: {
42
- nodeRuntime: string;
43
- lockfile: string;
44
- capacity: RunCapacity;
45
- selectedSet?: readonly string[];
46
- };
45
+ envParts?: EnvironmentParts;
47
46
  }
48
47
  export declare function computeVerificationIdentity(params: {
49
48
  worktree: string;
@@ -56,6 +55,7 @@ export declare function computeVerificationIdentity(params: {
56
55
  tree?: string;
57
56
  lockfile?: string;
58
57
  nodeRuntime?: string;
58
+ verification?: VerificationProtocol;
59
59
  }): Promise<VerificationIdentity | undefined>;
60
60
  export declare function verificationIdentityKey(id: VerificationIdentity): string;
61
61
  export declare function formatReusedDetails(originalDetails: string, id: VerificationIdentity): string;
@@ -89,6 +89,7 @@ export declare class VerdictStore {
89
89
  readonly dir: string;
90
90
  constructor(dir: string);
91
91
  private initSequenceFromDisk;
92
+ private static unknownPolicy;
92
93
  get(id?: VerificationIdentity): CachedVerdict | undefined;
93
94
  set(id: VerificationIdentity | undefined, verdict: GateResult | CachedVerdict): boolean;
94
95
  size(): number;
@@ -3,7 +3,7 @@ import { execSync } from "node:child_process";
3
3
  import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, rmSync, unlinkSync, writeFileSync } from "node:fs";
4
4
  import { dirname, join, resolve } from "node:path";
5
5
  import { tmpdir } from "node:os";
6
- import { describeCapacity, resolvedCapacity, shGit } from "../run/git.js";
6
+ import { describeCapacity, resolvedCapacity, shGit, verificationProtocol } from "../run/git.js";
7
7
  import { shq } from "../adapters/types.js";
8
8
  export const DEFAULT_VERDICT_CACHE_BOUND = 128;
9
9
  let testCacheBound;
@@ -83,17 +83,23 @@ export function environmentFingerprint(env) {
83
83
  const cap = env.capacity ?? resolvedCapacity();
84
84
  const capacity = { forkCap: cap.forkCap, cores: cap.cores };
85
85
  const selectedSet = env.selectedSet ? [...env.selectedSet].sort() : undefined;
86
+ const verification = env.verification ?? verificationProtocol(process.env, env.worktree ?? process.cwd());
87
+ // R41: the protocol and the EFFECTIVE lifecycle are IN the hashed payload, so every entry written
88
+ // before this stamp — green or red — keys differently and is never answered; no store surgery is
89
+ // needed. `source` is provenance (kept in parts, printed on the row) and never enters the key: an
90
+ // explicit `false` and an npmrc `false` are the same policy for the child that ran.
86
91
  const payload = canonicalJson({
87
92
  nodeRuntime,
88
93
  lockfile,
89
94
  capacity,
90
95
  selectedSet: selectedSet ?? null,
91
96
  scope: env.scope ?? "battery",
97
+ verification: { protocol: verification.protocol, lifecycle: verification.lifecycle },
92
98
  });
93
99
  const fingerprint = createHash("sha256").update(payload).digest("hex").slice(0, 16);
94
100
  return {
95
101
  fingerprint,
96
- parts: { nodeRuntime, lockfile, capacity, selectedSet },
102
+ parts: { nodeRuntime, lockfile, capacity, selectedSet, verification },
97
103
  };
98
104
  }
99
105
  export async function computeVerificationIdentity(params) {
@@ -108,6 +114,7 @@ export async function computeVerificationIdentity(params) {
108
114
  lockfile: params.lockfile,
109
115
  nodeRuntime: params.nodeRuntime,
110
116
  scope: params.scope,
117
+ verification: params.verification,
111
118
  });
112
119
  return {
113
120
  gate: params.gate,
@@ -130,7 +137,7 @@ export function verificationIdentityKey(id) {
130
137
  export function formatReusedDetails(originalDetails, id) {
131
138
  const unadorned = originalDetails.replace(/^reused verdict \(identity: [^)]+\):\s*/, "");
132
139
  const envDesc = id.envParts
133
- ? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}]`
140
+ ? ` [node=${id.envParts.nodeRuntime}, lockfile=${id.envParts.lockfile}, capacity=${describeCapacity(id.envParts.capacity)}${id.envParts.selectedSet ? `, selected=${id.envParts.selectedSet.join(",")}` : ""}, protocol=${id.envParts.verification.protocol}, lifecycle=${id.envParts.verification.lifecycle} (${id.envParts.verification.source})]`
134
141
  : "";
135
142
  const gate = id.gate ?? "gate";
136
143
  const prefix = `reused ${id.scope === "tip" ? "tip " : ""}verdict (identity: gate=${gate} tree=${id.tree}${id.worktree ? ` worktree=${id.worktree}` : ""} command=${id.command} baseline=${id.baseline} env=${id.environment}${envDesc})`;
@@ -258,8 +265,13 @@ export class VerdictStore {
258
265
  // ignore
259
266
  }
260
267
  }
268
+ // R41: an identity whose lifecycle policy could not be measured is never answered and never
269
+ // stored — an unknown policy is not comparable to anything, so the battery runs the command.
270
+ static unknownPolicy(id) {
271
+ return id.envParts?.verification?.lifecycle === "unknown";
272
+ }
261
273
  get(id) {
262
- if (!id || !id.tree)
274
+ if (!id || !id.tree || VerdictStore.unknownPolicy(id))
263
275
  return undefined;
264
276
  const key = verificationIdentityKey(id);
265
277
  const p = join(this.dir, `verdict-${key}.json`);
@@ -276,7 +288,7 @@ export class VerdictStore {
276
288
  return undefined;
277
289
  }
278
290
  set(id, verdict) {
279
- if (!id || !id.tree)
291
+ if (!id || !id.tree || VerdictStore.unknownPolicy(id))
280
292
  return false;
281
293
  if (isInfraResult(verdict))
282
294
  return false;
@@ -62,11 +62,14 @@ export interface LlmRunResult {
62
62
  exitCode?: number;
63
63
  timedOut: boolean;
64
64
  launchNeverStarted?: boolean;
65
+ /** OBS-1039: seat-authored bytes stayed under REVIEW_SILENT_BYTE_FLOOR at the first liveness beat. */
66
+ silentAtBeat?: boolean;
65
67
  seatAuthoredBytes?: number;
66
68
  }
67
69
  export declare const PROMPT_GLYPHS: readonly ["➜", "❯", "$", "%", ">>", ">"];
68
70
  export declare function reviewSeatOutput(raw: string, nonce: string, adapterBannerRows?: readonly string[]): string;
69
71
  export declare const REVIEW_FIRST_LIVENESS_MS = 30000;
72
+ export declare const REVIEW_SILENT_BYTE_FLOOR = 64;
70
73
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
71
74
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
72
75
  export declare function dewrapPaneVerdict(out: string, nonce: string): string;
package/dist/gates/llm.js CHANGED
@@ -300,6 +300,10 @@ export function reviewSeatOutput(raw, nonce, adapterBannerRows = []) {
300
300
  return trailer ? seat.slice(0, trailer.index) : seat;
301
301
  }
302
302
  export const REVIEW_FIRST_LIVENESS_MS = 30_000;
303
+ // OBS-1039: a seat that wrote ten bytes and went quiet escaped the zero-byte beat and sat to the
304
+ // ceiling. Below this many seat-authored bytes at the first beat the seat is `silent` — demoted and
305
+ // re-routed then, not at the ceiling. Pane path only; a headless runner buffers and keeps its ceiling.
306
+ export const REVIEW_SILENT_BYTE_FLOOR = 64;
303
307
  async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
304
308
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
305
309
  try {
@@ -356,6 +360,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
356
360
  let out;
357
361
  let timedOut = false;
358
362
  let launchNeverStarted = false;
363
+ let silentAtBeat = false;
359
364
  let seatAuthoredBytes = 0;
360
365
  const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
361
366
  if (!gatePrompt) {
@@ -412,6 +417,11 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
412
417
  forceClose = true;
413
418
  break;
414
419
  }
420
+ if (seatAuthoredBytes < REVIEW_SILENT_BYTE_FLOOR) {
421
+ silentAtBeat = true;
422
+ forceClose = true;
423
+ break;
424
+ }
415
425
  }
416
426
  // Producing reviews own their full ceiling; inactivity is not a review verdict.
417
427
  if (reviewing)
@@ -458,6 +468,7 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
458
468
  ...(Number.isFinite(exitCode) ? { exitCode } : {}),
459
469
  timedOut,
460
470
  launchNeverStarted,
471
+ silentAtBeat,
461
472
  seatAuthoredBytes,
462
473
  };
463
474
  }
@@ -64,6 +64,13 @@ export declare function matchClosureId(candidate: unknown, fingerprints: Iterabl
64
64
  * all route through matchClosureId.
65
65
  */
66
66
  export declare function isReviewClosureInvalid(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
67
+ /**
68
+ * OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
69
+ * it answered about the materials and missed the id (a retyped, truncated or paraphrased fingerprint).
70
+ * That is a no-verdict about the carried work (re-route), not a parse defect. A verdict that omits a
71
+ * list, carries a non-string or duplicates an id stays malformed: its shape, not its ids, is wrong.
72
+ */
73
+ export declare function isReviewClosureMismatch(v: Pick<ReviewVerdict, "resolved" | "reraised"> | null | undefined, priorIds: ReadonlySet<string> | readonly string[]): boolean;
67
74
  export type ReviewerFloorCause = "author-tier" | "task-floor" | "config" | "prior-reviewer";
68
75
  /**
69
76
  * RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
@@ -101,12 +108,30 @@ export declare function pickReviewer(author: Assignment, channels: BillingChanne
101
108
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
102
109
  floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
103
110
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
104
- onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>): BillingChannel | null;
105
- export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent";
111
+ onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>): BillingChannel | null;
112
+ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch";
106
113
  /**
107
114
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
108
115
  * remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
109
116
  * judgement rather than a guarantee made by this renderer.
110
117
  */
111
118
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
112
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[]): Promise<GateResult>;
119
+ /**
120
+ * OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
121
+ * never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
122
+ * its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
123
+ */
124
+ export declare function carriedAuthorVendors(channels: BillingChannel[], carriedAuthors?: readonly string[]): Set<string>;
125
+ /**
126
+ * OBS-1020: the compiled goal is the contract. After `resume --graph-changed` the worktree's copy of
127
+ * the spec is the pre-change text on the integration branch, so a reviewer that reads it grades a
128
+ * superseded contract. The daemon's repository root is where specs and planning records are current.
129
+ */
130
+ export declare function renderGoalSection(goal: string, repoRoot?: string): string;
131
+ /**
132
+ * OBS-1013 add.3: each carried id is printed ONCE, verbatim, inside a fenced block the reviewer can
133
+ * copy; the notes follow in the same order. A reviewer that retyped a 600-byte id from prose lost
134
+ * closure on a typo and that read as malformed — the block is what a closure list is copied from.
135
+ */
136
+ export declare function renderPriorMaterials(priorMaterials: readonly StructuredFinding[]): string;
137
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[], demotedReviewers?: ReadonlySet<string>, carriedFindings?: readonly StructuredFinding[], priorReviewers?: readonly PriorReviewer[], carriedAuthors?: readonly string[]): Promise<GateResult>;
@@ -1,5 +1,5 @@
1
1
  import { existsSync, writeFileSync } from "node:fs";
2
- import { join } from "node:path";
2
+ import { dirname, join } from "node:path";
3
3
  import { channelKey, shq } from "../adapters/types.js";
4
4
  import { criticalPathHits, DEFAULT_DIFF_CAP, DEFAULT_REVIEW_CRITICAL_PATHS, declaredReviewPolicy, isReviewLeafPath, raiseReviewPolicy, REVIEW_VERSION_MIRRORS, TIER_RANK, } from "../config/config.js";
5
5
  import { filesGlob } from "../graph/files-glob.js";
@@ -10,6 +10,7 @@ import { structuredFindings } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
11
  import { marginalCostRank } from "../route/router.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
+ import { resolveStateDir } from "./cache.js";
13
14
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, runLlmDetailed, verdictNonceLine } from "./llm.js";
14
15
  import { classifyVerdictCause } from "./verdict-cause.js";
15
16
  import { captureDiffCapFor, measureArtifactDiff, reviewableLogicDiff, } from "./artifact-manifest.js";
@@ -130,8 +131,14 @@ export function checkDiffCap(gate, measured, cap, prefix = "") {
130
131
  gate,
131
132
  pass: false,
132
133
  details: prefix + `diff exceeds verifiable cap (${measured} > ${cap}) — ${DIFF_CAP_REMEDY}`,
133
- // daemon/run-gates: park('human') immediately — the diff cannot shrink by retrying (OBS-48).
134
- meta: { park: "human" },
134
+ meta: {
135
+ park: "diff-cap",
136
+ parkKind: "diff-cap",
137
+ measuredBytes: measured,
138
+ permittedBytes: cap,
139
+ measured,
140
+ permitted: cap,
141
+ },
135
142
  };
136
143
  }
137
144
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
@@ -147,12 +154,19 @@ export function checkTaskDiffCaps(gate, measured, logicCap, prefix = "") {
147
154
  pass: false,
148
155
  details: prefix
149
156
  + `captured artifact diff exceeds verifiable capture cap (${measured.captureBytes} > ${captureCap}) — ${DIFF_CAP_REMEDY}`,
150
- meta: { park: "human" },
157
+ meta: {
158
+ park: "diff-cap",
159
+ parkKind: "diff-cap",
160
+ measuredBytes: measured.captureBytes,
161
+ permittedBytes: captureCap,
162
+ measured: measured.captureBytes,
163
+ permitted: captureCap,
164
+ },
151
165
  };
152
166
  }
153
167
  export function isDiffCapPark(result) {
154
168
  return result.pass === false
155
- && result.meta?.park === "human"
169
+ && result.meta?.parkKind === "diff-cap"
156
170
  && /diff exceeds verifiable (?:capture )?cap/i.test(result.details);
157
171
  }
158
172
  // ponytail: single policy hook for callers after runGates — skips the escalation ladder on diff-cap trips.
@@ -171,7 +185,10 @@ export { modelProvider };
171
185
  export function matchClosureId(candidate, target) {
172
186
  if (typeof candidate !== "string")
173
187
  return typeof target === "string" ? false : undefined;
174
- const normCandidate = candidate.replace(/\s+/g, "");
188
+ // OBS-1068: the brief's copy block printed each id as `Fingerprint: <id>` and told the seat to copy
189
+ // it EXACTLY — three faithful approvals in one run were discarded as closure-mismatch. A leading
190
+ // `Fingerprint:` label is the brief's own wording, never the reviewer's id; strip it before comparing.
191
+ const normCandidate = candidate.replace(/^\s*fingerprint\s*:\s*/i, "").replace(/\s+/g, "");
175
192
  if (typeof target === "string") {
176
193
  return normCandidate === target.replace(/\s+/g, "");
177
194
  }
@@ -193,6 +210,21 @@ export function isReviewClosureInvalid(v, priorIds) {
193
210
  || new Set(allCandidateIds.map((id) => matchClosureId(id, priors) ?? id)).size !== allCandidateIds.length
194
211
  || [...priors].some((id) => !allCandidateIds.some((candidate) => matchClosureId(candidate, id))));
195
212
  }
213
+ /**
214
+ * OBS-1013 add.3: the reviewer ECHOED closure ids and at least one matches no carried fingerprint —
215
+ * it answered about the materials and missed the id (a retyped, truncated or paraphrased fingerprint).
216
+ * That is a no-verdict about the carried work (re-route), not a parse defect. A verdict that omits a
217
+ * list, carries a non-string or duplicates an id stays malformed: its shape, not its ids, is wrong.
218
+ */
219
+ export function isReviewClosureMismatch(v, priorIds) {
220
+ if (!v || !Array.isArray(v.resolved) || !Array.isArray(v.reraised))
221
+ return false;
222
+ const ids = [...v.resolved, ...v.reraised];
223
+ if (!ids.every((id) => typeof id === "string"))
224
+ return false;
225
+ const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
226
+ return ids.some((id) => matchClosureId(id, priors) === undefined);
227
+ }
196
228
  // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
197
229
  // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
198
230
  // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
@@ -247,7 +279,9 @@ export function pickReviewer(author, channels, exclude = [], // v1.1 failover: r
247
279
  prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
248
280
  floor, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
249
281
  history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
250
- onSeat, demoted = new Set()) {
282
+ onSeat, demoted = new Set(),
283
+ // OBS-1033: vendors that authored a carried commit inside the accumulated diff — excluded for the round.
284
+ excludeVendors = new Set()) {
251
285
  // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
252
286
  // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
253
287
  // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
@@ -267,6 +301,7 @@ onSeat, demoted = new Set()) {
267
301
  && modelProvider(c.model, c.vendor) !== authorProvider
268
302
  && modelId(c.model) !== modelId(author.model)
269
303
  && !exclude.includes(channelKey(c))
304
+ && !excludeVendors.has(c.vendor)
270
305
  && TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
271
306
  .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
272
307
  const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
@@ -289,13 +324,66 @@ export function renderDeclaredWriteScope(files) {
289
324
  The task DECLARED these write-scope patterns:
290
325
  ${files.map((path) => `- ${path}`).join("\n")}`;
291
326
  }
327
+ /**
328
+ * OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
329
+ * never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
330
+ * its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
331
+ */
332
+ export function carriedAuthorVendors(channels, carriedAuthors = []) {
333
+ const vendors = new Set();
334
+ for (const key of carriedAuthors) {
335
+ const adapter = key.split(":")[0];
336
+ const exact = channels.filter((c) => channelKey(c) === key);
337
+ for (const c of exact.length ? exact : channels.filter((c) => c.adapter === adapter))
338
+ vendors.add(c.vendor);
339
+ }
340
+ return vendors;
341
+ }
342
+ /**
343
+ * OBS-1020: the compiled goal is the contract. After `resume --graph-changed` the worktree's copy of
344
+ * the spec is the pre-change text on the integration branch, so a reviewer that reads it grades a
345
+ * superseded contract. The daemon's repository root is where specs and planning records are current.
346
+ */
347
+ export function renderGoalSection(goal, repoRoot) {
348
+ return `## Goal (authoritative — compiled from the sealed graph; the worktree's spec file may be stale after resume --graph-changed)
349
+ ${goal}
350
+ ${repoRoot ? `Specs and planning records are read in the daemon's repository root ${repoRoot} (its specs/ and .planning/), never this worktree's copies.` : ""}`;
351
+ }
352
+ /**
353
+ * The daemon's repository root: the parent of the state dir the run's artifacts live under. Named
354
+ * only when that state dir exists — a guessed one would send the reviewer to a path that holds nothing.
355
+ */
356
+ function daemonRepoRoot(worktree, artifactDir) {
357
+ try {
358
+ const stateDir = resolveStateDir(worktree, artifactDir);
359
+ return stateDir.endsWith("/.tickmarkr") && existsSync(stateDir) ? dirname(stateDir) : undefined;
360
+ }
361
+ catch {
362
+ return undefined;
363
+ }
364
+ }
365
+ /**
366
+ * OBS-1013 add.3: each carried id is printed ONCE, verbatim, inside a fenced block the reviewer can
367
+ * copy; the notes follow in the same order. A reviewer that retyped a 600-byte id from prose lost
368
+ * closure on a typo and that read as malformed — the block is what a closure list is copied from.
369
+ */
370
+ export function renderPriorMaterials(priorMaterials) {
371
+ return `## Prior materials this attempt must close
372
+ Copy each fingerprint below EXACTLY (they appear once, in this block) into resolved or reraised:
373
+ \`\`\`text
374
+ ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}`).join("\n")}
375
+ \`\`\`
376
+ ${priorMaterials.map((finding, i) => `${i + 1}. ${finding.note}`).join("\n\n")}`;
377
+ }
292
378
  export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
293
379
  // OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
294
380
  // direct tests) skips persistence and changes nothing else.
295
381
  artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
296
382
  // RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
297
383
  // never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
298
- priorReviewers = []) {
384
+ priorReviewers = [],
385
+ // OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
386
+ carriedAuthors = []) {
299
387
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
300
388
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
301
389
  // retired — the branch that returned a green skip on a complexity comparison is gone, and with it the
@@ -370,7 +458,7 @@ priorReviewers = []) {
370
458
  const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
371
459
  const floorMeta = { reviewerFloor, reviewerFloorCause };
372
460
  let rotationSeat;
373
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers);
461
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers, carriedAuthorVendors(channels, carriedAuthors));
374
462
  if (!reviewer) {
375
463
  // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
376
464
  // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
@@ -390,6 +478,12 @@ priorReviewers = []) {
390
478
  if (capFail)
391
479
  return capFail;
392
480
  const nonce = generateVerdictNonce();
481
+ const repoRoot = daemonRepoRoot(worktree, artifactDir);
482
+ // OBS-880 add.1: guidance only. Do not expand scope globs into permission to run suites.
483
+ const ownTestFiles = [...new Set(task.files.filter((file) => !/[*?[\]{}()!]/.test(file) && /(?:^|\/)[^/]*\.(?:test|spec)\.[cm]?[jt]sx?$/.test(file)))];
484
+ const suiteBudget = ownTestFiles.length
485
+ ? `You may run at most the task's own test files explicitly named in files[]; these are the only suites you may run: ${ownTestFiles.map((file) => `\`${file}\``).join(", ")}.`
486
+ : "No suite may be run: files[] names no explicit test file owned by this task.";
393
487
  const prompt = `TICKMARKR-REVIEW
394
488
  You are a skeptical cross-vendor code reviewer. Another agent (vendor: ${author.adapter}) authored this diff.
395
489
  Look for correctness bugs, security issues, and acceptance-criteria gaps. Approve only if you would merge it.
@@ -397,13 +491,16 @@ Look for correctness bugs, security issues, and acceptance-criteria gaps. Approv
397
491
  ${COMPLETION_FAKING_CHECKLIST}
398
492
 
399
493
  ## Task ${task.id}: ${task.title} (complexity ${task.complexity})
494
+ ${renderGoalSection(task.goal, repoRoot)}
400
495
  ## Acceptance criteria
401
496
  ${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
402
497
 
403
498
  ${renderDeclaredWriteScope(task.files)}
404
499
 
405
- ${priorMaterials.length ? `## Prior materials this attempt must close
406
- ${priorMaterials.map((finding) => `Fingerprint: ${finding.fingerprint}\n${finding.note}`).join("\n\n")}
500
+ ## Reviewer suite budget
501
+ ${suiteBudget} Never run the whole suite (including an unfiltered npm test or vitest run). The gate suite owns the runner lease; a parallel full suite starves the gate.
502
+
503
+ ${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
407
504
 
408
505
  ` : ""}## Diff
409
506
  \`\`\`diff
@@ -483,17 +580,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
483
580
  const findings = v && Array.isArray(v.findings) ? v.findings : null;
484
581
  const priorIds = new Set(priorMaterials.map((finding) => finding.fingerprint));
485
582
  const closureInvalid = isReviewClosureInvalid(v, priorIds);
583
+ const closureMismatch = closureInvalid && isReviewClosureMismatch(v, priorIds);
486
584
  // findings decides the verdict on its own; the legacy path still needs approve + issues to parse.
487
585
  if (!v || closureInvalid || (findings === null && (typeof v.approve !== "boolean" || !Array.isArray(v.issues)))) {
488
586
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
489
587
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
490
588
  const bytes = llm.seatAuthoredBytes ?? Buffer.byteLength(raw.trim(), "utf8");
491
- const cause = closureInvalid ? "malformed-verdict" : llm.launchNeverStarted ? "launch-never-started"
492
- : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
493
- : classifyVerdictCause(raw, nonce, "approve", llm);
589
+ const cause = closureMismatch ? "closure-mismatch" : closureInvalid ? "malformed-verdict"
590
+ : llm.launchNeverStarted ? "launch-never-started"
591
+ : llm.silentAtBeat ? "silent"
592
+ : llm.timedOut ? (bytes > 0 ? "truncated" : "silent")
593
+ : classifyVerdictCause(raw, nonce, "approve", llm);
494
594
  const failure = cause === "malformed-verdict"
495
595
  ? "review output unparseable"
496
- : "review dispatch failed — no structurally valid nonce-bound response";
596
+ : cause === "closure-mismatch"
597
+ ? "review verdict closes no carried fingerprint — closure ids match none of the carried materials"
598
+ : "review dispatch failed — no structurally valid nonce-bound response";
497
599
  return {
498
600
  gate: "review",
499
601
  pass: false,
@@ -508,6 +610,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
508
610
  provider,
509
611
  ...(cause === "malformed-verdict" ? { unparseable: true } : { noVerdict: true, classification: "infra", infra: true }),
510
612
  cause,
613
+ ...(closureMismatch ? { resolved: v?.resolved, reraised: v?.reraised, carriedFingerprints: [...priorIds] } : {}),
511
614
  bytes, seatAuthoredBytes: bytes,
512
615
  ...(saved ? { rawPath: saved } : {}),
513
616
  ...(savedBrief ? { briefPath: savedBrief } : {}),
@@ -1,3 +1,4 @@
1
+ import type { CommandReceiptAttribution } from "../run/protocol.js";
1
2
  import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
2
3
  import { type TickmarkrConfig } from "../config/config.js";
3
4
  import { type GateName, type Task } from "../graph/schema.js";
@@ -5,6 +6,7 @@ import { type Baseline } from "./baseline.js";
5
6
  import { type GateVia } from "./llm.js";
6
7
  import { type PriorReviewer } from "./review.js";
7
8
  import type { GateResult } from "./types.js";
9
+ import { type VerificationRetryCause } from "../run/recovery.js";
8
10
  import { type StructuredFinding } from "../run/journal.js";
9
11
  import { type VerificationScope } from "./cache.js";
10
12
  export type LoadProvider = () => number;
@@ -41,9 +43,11 @@ export type GateEvent = {
41
43
  gate: GateName;
42
44
  name: string;
43
45
  payload: Record<string, unknown>;
44
- result: GateResult;
46
+ result?: GateResult;
45
47
  };
46
48
  export interface GateContext {
49
+ buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
50
+ authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
47
51
  verificationScope?: VerificationScope;
48
52
  worktree: string;
49
53
  baseRef: string;
@@ -59,11 +63,15 @@ export interface GateContext {
59
63
  carriedFindings?: readonly StructuredFinding[];
60
64
  excludeReviewers?: string[];
61
65
  demotedReviewers?: Set<string>;
66
+ reviewNoVerdicts?: Map<string, string[]>;
67
+ recheck?: boolean;
68
+ carriedAuthors?: readonly string[];
62
69
  reviewHistory?: string[];
63
70
  priorReviewers?: PriorReviewer[];
64
71
  artifactDir?: string;
65
- pipeline?: "v185" | "legacy";
66
72
  selectTests?: boolean;
73
+ requiredRepairTests?: readonly string[];
74
+ selectionReason?: string;
67
75
  collateral?: ReadonlyArray<string>;
68
76
  onGate?: (e: GateEvent) => void | Promise<void>;
69
77
  stateDir?: string;