tickmarkr 2.5.9 → 2.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/dist/adapters/claude-code.js +19 -6
  2. package/dist/adapters/prompt.d.ts +2 -1
  3. package/dist/adapters/prompt.js +11 -1
  4. package/dist/adapters/types.d.ts +4 -0
  5. package/dist/adapters/types.js +10 -0
  6. package/dist/cli/commands/fleet.js +26 -1
  7. package/dist/cli/commands/init.js +1 -1
  8. package/dist/cli/commands/plan.js +7 -3
  9. package/dist/cli/commands/report.js +11 -1
  10. package/dist/cli/commands/status.js +13 -3
  11. package/dist/cli/commands/verify.js +2 -0
  12. package/dist/cli/help.d.ts +2 -0
  13. package/dist/cli/help.js +2 -0
  14. package/dist/compile/native.js +39 -4
  15. package/dist/drivers/orca.d.ts +1 -1
  16. package/dist/drivers/orca.js +27 -5
  17. package/dist/gates/baseline.d.ts +2 -0
  18. package/dist/gates/baseline.js +9 -1
  19. package/dist/gates/cache.d.ts +11 -1
  20. package/dist/gates/cache.js +35 -15
  21. package/dist/gates/llm.js +4 -1
  22. package/dist/gates/review.d.ts +2 -3
  23. package/dist/gates/review.js +28 -35
  24. package/dist/gates/run-gates.d.ts +4 -0
  25. package/dist/gates/run-gates.js +17 -4
  26. package/dist/gates/test-manifest.d.ts +17 -0
  27. package/dist/gates/test-manifest.js +109 -9
  28. package/dist/gates/test-reporter.js +4 -0
  29. package/dist/graph/graph.d.ts +7 -3
  30. package/dist/graph/graph.js +21 -5
  31. package/dist/graph/schema.d.ts +2 -0
  32. package/dist/graph/schema.js +2 -0
  33. package/dist/route/preference.d.ts +1 -1
  34. package/dist/route/preference.js +10 -39
  35. package/dist/route/role-pick.d.ts +16 -0
  36. package/dist/route/role-pick.js +15 -0
  37. package/dist/route/router.d.ts +14 -0
  38. package/dist/route/router.js +9 -1
  39. package/dist/run/consult.js +5 -9
  40. package/dist/run/daemon.d.ts +16 -1
  41. package/dist/run/daemon.js +554 -77
  42. package/dist/run/git.d.ts +46 -1
  43. package/dist/run/git.js +138 -6
  44. package/dist/run/host-health.d.ts +20 -0
  45. package/dist/run/host-health.js +64 -0
  46. package/dist/run/journal.d.ts +1 -1
  47. package/dist/run/journal.js +51 -6
  48. package/dist/run/merge.d.ts +1 -1
  49. package/dist/run/merge.js +30 -6
  50. package/dist/run/operator-state.d.ts +24 -2
  51. package/dist/run/operator-state.js +41 -5
  52. package/dist/run/stall.d.ts +38 -2
  53. package/dist/run/stall.js +276 -6
  54. package/dist/tui/cockpit/board.d.ts +1 -1
  55. package/dist/tui/cockpit/board.js +27 -19
  56. package/dist/tui/cockpit/derive.d.ts +2 -0
  57. package/dist/tui/cockpit/derive.js +4 -0
  58. package/dist/tui/cockpit/live-runtime.d.ts +4 -0
  59. package/dist/tui/cockpit/live-runtime.js +37 -3
  60. package/dist/tui/cockpit/live-store.d.ts +3 -0
  61. package/dist/tui/cockpit/live-store.js +31 -8
  62. package/dist/tui/cockpit/run-cockpit.js +2 -1
  63. package/dist/tui/cockpit/run-view.d.ts +2 -4
  64. package/dist/tui/cockpit/run-view.js +9 -8
  65. package/package.json +1 -1
  66. package/schema/rungraph.schema.json +7 -0
  67. package/skills/tickmarkr-overseer/SKILL.md +76 -14
@@ -1,9 +1,9 @@
1
1
  import { createHash } from "node:crypto";
2
2
  import { execSync } from "node:child_process";
3
- import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, rmSync, unlinkSync, writeFileSync } from "node:fs";
4
- import { dirname, join, resolve } from "node:path";
3
+ import { existsSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, realpathSync, renameSync, rmSync, unlinkSync, writeFileSync } from "node:fs";
4
+ import { dirname, join, relative, resolve } from "node:path";
5
5
  import { tmpdir } from "node:os";
6
- import { describeCapacity, resolvedCapacity, shGit, verificationProtocol } from "../run/git.js";
6
+ import { checkoutIncarnation, describeCapacity, inventoryDependencyLinks, resolvedCapacity, shGit, verificationProtocol } from "../run/git.js";
7
7
  import { shq } from "../adapters/types.js";
8
8
  export const DEFAULT_VERDICT_CACHE_BOUND = 128;
9
9
  let testCacheBound;
@@ -84,6 +84,13 @@ export function environmentFingerprint(env) {
84
84
  const capacity = { forkCap: cap.forkCap, cores: cap.cores };
85
85
  const selectedSet = env.selectedSet ? [...env.selectedSet].sort() : undefined;
86
86
  const verification = env.verification ?? verificationProtocol(process.env, env.worktree ?? process.cwd());
87
+ // Admitted links still affect resolution. Normalize against the classified root so relocating
88
+ // an otherwise identical checkout (including its dependency store) preserves the identity.
89
+ const resolution = env.worktree ? inventoryDependencyLinks(env.worktree).map(({ link, target, classification }) => ({
90
+ link,
91
+ classification,
92
+ target: classification === "outside" ? target : relative(realpathSync(classification === "worktree" ? env.worktree : join(env.worktree, "node_modules")), target),
93
+ })).sort((a, b) => a.link < b.link ? -1 : a.link > b.link ? 1 : 0) : [];
87
94
  // R41: the protocol and the EFFECTIVE lifecycle are IN the hashed payload, so every entry written
88
95
  // before this stamp — green or red — keys differently and is never answered; no store surgery is
89
96
  // needed. `source` is provenance (kept in parts, printed on the row) and never enters the key: an
@@ -92,6 +99,7 @@ export function environmentFingerprint(env) {
92
99
  nodeRuntime,
93
100
  lockfile,
94
101
  capacity,
102
+ resolution,
95
103
  selectedSet: selectedSet ?? null,
96
104
  scope: env.scope ?? "battery",
97
105
  verification: { protocol: verification.protocol, lifecycle: verification.lifecycle },
@@ -106,6 +114,9 @@ export async function computeVerificationIdentity(params) {
106
114
  const tree = params.tree ?? (await getWorktreeTree(params.worktree));
107
115
  if (!tree)
108
116
  return undefined;
117
+ const incarnation = params.gate === "build" ? checkoutIncarnation(params.worktree) : undefined;
118
+ if (params.gate === "build" && !incarnation)
119
+ return undefined;
109
120
  const baseline = baselineIdentity(params.baseline);
110
121
  const env = environmentFingerprint({
111
122
  worktree: params.worktree,
@@ -121,6 +132,7 @@ export async function computeVerificationIdentity(params) {
121
132
  scope: params.scope ?? "battery",
122
133
  worktree: realpathSync(params.worktree),
123
134
  tree,
135
+ ...(incarnation ? { checkoutIncarnation: incarnation } : {}),
124
136
  command: params.command,
125
137
  baseline,
126
138
  environment: env.fingerprint,
@@ -130,8 +142,10 @@ export async function computeVerificationIdentity(params) {
130
142
  export function verificationIdentityKey(id) {
131
143
  const cmdHash = createHash("sha256").update(id.command).digest("hex").slice(0, 16);
132
144
  const gate = id.gate ?? "gate";
133
- // Checkout location is diagnostic only. Scope separates each verifier's evidence policy.
134
- const workingTree = createHash("sha256").update(id.tree).digest("hex");
145
+ // Lint/test remain content-addressed. Builds also promise outputs in this physical checkout.
146
+ // Scope continues to separate each verifier's evidence policy.
147
+ const workingTree = createHash("sha256").update(id.tree)
148
+ .update(id.gate === "build" ? `\0checkout:${id.checkoutIncarnation ?? "unbound"}` : "").digest("hex");
135
149
  return `${id.scope ?? "battery"}-${gate}-${workingTree}-${cmdHash}-${id.baseline}-${id.environment}`;
136
150
  }
137
151
  export function formatReusedDetails(originalDetails, id) {
@@ -155,6 +169,9 @@ export function formatReusedRow(cached, id) {
155
169
  pass: cached.pass,
156
170
  details: cached.pass ? reusedDetails : cached.details,
157
171
  capacity: cached.capacity,
172
+ ...(cached.evidenceReceipt ? { evidenceReceipt: cached.evidenceReceipt } : {}),
173
+ ...(cached.evidenceReceipts ? { evidenceReceipts: cached.evidenceReceipts } : {}),
174
+ ...(cached.originRunRoot ? { originRunRoot: cached.originRunRoot } : {}),
158
175
  meta: {
159
176
  ...cached.meta,
160
177
  reused: true,
@@ -168,6 +185,7 @@ export function reusedIdentity(id) {
168
185
  gate: id.gate ?? "gate",
169
186
  scope: id.scope ?? "battery",
170
187
  tree: id.tree,
188
+ ...(id.checkoutIncarnation ? { checkoutIncarnation: id.checkoutIncarnation } : {}),
171
189
  ...(id.worktree ? { worktree: id.worktree } : {}),
172
190
  command: id.command,
173
191
  baseline: id.baseline,
@@ -306,6 +324,10 @@ export class VerdictStore {
306
324
  ...("exitCode" in verdict ? { exitCode: verdict.exitCode } : {}),
307
325
  ...(verdict.capacity ? { capacity: verdict.capacity } : {}),
308
326
  ...(verdict.meta ? { meta: verdict.meta } : {}),
327
+ ...(verdict.evidenceReceipt ? { evidenceReceipt: verdict.evidenceReceipt } : {}),
328
+ ...(verdict.evidenceReceipts ? { evidenceReceipts: verdict.evidenceReceipts } : {}),
329
+ ...(verdict.originRunRoot ? { originRunRoot: verdict.originRunRoot }
330
+ : typeof verdict.meta?.runDir === "string" ? { originRunRoot: resolve(verdict.meta.runDir) } : {}),
309
331
  },
310
332
  timestamp: Date.now(),
311
333
  sequence: ++globalStoreSequence,
@@ -314,17 +336,15 @@ export class VerdictStore {
314
336
  const tmpPath = join(this.dir, `verdict-${key}.${process.pid}.${Date.now()}.${Math.random().toString(16).slice(2)}.tmp`);
315
337
  writeFileSync(tmpPath, JSON.stringify(record, null, 2) + "\n");
316
338
  try {
317
- unlinkSync(finalPath);
318
- }
319
- catch {
320
- // ignore
321
- }
322
- writeFileSync(finalPath, readFileSync(tmpPath));
323
- try {
324
- unlinkSync(tmpPath);
339
+ renameSync(tmpPath, finalPath);
325
340
  }
326
- catch {
327
- // ignore
341
+ finally {
342
+ try {
343
+ unlinkSync(tmpPath);
344
+ }
345
+ catch {
346
+ // Already renamed, or best-effort cleanup after a failed replacement.
347
+ }
328
348
  }
329
349
  this.evictOldest(bound);
330
350
  return true;
package/dist/gates/llm.js CHANGED
@@ -403,7 +403,10 @@ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs
403
403
  // That seat was killed by its configured timeout, not an early launch reroute.
404
404
  if (now - startedAt >= timeoutMs)
405
405
  break;
406
- if (reviewing && !firstLivenessObserved && now - startedAt >= REVIEW_FIRST_LIVENESS_MS) {
406
+ // OBS-1177: the beat is armed for an INTERACTIVE (pane) driver only. On the subprocess driver
407
+ // a headless `claude -p` buffers every byte until it exits, so 0 seat bytes at 30 s is not a
408
+ // dead launch; that seat keeps its full ceiling, which still bounds it (fail-closed by timeout).
409
+ if (reviewing && via.driver.interactive && !firstLivenessObserved && now - startedAt >= REVIEW_FIRST_LIVENESS_MS) {
407
410
  firstLivenessObserved = true;
408
411
  // RS-2: the beat reads seat-authored bytes ALONE. CPU evidence never holds a preamble-only
409
412
  // capture open to the ceiling; a seat that has not written one byte of its own is re-routed.
@@ -110,7 +110,7 @@ export declare function pickReviewer(author: Assignment, channels: BillingChanne
110
110
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
111
111
  floor?: Tier, // task/config/prior floor from the caller; the author's own tier is ALWAYS applied here (RF-1)
112
112
  history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
113
- onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>): BillingChannel | null;
113
+ onSeat?: (seat: number, count: number) => void, demoted?: ReadonlySet<string>, excludeVendors?: ReadonlySet<string>, authors?: readonly string[]): BillingChannel | null;
114
114
  export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-started" | "truncated" | "silent" | "closure-mismatch";
115
115
  /**
116
116
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
@@ -120,8 +120,7 @@ export type ReviewUnparseableCause = VerdictUnparseableCause | "launch-never-sta
120
120
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
121
121
  /**
122
122
  * OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
123
- * never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
124
- * its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
123
+ * never handed its own earlier work to approve. Unresolved channels are refused by reviewGate.
125
124
  */
126
125
  export declare function carriedAuthorVendors(channels: BillingChannel[], carriedAuthors?: readonly string[]): Set<string>;
127
126
  /**
@@ -8,7 +8,7 @@ import { getAdapter } from "../adapters/registry.js";
8
8
  import { shOk } from "../run/git.js";
9
9
  import { carryReviewFindings, observedReviewFingerprints, reviewFingerprintMatches, structuredFindings, UNIDENTIFIED } from "../run/journal.js";
10
10
  import { redactSecrets } from "../run/redact.js";
11
- import { marginalCostRank } from "../route/router.js";
11
+ import { rankPreferredChannels, reviewPreferenceTieBreak } from "../route/role-pick.js";
12
12
  import { modelProvider } from "../route/preference.js";
13
13
  import { resolveStateDir } from "./cache.js";
14
14
  import { appendAnchoredReview, COMPLETION_FAKING_CHECKLIST, extractVerdictJson, generateVerdictNonce, parseAnchoredComments, runLlmDetailed, verdictNonceLine } from "./llm.js";
@@ -227,13 +227,6 @@ export function isReviewClosureMismatch(v, priorIds) {
227
227
  const priors = priorIds instanceof Set ? priorIds : new Set(priorIds);
228
228
  return ids.some((id) => matchClosureId(id, priors) === undefined);
229
229
  }
230
- // v1.53 T2: same entry grammar as routing.map.prefer (router.ts preferIndex — router is out of this
231
- // module's dependency direction for a private fn, so the 3 lines live here too): `adapter` matches
232
- // every channel of that adapter, `adapter:model` exactly one; unmatched channels sort after all entries.
233
- function reviewPreferIndex(c, prefer) {
234
- const i = prefer.findIndex((p) => p === c.adapter || p === channelKey(c));
235
- return i === -1 ? prefer.length : i;
236
- }
237
230
  /**
238
231
  * RF-1 (OBS-922 add.2/3): the tier a reviewer must meet is the maximum of the author's tier, the
239
232
  * task-declared floor, a configured `review.floor` tier and, on a second round or a retry, the prior
@@ -283,29 +276,28 @@ floor, // task/config/prior floor from the caller; the author's own tier is ALWA
283
276
  history = [], // run-scoped picks, oldest to newest; empty preserves the established ranking
284
277
  onSeat, demoted = new Set(),
285
278
  // OBS-1033: vendors that authored a carried commit inside the accumulated diff — excluded for the round.
286
- excludeVendors = new Set()) {
287
- // FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
288
- // The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
289
- // admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
290
- // fail-closed branch under review.required.
291
- const authorChannel = channels.find((c) => c.adapter === author.adapter && c.model === author.model);
292
- if (!authorChannel)
279
+ excludeVendors = new Set(), authors = [channelKey(author)]) {
280
+ // Resolve every actual author, never guess a vendor from an adapter id.
281
+ const authorChannels = authors.map((key) => channels.find((c) => channelKey(c) === key));
282
+ if (authorChannels.some((c) => !c))
293
283
  return null;
294
- const authorProvider = modelProvider(author.model, authorChannel.vendor);
295
284
  // RF-1: every caller inherits the author-tier floor — a reviewer is never seated below its author.
296
285
  const effectiveFloor = resolveReviewerFloor(author.tier, floor).floor;
297
- const ranked = channels
286
+ const eligible = channels
298
287
  // Three independent axes: different vendor, different resolved provider identity (OBS-946: on initial pick
299
288
  // as well as failover, so an aggregator channel stamped "mixed" never seats the author's own provider),
300
289
  // and different base-model identity (ADDED TO the vendor rule, never replacing it). The diversity
301
290
  // filter runs BEFORE preference ranking, so prefer cannot resurrect an excluded channel.
302
- .filter((c) => c.vendor !== authorChannel.vendor
303
- && modelProvider(c.model, c.vendor) !== authorProvider
304
- && modelId(c.model) !== modelId(author.model)
291
+ .filter((c) => authorChannels.every((a) => a && c.vendor !== a.vendor
292
+ && modelProvider(c.model, c.vendor) !== modelProvider(a.model, a.vendor)
293
+ && modelId(c.model) !== modelId(a.model))
305
294
  && !exclude.includes(channelKey(c))
306
295
  && !excludeVendors.has(c.vendor)
307
- && TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor])
308
- .sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b));
296
+ && TIER_RANK[c.tier] >= TIER_RANK[effectiveFloor]);
297
+ const ranked = rankPreferredChannels(eligible, prefer, {
298
+ includeUnpreferred: true,
299
+ tieBreak: reviewPreferenceTieBreak,
300
+ });
309
301
  const reviewer = [...ranked].sort((a, b) => Number(demoted.has(channelKey(a))) - Number(demoted.has(channelKey(b)))
310
302
  || history.lastIndexOf(channelKey(a)) - history.lastIndexOf(channelKey(b))
311
303
  || ranked.indexOf(a) - ranked.indexOf(b))[0] ?? null;
@@ -328,15 +320,12 @@ ${files.map((path) => `- ${path}`).join("\n")}`;
328
320
  }
329
321
  /**
330
322
  * OBS-1033: the vendors of the seats that authored commits inside the accumulated diff. A seat is
331
- * never handed its own earlier work to approve. A prior author not resolvable in the pool excludes
332
- * its adapter's vendors instead (fail closed: the seat is known, its vendor is whatever it bills as).
323
+ * never handed its own earlier work to approve. Unresolved channels are refused by reviewGate.
333
324
  */
334
325
  export function carriedAuthorVendors(channels, carriedAuthors = []) {
335
326
  const vendors = new Set();
336
327
  for (const key of carriedAuthors) {
337
- const adapter = key.split(":")[0];
338
- const exact = channels.filter((c) => channelKey(c) === key);
339
- for (const c of exact.length ? exact : channels.filter((c) => c.adapter === adapter))
328
+ for (const c of channels.filter((c) => channelKey(c) === key))
340
329
  vendors.add(c.vendor);
341
330
  }
342
331
  return vendors;
@@ -391,7 +380,7 @@ artifactDir, reviewHistory, demotedReviewers, carriedFindings = [],
391
380
  // RF-1: channel keys of THIS task's prior reviewers (earlier rounds, a flaked seat) — task-scoped,
392
381
  // never the run-wide rotation history nor excludeReviewers; the seat holds the highest of their tiers.
393
382
  priorReviewers = [],
394
- // OBS-1033: channel keys of the seats that authored the carried commits (the daemon's tried list).
383
+ // Closed list of actual subject authors; empty preserves legacy callers' single-author contract.
395
384
  carriedAuthors = [], operatorContext) {
396
385
  // R3 (OBS-186): participation is keyed on PATHS. The compiler's assignment comes from the DECLARED
397
386
  // files[]; the operator's floor may RAISE it to full and can never lower it. `complexityThreshold` is
@@ -466,15 +455,15 @@ carriedAuthors = [], operatorContext) {
466
455
  // review.floor is read from config — cfg.routing.floors governs workers and never moves review seats.
467
456
  const { floor: reviewerFloor, cause: reviewerFloorCause } = gateReviewerFloor(task, cfg, author, channels, priorReviewers);
468
457
  const floorMeta = { reviewerFloor, reviewerFloorCause };
458
+ const authors = carriedAuthors.length ? carriedAuthors : [channelKey(author)];
459
+ const authorVendors = carriedAuthorVendors(channels, authors);
460
+ const unresolvedAuthors = authors.filter((key) => !channels.some((c) => channelKey(c) === key));
469
461
  let rotationSeat;
470
- const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers, carriedAuthorVendors(channels, carriedAuthors));
462
+ const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor, reviewHistory, reviewHistory ? (seat) => { rotationSeat = seat; } : undefined, demotedReviewers, authorVendors, authors);
471
463
  if (!reviewer) {
472
- // meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
473
- // the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
474
- const reason = `no cross-vendor reviewer available at or above ${reviewerFloor} floor (${reviewerFloorCause}; diversity rule)`;
475
- return cfg.review.required || priorMaterials.length > 0
476
- ? { gate: "review", pass: false, details: `unreadable — ${reason}; ${priorMaterials.length ? "carried materials require a review verdict" : "set review.required:false to waive"}`, meta: { noEligibleReviewer: true, unreadable: true, ...floorMeta } }
477
- : { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...floorMeta } };
464
+ const reason = `no cross-vendor reviewer available at or above ${reviewerFloor} floor (${reviewerFloorCause}; diversity rule); author vendors: ${[...authorVendors].sort().join(", ") || "unknown"}${unresolvedAuthors.length ? `; unresolved author channels: ${unresolvedAuthors.join(", ")}` : ""}`;
465
+ return { gate: "review", pass: false, details: `unreadable — ${reason}; operator approve --waive required to proceed`,
466
+ meta: { noEligibleReviewer: true, unreadable: true, findings: [], authorVendors: [...authorVendors].sort(), unresolvedAuthors, ...floorMeta } };
478
467
  }
479
468
  reviewHistory?.push(channelKey(reviewer));
480
469
  const rotationMeta = rotationSeat === undefined ? {} : { rotationSeat };
@@ -515,6 +504,10 @@ ${priorMaterials.length ? `${renderPriorMaterials(priorMaterials)}
515
504
  Context only: this never substitutes for an acceptance criterion or closes a prior material.
516
505
  ${operatorContext.trim()}
517
506
 
507
+ ` : ""}${task.outOfScope?.length ? `## Out of scope
508
+ The task declares these items out of scope. A finding inside these declared bounds is not material and must not block approval.
509
+ ${task.outOfScope.map((item) => `- ${item}`).join("\n")}
510
+
518
511
  ` : ""}## Diff
519
512
  \`\`\`diff
520
513
  ${diff}
@@ -7,6 +7,7 @@ import { type GateVia } from "./llm.js";
7
7
  import { type PriorReviewer } from "./review.js";
8
8
  import type { GateResult } from "./types.js";
9
9
  import { type VerificationRetryCause } from "../run/recovery.js";
10
+ import { type PreserveProducer } from "../run/git.js";
10
11
  import { type StructuredFinding } from "../run/journal.js";
11
12
  import { type VerificationScope } from "./cache.js";
12
13
  export type LoadProvider = () => number;
@@ -71,6 +72,9 @@ export interface GateContext {
71
72
  /** Explicit worker funding requires fresh red measurements, never a gate waiver. */
72
73
  cachedRedBypass?: "operator-rerun";
73
74
  carriedAuthors?: readonly string[];
75
+ /** The attempt whose worker last wrote the gated checkout; a dirty-tree refusal stamps it on the
76
+ * preserve commit and its row. Absent (standalone verify, gate-only restores) preserves as "unknown". */
77
+ producer?: PreserveProducer;
74
78
  reviewHistory?: string[];
75
79
  priorReviewers?: PriorReviewer[];
76
80
  artifactDir?: string;
@@ -17,7 +17,7 @@ import { scopeGate } from "./scope.js";
17
17
  import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
18
18
  import { executionSignal } from "../run/execution-budget.js";
19
19
  import { failureDisposition } from "../run/recovery.js";
20
- import { preserveWorktree, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
20
+ import { dependencyLinkRefusal, preserveWorktree, producerFields, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
21
21
  import { withJudgeInvocationEvidence } from "../run/journal.js";
22
22
  import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
23
23
  const productionLoadProvider = () => loadavg()[0] ?? 0;
@@ -223,6 +223,7 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
223
223
  overallCeilingMs: effectiveCeilingMs(entry),
224
224
  artifactDir,
225
225
  evidence: retry.evidence,
226
+ retryBaseCommand: retry.retryBaseCommand, // OBS-1166: the un-narrowed command for a selected screen's stranded retry
226
227
  });
227
228
  const reportPath = outcome.reportPath;
228
229
  const evidence = { evidenceReceipt: outcome.evidenceReceipt, evidenceReceipts: outcome.evidenceReceipts };
@@ -304,6 +305,16 @@ export async function runGates(task, ctx) {
304
305
  };
305
306
  let selectionDecision;
306
307
  let commits = [];
308
+ // Check before cache identity, npm policy probes, or any gate command.
309
+ const dependencyRefusal = dependencyLinkRefusal(ctx.worktree);
310
+ if (dependencyRefusal) {
311
+ const gate = GATE_NAMES.find(g => task.gates.includes(g)) ?? "build";
312
+ const result = { gate, pass: false, details: dependencyRefusal,
313
+ meta: { infra: true, classification: "infra", retryable: false, kind: "workspace-dependency" } };
314
+ await noBuild("refused", dependencyRefusal);
315
+ await ctx.onGate?.({ phase: "end", gate, result });
316
+ return { results: [result], commits: [] };
317
+ }
307
318
  const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
308
319
  const verdictStore = getVerdictStore(stateDir);
309
320
  // R41: the verification protocol and the EFFECTIVE npm lifecycle policy measured for THIS
@@ -472,7 +483,7 @@ export async function runGates(task, ctx) {
472
483
  let preservedRef;
473
484
  let preservationError;
474
485
  try {
475
- preservedRef = await preserveWorktree(ctx.worktree);
486
+ preservedRef = await preserveWorktree(ctx.worktree, ctx.producer);
476
487
  }
477
488
  catch (error) {
478
489
  // Never masks the refusal, but never pretends a snapshot exists either — surfaced below.
@@ -527,6 +538,7 @@ export async function runGates(task, ctx) {
527
538
  dirtyWorktree: true,
528
539
  ref: preservedRef,
529
540
  preservedRef,
541
+ ...producerFields(ctx.producer),
530
542
  paths: dirtyPaths,
531
543
  files: dirtyPaths,
532
544
  path: primaryFile,
@@ -573,7 +585,7 @@ export async function runGates(task, ctx) {
573
585
  let preservedRef;
574
586
  let preservationError;
575
587
  try {
576
- preservedRef = await preserveWorktree(ctx.worktree);
588
+ preservedRef = await preserveWorktree(ctx.worktree, ctx.producer);
577
589
  }
578
590
  catch (error) {
579
591
  preservationError = error instanceof Error ? error.message : String(error);
@@ -604,6 +616,7 @@ export async function runGates(task, ctx) {
604
616
  dirtyAtRoundEnd: true,
605
617
  ref: preservedRef,
606
618
  preservedRef,
619
+ ...producerFields(ctx.producer),
607
620
  paths: dirtyPaths,
608
621
  files: dirtyPaths,
609
622
  path: primaryFile,
@@ -667,7 +680,7 @@ export async function runGates(task, ctx) {
667
680
  // other scripted test command keeps today's exit-code contract byte-identically.
668
681
  const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
669
682
  r = useManifest
670
- ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence }))
683
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence, ...(selected ? { retryBaseCommand: ctx.commands.test } : {}) }))
671
684
  : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), evidence, ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
672
685
  }
673
686
  finally {
@@ -1,5 +1,7 @@
1
1
  import { type GateEvidenceOptions, type BaselineFileDuration } from "./baseline.js";
2
2
  import type { GateEvidenceReceipt } from "../run/protocol.js";
3
+ export declare const VITEST_CACHE_ENV = "TICKMARKR_VITEST_CACHE_DIR";
4
+ export declare function worktreeVitestCache(cwd: string, inherited?: string): string;
3
5
  export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
4
6
  /** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
5
7
  * --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
@@ -37,6 +39,11 @@ export interface TestReport {
37
39
  requested: string[];
38
40
  started: Record<string, number>;
39
41
  completed: Record<string, TestReportCompletion>;
42
+ /** Resolved scheduling observed at run start; absent/partial means unknown, never inferred. */
43
+ scheduling?: Record<string, {
44
+ pool: string;
45
+ singleFork: boolean;
46
+ }>;
40
47
  /** Files the reporter observed complete MORE than once — `completed`'s object keys cannot show
41
48
  * this themselves (a second write silently overwrites the first), so the reporter records the
42
49
  * evidence separately before it is lost. */
@@ -139,6 +146,13 @@ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
139
146
  * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
140
147
  * run here; the capture already ran it once. */
141
148
  export declare function manifestFileCount(cmd: string, cwd: string): Promise<number | null>;
149
+ /** The stranded single-fork retry: positional filters are substring matches that OR with any filter
150
+ * the command already carries, and under a `projects` config the CLI `--exclude` never subtracts such a
151
+ * selection (OBS-1166: a selected screen's retry rediscovered the whole selection and refused). So the
152
+ * retry is built from the UN-narrowed base command, its own `--` rule, the stranded files as the only
153
+ * positional filters, and an `--exclude` of every completed file; the caller then requires discovery
154
+ * to prove the exact retry set before launch. */
155
+ export declare function singleForkRetryCommand(base: string, cwd: string, stranded: readonly string[], completed: readonly string[]): string;
142
156
  /** One configured runner execution, and its own collection under the same arguments and environment.
143
157
  * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
144
158
  export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
@@ -147,4 +161,7 @@ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
147
161
  overallCeilingMs?: number;
148
162
  artifactDir?: string;
149
163
  evidence?: GateEvidenceOptions;
164
+ /** OBS-1166: the configured command BEFORE a selected screen narrowed it, so a stranded retry's
165
+ * only positional filters are the stranded files. Absent (a full suite), `cmd` is that command. */
166
+ retryBaseCommand?: string;
150
167
  }): Promise<ManifestGateOutcome>;
@@ -1,11 +1,20 @@
1
1
  import { createHash, randomBytes } from "node:crypto";
2
- import { existsSync, mkdtempSync, readFileSync, realpathSync, writeFileSync } from "node:fs";
2
+ import { existsSync, mkdtempSync, readFileSync, realpathSync, rmSync, writeFileSync } from "node:fs";
3
3
  import { tmpdir } from "node:os";
4
- import { isAbsolute, join, relative, sep } from "node:path";
4
+ import { isAbsolute, join, relative, resolve, sep } from "node:path";
5
5
  import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
6
6
  import { shq } from "../adapters/types.js";
7
7
  import { beginGateEvidence, redactGateOutput } from "./baseline.js";
8
8
  import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
9
+ // Outside the shared dependency symlink and ignored by the shipped repository.
10
+ export const VITEST_CACHE_ENV = "TICKMARKR_VITEST_CACHE_DIR";
11
+ export function worktreeVitestCache(cwd, inherited) {
12
+ const local = resolve(cwd, ".vitest-cache");
13
+ const candidate = inherited ? resolve(cwd, inherited) : local;
14
+ const within = relative(local, candidate);
15
+ return within === "" || (!isAbsolute(within) && within !== ".." && !within.startsWith(`..${sep}`))
16
+ ? candidate : local;
17
+ }
9
18
  /**
10
19
  * VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
11
20
  * count. `fileCountDeficit` (baseline.ts) reads a summary LINE — a selected screen's smaller count
@@ -387,6 +396,7 @@ export function runManifestedTest(cmd, cwd, opts) {
387
396
  * verdict measured with `pretest` hooks is never compared to one without. */
388
397
  function manifestEnvironment(cwd) {
389
398
  const env = { ...process.env, PATH: `${join(cwd, "node_modules/.bin")}:${process.env.PATH ?? ""}`,
399
+ [VITEST_CACHE_ENV]: worktreeVitestCache(cwd),
390
400
  [FORK_CAP_ENV]: String(resolvedCapacity().forkCap), [SUITE_PARENT_ENV]: String(process.pid) };
391
401
  const verification = verificationProtocol(env, cwd);
392
402
  for (const key of ROUTING_ENV_SEAMS)
@@ -436,17 +446,68 @@ export async function manifestFileCount(cmd, cwd) {
436
446
  catch {
437
447
  return null;
438
448
  }
449
+ finally {
450
+ rmSync(dir, { recursive: true, force: true });
451
+ } // OBS-1155: the listing directory is the capture's alone, listed or not
452
+ }
453
+ /** Vitest's forks pool awaits the parallel phase, then throws before the single-fork phase
454
+ * on any rejected worker. Recover only that exact, fully accounted-for boundary. */
455
+ function strandedSingleForkFiles(files, nonce, run) {
456
+ const r = run.report;
457
+ if (run.killedFile || run.exitCode !== 1 || !r || r.nonce !== nonce || r.certificate?.exitCode !== 1)
458
+ return;
459
+ if (r.requested.length !== files.length || new Set(r.requested).size !== files.length
460
+ || r.requested.some(f => !files.includes(f)) || r.duplicateCompletions?.length)
461
+ return;
462
+ if (Object.keys(r.started).some(f => !files.includes(f))
463
+ || Object.keys(r.completed).some(f => !files.includes(f) || !Object.hasOwn(r.started, f)))
464
+ return;
465
+ const scheduling = r.scheduling;
466
+ if (!scheduling || Object.keys(scheduling).length !== files.length
467
+ || files.some(f => !Object.hasOwn(scheduling, f) || scheduling[f]?.pool !== "forks"
468
+ || typeof scheduling[f]?.singleFork !== "boolean"))
469
+ return;
470
+ const diagnostics = r.certificate.diagnostics;
471
+ if (!Array.isArray(diagnostics) || !diagnostics.length || r.certificate.errors !== diagnostics.length
472
+ || diagnostics.some(d => {
473
+ if (typeof d !== "string")
474
+ return true;
475
+ const timeout = /^(?:(.+): )?Error: \[vitest-worker\]: Timeout calling "[A-Za-z_$][\w$]*"$/.exec(d);
476
+ // The optional prefix is the reporter's testPath, not another error or arbitrary prose.
477
+ return !timeout || (timeout[1] !== undefined && !files.some(f => timeout[1] === f || timeout[1].endsWith(`/${f}`)));
478
+ }))
479
+ return;
480
+ const single = files.filter(f => scheduling[f].singleFork);
481
+ const parallel = files.filter(f => !scheduling[f].singleFork);
482
+ if (!single.length || !parallel.length || single.some(f => Object.hasOwn(r.started, f) || Object.hasOwn(r.completed, f)))
483
+ return;
484
+ if (parallel.some(f => !Object.hasOwn(r.started, f)
485
+ || !["passed", "skipped"].includes(r.completed[f]?.status ?? "")
486
+ || (r.completed[f]?.tests?.failed ?? 0) !== 0))
487
+ return;
488
+ return single;
489
+ }
490
+ /** The stranded single-fork retry: positional filters are substring matches that OR with any filter
491
+ * the command already carries, and under a `projects` config the CLI `--exclude` never subtracts such a
492
+ * selection (OBS-1166: a selected screen's retry rediscovered the whole selection and refused). So the
493
+ * retry is built from the UN-narrowed base command, its own `--` rule, the stranded files as the only
494
+ * positional filters, and an `--exclude` of every completed file; the caller then requires discovery
495
+ * to prove the exact retry set before launch. */
496
+ export function singleForkRetryCommand(base, cwd, stranded, completed) {
497
+ const excluded = completed.map(f => `--exclude=${shq(f.replace(/[\\*?[\]{}()!+@]/g, "\\$&"))}`).join(" ");
498
+ return `${base}${runnerInvocation(base, cwd).separator} ${stranded.map(f => shq(join(cwd, f))).join(" ")} ${excluded}`;
439
499
  }
440
500
  /** One configured runner execution, and its own collection under the same arguments and environment.
441
501
  * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
442
502
  export async function evaluateManifestedTest(cmd, cwd, opts) {
443
503
  const dir = opts.artifactDir ?? mkdtempSync(join(tmpdir(), "tickmarkr-test-report-"));
444
- const nonce = randomBytes(16).toString("hex");
445
- const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
504
+ let nonce = randomBytes(16).toString("hex");
505
+ let reportPath = join(dir, `test-manifest-report-${nonce}.json`);
446
506
  const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
447
507
  const evidenceReceipts = [];
448
508
  let spawnedCommand = cmd;
449
509
  let manifestPath;
510
+ let recovery;
450
511
  const { env, verification } = manifestEnvironment(cwd);
451
512
  try {
452
513
  const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
@@ -463,18 +524,55 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
463
524
  }, null, 2) + "\n");
464
525
  writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
465
526
  spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
466
- const invoked = await runManifestedTest(spawnedCommand, cwd, {
527
+ let invoked = await runManifestedTest(spawnedCommand, cwd, {
467
528
  evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: files, nonce, reportPath, env,
468
529
  baselineDurations: opts.baselineDurations?.map((d) => ({ ...d, file: toManifestPath(d.file, cwd) })),
469
530
  longestFile: opts.longestFile,
470
531
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
471
532
  pollMs: 20,
472
533
  });
473
- const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
534
+ let verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
474
535
  report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
536
+ evidenceReceipts.push(...invoked.evidenceReceipts);
537
+ const stranded = strandedSingleForkFiles(files, nonce, invoked);
538
+ if (stranded) {
539
+ const first = invoked.report;
540
+ const retryNonce = randomBytes(16).toString("hex");
541
+ recovery = { firstNonce: nonce, firstReportPath: reportPath, retryNonce, files: stranded };
542
+ nonce = retryNonce;
543
+ reportPath = join(dir, `test-manifest-report-${nonce}.json`);
544
+ const retryCommand = singleForkRetryCommand(opts.retryBaseCommand ?? cmd, cwd, stranded, files.filter(f => !stranded.includes(f)));
545
+ const listed = await discoverTestManifest(retryCommand, cwd, { dir, nonce, env,
546
+ overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
547
+ evidenceReceipts.push(...listed.evidenceReceipts);
548
+ if (listed.files.length !== stranded.length || listed.files.some(f => !stranded.includes(f)))
549
+ throw new Error("single fork retry discovery does not match the stranded set");
550
+ writeFileSync(join(dir, `test-manifest-expected-${nonce}.json`), JSON.stringify({ nonce, files: stranded,
551
+ firstNonce: recovery.firstNonce, listingCommand: listed.listing, verification }, null, 2) + "\n");
552
+ spawnedCommand = `${retryCommand}${listed.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
553
+ invoked = await runManifestedTest(spawnedCommand, cwd, {
554
+ evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: stranded, nonce, reportPath, env,
555
+ baselineDurations: opts.baselineDurations?.map(d => ({ ...d, file: toManifestPath(d.file, cwd) })),
556
+ longestFile: opts.longestFile, overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS, pollMs: 20,
557
+ });
558
+ evidenceReceipts.push(...invoked.evidenceReceipts);
559
+ verdict = verifyManifestReport({ manifest: stranded, nonce, exitCode: invoked.exitCode,
560
+ report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
561
+ // An all-skipped retry may contribute lifecycle accounting, but only the combined manifest
562
+ // can establish executed success. Never rewrite either invocation's persisted certificate.
563
+ if (verdict.pass || verdict.meta.noExecutedModules) {
564
+ const retry = invoked.report;
565
+ if (retry.certificate?.errors !== 0 || !Array.isArray(retry.certificate.diagnostics) || retry.certificate.diagnostics.length)
566
+ verdict = { kind: "fail-closed", pass: false, details: "single fork retry has unknown or nonempty runner diagnostics", meta: { classification: "infra", infra: true } };
567
+ else
568
+ verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode, report: {
569
+ ...retry, requested: files, started: { ...first.started, ...retry.started }, completed: { ...first.completed, ...retry.completed },
570
+ } });
571
+ }
572
+ verdict.details = `single fork retry ${nonce} after worker RPC timeout in ${recovery.firstNonce}: ${verdict.details}`;
573
+ }
475
574
  // Preserve the validator's verdict and classification; runner evidence only explains it.
476
575
  const evidenceReceipt = invoked.evidenceReceipt;
477
- evidenceReceipts.push(...invoked.evidenceReceipts);
478
576
  const evidenceRoot = opts.evidence?.artifactDir ?? dir;
479
577
  const stdoutPath = join(evidenceRoot, evidenceReceipt.stdout.path);
480
578
  const stderrPath = join(evidenceRoot, evidenceReceipt.stderr.path);
@@ -495,6 +593,7 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
495
593
  details: verdict.details + diagnostics,
496
594
  classification: verdict.meta.classification,
497
595
  meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
596
+ ...(recovery ? { recovery, retryable: false } : {}),
498
597
  spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
499
598
  exitCode: invoked.exitCode ?? -1, reportPath };
500
599
  }
@@ -503,7 +602,8 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
503
602
  if (evidenceReceipt)
504
603
  evidenceReceipts.push(...(error.evidenceReceipts ?? [evidenceReceipt]));
505
604
  return { evidenceReceipt, evidenceReceipts, pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
506
- details: error instanceof Error ? error.message : String(error),
507
- meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
605
+ details: (recovery ? `single fork retry ${recovery.retryNonce} after ${recovery.firstNonce}: ` : "") + (error instanceof Error ? error.message : String(error)),
606
+ meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification,
607
+ ...(recovery ? { recovery, retryable: false } : {}), ...(manifestPath ? { manifestPath } : {}) } };
508
608
  }
509
609
  }
@@ -35,6 +35,10 @@ export default class TickmarkrReporter {
35
35
  }
36
36
  onTestRunStart(specifications) {
37
37
  this.report.requested = specifications.map(s => this.file(s));
38
+ // The files-only CLI listing has no scheduling information. These are the resolved
39
+ // specifications the pool will consume, including per-file pool overrides.
40
+ this.report.scheduling = Object.fromEntries(specifications.filter(s => s.project?.config && typeof s.pool === 'string')
41
+ .map(s => [this.file(s), { pool: s.pool, singleFork: s.project.config.poolOptions?.forks?.singleFork === true }]));
38
42
  this.save();
39
43
  }
40
44
  onTestModuleStart(module) {