tickmarkr 2.5.8 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/dist/adapters/claude-code.js +19 -6
  2. package/dist/adapters/prompt.js +1 -1
  3. package/dist/adapters/qwen.js +30 -3
  4. package/dist/adapters/types.d.ts +3 -0
  5. package/dist/cli/commands/approve.js +15 -3
  6. package/dist/cli/commands/beat.d.ts +2 -0
  7. package/dist/cli/commands/beat.js +28 -29
  8. package/dist/cli/commands/compile.js +11 -0
  9. package/dist/cli/commands/fleet.js +61 -53
  10. package/dist/cli/commands/init.js +1 -1
  11. package/dist/cli/commands/report.js +11 -1
  12. package/dist/cli/commands/verify.d.ts +4 -1
  13. package/dist/cli/commands/verify.js +18 -5
  14. package/dist/cli/help.d.ts +2 -0
  15. package/dist/cli/help.js +6 -4
  16. package/dist/compile/native.js +39 -4
  17. package/dist/config/config.d.ts +41 -3
  18. package/dist/config/config.js +48 -17
  19. package/dist/config/fleet-overlay.js +47 -62
  20. package/dist/drivers/orca.d.ts +1 -1
  21. package/dist/drivers/orca.js +21 -4
  22. package/dist/gates/baseline.d.ts +67 -0
  23. package/dist/gates/baseline.js +172 -10
  24. package/dist/gates/cache.d.ts +8 -0
  25. package/dist/gates/cache.js +26 -13
  26. package/dist/gates/review.d.ts +8 -7
  27. package/dist/gates/review.js +83 -45
  28. package/dist/gates/run-gates.d.ts +4 -1
  29. package/dist/gates/run-gates.js +32 -12
  30. package/dist/gates/test-manifest.d.ts +16 -1
  31. package/dist/gates/test-manifest.js +127 -27
  32. package/dist/gates/test-reporter.js +4 -0
  33. package/dist/graph/graph.d.ts +1 -1
  34. package/dist/graph/graph.js +6 -1
  35. package/dist/graph/schema.d.ts +2 -0
  36. package/dist/graph/schema.js +2 -0
  37. package/dist/route/preference.d.ts +1 -1
  38. package/dist/route/preference.js +10 -39
  39. package/dist/run/daemon.d.ts +7 -1
  40. package/dist/run/daemon.js +237 -53
  41. package/dist/run/git.d.ts +10 -0
  42. package/dist/run/git.js +49 -1
  43. package/dist/run/journal.d.ts +18 -3
  44. package/dist/run/journal.js +141 -17
  45. package/dist/run/merge.d.ts +13 -2
  46. package/dist/run/merge.js +98 -12
  47. package/dist/run/protocol.d.ts +82 -0
  48. package/dist/run/protocol.js +35 -0
  49. package/dist/run/receipt-resolver.d.ts +18 -0
  50. package/dist/run/receipt-resolver.js +132 -0
  51. package/dist/run/supervision.d.ts +14 -1
  52. package/dist/run/supervision.js +122 -24
  53. package/dist/tui/cockpit/evidence-view.d.ts +10 -1
  54. package/dist/tui/cockpit/evidence-view.js +37 -5
  55. package/dist/tui/cockpit/live-runtime.d.ts +4 -0
  56. package/dist/tui/cockpit/live-runtime.js +37 -3
  57. package/dist/tui/cockpit/live-store.d.ts +2 -0
  58. package/dist/tui/ink/fleet-app.d.ts +12 -22
  59. package/dist/tui/ink/fleet-app.js +520 -131
  60. package/package.json +1 -1
  61. package/schema/rungraph.schema.json +7 -0
  62. package/skills/tickmarkr-overseer/SKILL.md +91 -41
  63. package/skills/tickmarkr-overseer/scripts/watch-context.sh +14 -2
@@ -2,7 +2,7 @@ import type { CommandReceiptAttribution } from "../run/protocol.js";
2
2
  import { type Assignment, type BillingChannel, type WorkerAdapter, type WorkerResult } from "../adapters/types.js";
3
3
  import { type TickmarkrConfig } from "../config/config.js";
4
4
  import { type GateName, type Task } from "../graph/schema.js";
5
- import { type Baseline } from "./baseline.js";
5
+ import { type Baseline, type GateEvidenceOptions } from "./baseline.js";
6
6
  import { type GateVia } from "./llm.js";
7
7
  import { type PriorReviewer } from "./review.js";
8
8
  import type { GateResult } from "./types.js";
@@ -46,6 +46,7 @@ export type GateEvent = {
46
46
  result?: GateResult;
47
47
  };
48
48
  export interface GateContext {
49
+ evidence?: GateEvidenceOptions;
49
50
  buildReceiptIdentity?: Omit<CommandReceiptAttribution, "invocation">;
50
51
  authorizeInfraRetry?: (subject: string, cause?: VerificationRetryCause) => boolean;
51
52
  verificationScope?: VerificationScope;
@@ -61,6 +62,8 @@ export interface GateContext {
61
62
  cfg: TickmarkrConfig;
62
63
  via?: GateVia;
63
64
  carriedFindings?: readonly StructuredFinding[];
65
+ /** Approval reason bound to this attempt; guidance, never criterion or closure authority. */
66
+ operatorContext?: string;
64
67
  excludeReviewers?: string[];
65
68
  demotedReviewers?: Set<string>;
66
69
  reviewNoVerdicts?: Map<string, string[]>;
@@ -17,7 +17,7 @@ import { scopeGate } from "./scope.js";
17
17
  import { evaluateManifestedTest, isVitestTestCommand } from "./test-manifest.js";
18
18
  import { executionSignal } from "../run/execution-budget.js";
19
19
  import { failureDisposition } from "../run/recovery.js";
20
- import { preserveWorktree, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
20
+ import { dependencyLinkRefusal, preserveWorktree, shGit, resolvedCapacity, verificationProtocol } from "../run/git.js";
21
21
  import { withJudgeInvocationEvidence } from "../run/journal.js";
22
22
  import { computeVerificationIdentity, verificationIdentityKey, formatReusedRow, getVerdictStore, isInfraResult, resolveStateDir, reusedIdentity, } from "./cache.js";
23
23
  const productionLoadProvider = () => loadavg()[0] ?? 0;
@@ -222,22 +222,25 @@ async function runVitestManifestGate(worktree, cmd, baseline, selected, artifact
222
222
  longestFile: entry?.longestFile,
223
223
  overallCeilingMs: effectiveCeilingMs(entry),
224
224
  artifactDir,
225
+ evidence: retry.evidence,
225
226
  });
226
227
  const reportPath = outcome.reportPath;
228
+ const evidence = { evidenceReceipt: outcome.evidenceReceipt, evidenceReceipts: outcome.evidenceReceipts };
227
229
  if (!retried && retry.authorizeRetry && failureDisposition(outcome) === "infrastructure"
228
230
  && outcome.meta?.retryable !== false) {
229
231
  const waitedMs = await waitForCalmWindow(executionSignal());
230
232
  if (!calmWindowReady())
231
- return { gate: "test", pass: false, details: outcome.details,
233
+ return { ...evidence, gate: "test", pass: false, details: outcome.details,
232
234
  meta: { ...outcome.meta, reportPath, recoveryBlocked: "calm window unavailable within the existing wait ceiling" } };
233
235
  if (!retry.authorizeRetry("infra")) {
234
- return { gate: "test", pass: false, details: outcome.details,
236
+ return { ...evidence, gate: "test", pass: false, details: outcome.details,
235
237
  meta: { ...outcome.meta, reportPath, recoveryBlocked: "infrastructure retry allowance exhausted or subject unavailable" } };
236
238
  }
237
239
  const result = await runVitestManifestGate(worktree, cmd, baseline, selected, artifactDir, retry, true);
238
- return { ...result, meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
240
+ return { ...result, evidenceReceipts: [...(outcome.evidenceReceipts ?? []), ...(result.evidenceReceipts ?? [])], meta: { ...result.meta, runnerInfraRerun: { count: 1, waitedMs, firstReportPath: reportPath } } };
239
241
  }
240
242
  return {
243
+ ...evidence,
241
244
  gate: "test",
242
245
  pass: outcome.pass,
243
246
  details: outcome.details,
@@ -259,6 +262,12 @@ function classifySignalOnlyTest(g) {
259
262
  }
260
263
  export async function runGates(task, ctx) {
261
264
  const results = [];
265
+ const evidence = {
266
+ artifactDir: ctx.artifactDir,
267
+ runId: ctx.buildReceiptIdentity?.runId ?? ctx.artifactDir ?? "standalone",
268
+ taskId: task.id, attempt: ctx.buildReceiptIdentity?.attempt ?? 0,
269
+ ...ctx.evidence,
270
+ };
262
271
  // Receipt identity belongs to this round, never to a cached verdict. Each call from the shell
263
272
  // allocates a new invocation, including retries whose local spawn counter starts at one again.
264
273
  let currentBuild;
@@ -295,6 +304,16 @@ export async function runGates(task, ctx) {
295
304
  };
296
305
  let selectionDecision;
297
306
  let commits = [];
307
+ // Check before cache identity, npm policy probes, or any gate command.
308
+ const dependencyRefusal = dependencyLinkRefusal(ctx.worktree);
309
+ if (dependencyRefusal) {
310
+ const gate = GATE_NAMES.find(g => task.gates.includes(g)) ?? "build";
311
+ const result = { gate, pass: false, details: dependencyRefusal,
312
+ meta: { infra: true, classification: "infra", retryable: false, kind: "workspace-dependency" } };
313
+ await noBuild("refused", dependencyRefusal);
314
+ await ctx.onGate?.({ phase: "end", gate, result });
315
+ return { results: [result], commits: [] };
316
+ }
298
317
  const stateDir = ctx.stateDir ?? resolveStateDir(ctx.worktree, ctx.artifactDir);
299
318
  const verdictStore = getVerdictStore(stateDir);
300
319
  // R41: the verification protocol and the EFFECTIVE npm lifecycle policy measured for THIS
@@ -658,8 +677,8 @@ export async function runGates(task, ctx) {
658
677
  // other scripted test command keeps today's exit-code contract byte-identically.
659
678
  const useManifest = g === "test" && commands.test !== undefined && isVitestTestCommand(commands.test, ctx.worktree);
660
679
  r = useManifest
661
- ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, retryOptions(identity)))
662
- : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
680
+ ? await measure(g, () => runVitestManifestGate(ctx.worktree, commands.test, ctx.baseline, selected, ctx.artifactDir, { ...retryOptions(identity), evidence }))
681
+ : (await measure(g, () => compareToBaseline(ctx.worktree, commands, ctx.baseline, [g], { ...retryOptions(identity), evidence, ...(g === "build" ? { onReceipt: buildReceipt, taskBuildAttribution: beginBuild } : {}), ...(g === "test" && selected ? { selected } : {}) })))[0];
663
682
  }
664
683
  finally {
665
684
  await receiptNotes;
@@ -676,7 +695,7 @@ export async function runGates(task, ctx) {
676
695
  if (!cached && r.pass && commands[g]) {
677
696
  const dirt = await dirtyWorktree();
678
697
  if (dirt) {
679
- await record(await dirtyRefusal(g, dirt, commands[g]));
698
+ await record({ ...await dirtyRefusal(g, dirt, commands[g]), evidenceReceipt: r.evidenceReceipt, evidenceReceipts: r.evidenceReceipts });
680
699
  return;
681
700
  }
682
701
  }
@@ -903,7 +922,7 @@ export async function runGates(task, ctx) {
903
922
  const priorReviewers = [...(ctx.priorReviewers ?? []), ...(ctx.excludeReviewers ?? [])];
904
923
  const carriedAuthors = ctx.carriedAuthors ?? [];
905
924
  let exclusions = [...(ctx.excludeReviewers ?? []), ...retired];
906
- let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors));
925
+ let rv = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, ctx.via, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, priorReviewers, carriedAuthors, ctx.operatorContext));
907
926
  // OBS-193/574: an unparseable review verdict retries the REVIEW, preferring a different adapter. Only
908
927
  // a single-adapter eligible pool may fall back to another channel on the flaked adapter. The flaked
909
928
  // verdict never enters results; an exhausted pool preserves its cause.
@@ -936,7 +955,7 @@ export async function runGates(task, ctx) {
936
955
  const crossAdapter = pickReviewer(ctx.author, ctx.channels, [...exclusions, ...adapterExclusions], ctx.cfg.review.prefer ?? [], retryFloor, undefined, undefined, undefined, carriedAuthorVendors(ctx.channels, carriedAuthors));
937
956
  const exclusion = crossAdapter ? "adapter" : "channel";
938
957
  exclusions = [...exclusions, ...(crossAdapter ? adapterExclusions : [flaked])];
939
- const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors));
958
+ const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, exclusions, ctx.artifactDir, ctx.reviewHistory, ctx.demotedReviewers, ctx.carriedFindings, retryPrior, carriedAuthors, ctx.operatorContext));
940
959
  if (second.meta?.noEligibleReviewer !== true) {
941
960
  const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
942
961
  const route = exclusion === "adapter"
@@ -1104,8 +1123,8 @@ export async function runGates(task, ctx) {
1104
1123
  if (!full) {
1105
1124
  const fullUsesManifest = ctx.commands.test !== undefined && isVitestTestCommand(ctx.commands.test, ctx.worktree);
1106
1125
  full = fullUsesManifest
1107
- ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, retryOptions(identity)))
1108
- : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], retryOptions(identity))))[0];
1126
+ ? await measure("test", () => runVitestManifestGate(ctx.worktree, ctx.commands.test, ctx.baseline, undefined, ctx.artifactDir, { ...retryOptions(identity), evidence }))
1127
+ : (await measure("test", () => compareToBaseline(ctx.worktree, ctx.commands, ctx.baseline, ["test"], { ...retryOptions(identity), evidence })))[0];
1109
1128
  }
1110
1129
  fullDurationMs = spans.get("test") ? spans.get("test").durationMs - (selectedDurationMs ?? 0) : 0;
1111
1130
  const dirt = (!cached && full.pass) ? await dirtyWorktree() : undefined;
@@ -1115,8 +1134,9 @@ export async function runGates(task, ctx) {
1115
1134
  verdictStore.set(identity, { ...full, meta: { ...full.meta, source: "gate", runDir: ctx.artifactDir } });
1116
1135
  }
1117
1136
  const merged = withTelemetry(dirt
1118
- ? await dirtyRefusal("test", dirt, ctx.commands.test)
1137
+ ? { ...await dirtyRefusal("test", dirt, ctx.commands.test), evidenceReceipt: full.evidenceReceipt, evidenceReceipts: full.evidenceReceipts }
1119
1138
  : { ...full, meta: { ...full.meta, fullSuite: true, selectedTests: selected } });
1139
+ merged.evidenceReceipts = [...(heldTest?.evidenceReceipts ?? []), ...(full?.evidenceReceipts ?? [])];
1120
1140
  results[results.findIndex((r) => r.gate === "test")] = merged;
1121
1141
  heldTest = undefined;
1122
1142
  await ctx.onGate?.({ phase: "end", gate: "test", result: merged });
@@ -1,4 +1,5 @@
1
- import type { BaselineFileDuration } from "./baseline.js";
1
+ import { type GateEvidenceOptions, type BaselineFileDuration } from "./baseline.js";
2
+ import type { GateEvidenceReceipt } from "../run/protocol.js";
2
3
  export declare function isVitestTestCommand(cmd: string, cwd: string): boolean;
3
4
  /** One identity for every path this module compares: repo-relative, forward-slash. `vitest list
4
5
  * --json` and `TestModule.moduleId` both hand back an absolute filesystem path already resolved
@@ -36,6 +37,11 @@ export interface TestReport {
36
37
  requested: string[];
37
38
  started: Record<string, number>;
38
39
  completed: Record<string, TestReportCompletion>;
40
+ /** Resolved scheduling observed at run start; absent/partial means unknown, never inferred. */
41
+ scheduling?: Record<string, {
42
+ pool: string;
43
+ singleFork: boolean;
44
+ }>;
39
45
  /** Files the reporter observed complete MORE than once — `completed`'s object keys cannot show
40
46
  * this themselves (a second write silently overwrites the first), so the reporter records the
41
47
  * evidence separately before it is lost. */
@@ -78,6 +84,8 @@ export declare const FILE_HANG_SLACK = 3;
78
84
  export declare const DEFAULT_FILE_HANG_BUDGET_MS = 60000;
79
85
  export declare function fileHangBudgetMs(file: string, baselineDurations?: readonly BaselineFileDuration[] | null, ceilingMs?: number, longestFile?: BaselineFileDuration | null): number;
80
86
  export interface ManifestRunResult {
87
+ evidenceReceipt: GateEvidenceReceipt;
88
+ evidenceReceipts: GateEvidenceReceipt[];
81
89
  exitCode: number | undefined;
82
90
  stdout: string;
83
91
  stderr: string;
@@ -91,6 +99,7 @@ export interface ManifestRunResult {
91
99
  /** Supervise the configured command and poll the runner's atomic lifecycle snapshots. Every
92
100
  * timeout kills the detached process group, including descendants holding the output pipes. */
93
101
  export declare function runManifestedTest(cmd: string, cwd: string, opts: {
102
+ evidence?: GateEvidenceOptions;
94
103
  manifest: readonly string[];
95
104
  nonce: string;
96
105
  reportPath: string;
@@ -101,6 +110,8 @@ export declare function runManifestedTest(cmd: string, cwd: string, opts: {
101
110
  overallCeilingMs?: number;
102
111
  }): Promise<ManifestRunResult>;
103
112
  export interface ManifestGateOutcome {
113
+ evidenceReceipt?: GateEvidenceReceipt;
114
+ evidenceReceipts?: GateEvidenceReceipt[];
104
115
  pass: boolean;
105
116
  kind: ManifestVerdictKind;
106
117
  details: string;
@@ -115,6 +126,8 @@ export interface DiscoveredManifest {
115
126
  separator: string;
116
127
  listingExit: number | undefined;
117
128
  listingStdout: string;
129
+ evidenceReceipt: GateEvidenceReceipt;
130
+ evidenceReceipts: GateEvidenceReceipt[];
118
131
  }
119
132
  /** OBS-1044: THE discovery seam — the files the runner would collect under this exact invocation,
120
133
  * from its own listing and never from a stdout summary. The gate and the baseline capture share it,
@@ -125,6 +138,7 @@ export declare function discoverTestManifest(cmd: string, cwd: string, opts: {
125
138
  nonce: string;
126
139
  env: NodeJS.ProcessEnv;
127
140
  overallCeilingMs?: number;
141
+ evidence?: GateEvidenceOptions;
128
142
  }): Promise<DiscoveredManifest>;
129
143
  /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
130
144
  * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
@@ -137,4 +151,5 @@ export declare function evaluateManifestedTest(cmd: string, cwd: string, opts: {
137
151
  longestFile?: BaselineFileDuration | null;
138
152
  overallCeilingMs?: number;
139
153
  artifactDir?: string;
154
+ evidence?: GateEvidenceOptions;
140
155
  }): Promise<ManifestGateOutcome>;
@@ -4,6 +4,7 @@ import { tmpdir } from "node:os";
4
4
  import { isAbsolute, join, relative, sep } from "node:path";
5
5
  import { TEST_REPORTER_SOURCE } from "./test-reporter.js";
6
6
  import { shq } from "../adapters/types.js";
7
+ import { beginGateEvidence, redactGateOutput } from "./baseline.js";
7
8
  import { FORK_CAP_ENV, ROUTING_ENV_SEAMS, SUITE_PARENT_ENV, shell, resolvedCapacity, verificationProtocol } from "../run/git.js";
8
9
  /**
9
10
  * VL-1 (OBS-985 lineage): a test gate's completion must be the runner's OWN report, never a stdout
@@ -330,6 +331,7 @@ export function runManifestedTest(cmd, cwd, opts) {
330
331
  const pollMs = opts.pollMs ?? 20;
331
332
  const overallCeilingMs = usable(opts.overallCeilingMs) ? opts.overallCeilingMs : DEFAULT_FILE_HANG_BUDGET_MS;
332
333
  const env = { ...(opts.env ?? process.env), TICKMARKR_TEST_REPORT: opts.reportPath, TICKMARKR_TEST_NONCE: opts.nonce };
334
+ const evidence = beginGateEvidence(cwd, "test", cmd, { ...opts.evidence, env }, opts.nonce);
333
335
  const controller = new AbortController();
334
336
  let pid;
335
337
  let killedFile;
@@ -355,6 +357,7 @@ export function runManifestedTest(cmd, cwd, opts) {
355
357
  }
356
358
  };
357
359
  return shell(cmd, cwd, overallCeilingMs, false, {
360
+ onReceipt: receipt => evidence.observe(killedFile && receipt.outcome === "cancelled" ? { ...receipt, outcome: "timed-out" } : receipt),
358
361
  env, signal: controller.signal, onTimeout: () => checkHang(true),
359
362
  onSpawn: (childPid) => {
360
363
  pid = childPid;
@@ -362,11 +365,21 @@ export function runManifestedTest(cmd, cwd, opts) {
362
365
  clearInterval(poll);
363
366
  poll = setInterval(checkHang, pollMs);
364
367
  },
365
- }).then((result) => ({
366
- exitCode: result.signalExit ? undefined : result.code,
367
- stdout: result.stdout, stderr: result.stderr,
368
- report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
369
- })).finally(() => { clearInterval(poll); });
368
+ }).then((result) => {
369
+ const evidenceReceipt = evidence.finish(result.stdout, result.stderr);
370
+ return {
371
+ evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt],
372
+ exitCode: result.signalExit ? undefined : result.code,
373
+ stdout: result.stdout, stderr: result.stderr,
374
+ report: readTestReport(opts.reportPath), killedFile, hangBudgetMs, pid,
375
+ };
376
+ }).catch((error) => {
377
+ if (error instanceof Error) {
378
+ const evidenceReceipt = evidence.finish();
379
+ Object.assign(error, { evidenceReceipt, evidenceReceipts: [...evidence.history, evidenceReceipt] });
380
+ }
381
+ throw error;
382
+ }).finally(() => { clearInterval(poll); });
370
383
  }
371
384
  /** The child environment every manifest invocation (listing and run) receives, and the lifecycle
372
385
  * protocol it records. R41: the policy is whatever THIS process was launched with (npm reads
@@ -391,11 +404,12 @@ function manifestEnvironment(cwd) {
391
404
  export async function discoverTestManifest(cmd, cwd, opts) {
392
405
  const invocation = runnerInvocation(cmd, cwd);
393
406
  const listed = await runManifestedTest(invocation.listing, cwd, {
394
- manifest: [], nonce: opts.nonce, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
407
+ evidence: opts.evidence, manifest: [], nonce: `${opts.nonce}-listing`, reportPath: join(opts.dir, `listing-${opts.nonce}.json`), env: opts.env,
395
408
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
396
409
  });
410
+ const discoveryError = (message) => Object.assign(new Error(message), { evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts });
397
411
  if (listed.exitCode !== 0)
398
- throw new Error(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
412
+ throw discoveryError(`vitest cannot list files (exit ${listed.exitCode ?? "signal"}): ${listed.stderr || listed.stdout}`);
399
413
  let files;
400
414
  try {
401
415
  const rows = JSON.parse(listed.stdout.slice(listed.stdout.indexOf("[")));
@@ -404,11 +418,11 @@ export async function discoverTestManifest(cmd, cwd, opts) {
404
418
  files = [...new Set(rows.map((r) => toManifestPath(r.file, cwd)))].sort();
405
419
  }
406
420
  catch {
407
- throw new Error(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
421
+ throw discoveryError(`vitest cannot list files: invalid JSON listing: ${listed.stdout}`);
408
422
  }
409
423
  if (!files.length)
410
- throw new Error("vitest cannot list files: empty manifest");
411
- return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout };
424
+ throw discoveryError("vitest cannot list files: empty manifest");
425
+ return { files, listing: invocation.listing, separator: invocation.separator, listingExit: listed.exitCode, listingStdout: listed.stdout, evidenceReceipt: listed.evidenceReceipt, evidenceReceipts: listed.evidenceReceipts };
412
426
  }
413
427
  /** The baseline capture's reading of the same seam: the manifest's file count, or null when the
414
428
  * runner cannot list (a null compares nothing — it never manufactures a deficit). The suite is not
@@ -423,18 +437,58 @@ export async function manifestFileCount(cmd, cwd) {
423
437
  return null;
424
438
  }
425
439
  }
440
+ /** Vitest's forks pool awaits the parallel phase, then throws before the single-fork phase
441
+ * on any rejected worker. Recover only that exact, fully accounted-for boundary. */
442
+ function strandedSingleForkFiles(files, nonce, run) {
443
+ const r = run.report;
444
+ if (run.killedFile || run.exitCode !== 1 || !r || r.nonce !== nonce || r.certificate?.exitCode !== 1)
445
+ return;
446
+ if (r.requested.length !== files.length || new Set(r.requested).size !== files.length
447
+ || r.requested.some(f => !files.includes(f)) || r.duplicateCompletions?.length)
448
+ return;
449
+ if (Object.keys(r.started).some(f => !files.includes(f))
450
+ || Object.keys(r.completed).some(f => !files.includes(f) || !Object.hasOwn(r.started, f)))
451
+ return;
452
+ const scheduling = r.scheduling;
453
+ if (!scheduling || Object.keys(scheduling).length !== files.length
454
+ || files.some(f => !Object.hasOwn(scheduling, f) || scheduling[f]?.pool !== "forks"
455
+ || typeof scheduling[f]?.singleFork !== "boolean"))
456
+ return;
457
+ const diagnostics = r.certificate.diagnostics;
458
+ if (!Array.isArray(diagnostics) || !diagnostics.length || r.certificate.errors !== diagnostics.length
459
+ || diagnostics.some(d => {
460
+ if (typeof d !== "string")
461
+ return true;
462
+ const timeout = /^(?:(.+): )?Error: \[vitest-worker\]: Timeout calling "[A-Za-z_$][\w$]*"$/.exec(d);
463
+ // The optional prefix is the reporter's testPath, not another error or arbitrary prose.
464
+ return !timeout || (timeout[1] !== undefined && !files.some(f => timeout[1] === f || timeout[1].endsWith(`/${f}`)));
465
+ }))
466
+ return;
467
+ const single = files.filter(f => scheduling[f].singleFork);
468
+ const parallel = files.filter(f => !scheduling[f].singleFork);
469
+ if (!single.length || !parallel.length || single.some(f => Object.hasOwn(r.started, f) || Object.hasOwn(r.completed, f)))
470
+ return;
471
+ if (parallel.some(f => !Object.hasOwn(r.started, f)
472
+ || !["passed", "skipped"].includes(r.completed[f]?.status ?? "")
473
+ || (r.completed[f]?.tests?.failed ?? 0) !== 0))
474
+ return;
475
+ return single;
476
+ }
426
477
  /** One configured runner execution, and its own collection under the same arguments and environment.
427
478
  * The installed runner is trusted (R28 add.1 option B); the nonce catches stale artifacts, not forgery. */
428
479
  export async function evaluateManifestedTest(cmd, cwd, opts) {
429
480
  const dir = opts.artifactDir ?? mkdtempSync(join(tmpdir(), "tickmarkr-test-report-"));
430
- const nonce = randomBytes(16).toString("hex");
431
- const reportPath = join(dir, `test-manifest-report-${nonce}.json`);
481
+ let nonce = randomBytes(16).toString("hex");
482
+ let reportPath = join(dir, `test-manifest-report-${nonce}.json`);
432
483
  const reporterPath = join(dir, `test-reporter-${nonce}.mjs`);
484
+ const evidenceReceipts = [];
433
485
  let spawnedCommand = cmd;
434
486
  let manifestPath;
487
+ let recovery;
435
488
  const { env, verification } = manifestEnvironment(cwd);
436
489
  try {
437
- const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs });
490
+ const invocation = await discoverTestManifest(cmd, cwd, { dir, nonce, env, overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
491
+ evidenceReceipts.push(...invocation.evidenceReceipts);
438
492
  const files = invocation.files;
439
493
  // R41: the EXPECTED manifest is evidence in its own right — persisted beside the report with the
440
494
  // exact discovery invocation, so a later reader can tell what this invocation was asked to prove
@@ -447,22 +501,63 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
447
501
  }, null, 2) + "\n");
448
502
  writeFileSync(reporterPath, TEST_REPORTER_SOURCE);
449
503
  spawnedCommand = `${cmd}${invocation.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
450
- const invoked = await runManifestedTest(spawnedCommand, cwd, {
451
- manifest: files, nonce, reportPath, env,
504
+ let invoked = await runManifestedTest(spawnedCommand, cwd, {
505
+ evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: files, nonce, reportPath, env,
452
506
  baselineDurations: opts.baselineDurations?.map((d) => ({ ...d, file: toManifestPath(d.file, cwd) })),
453
507
  longestFile: opts.longestFile,
454
508
  overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS,
455
509
  pollMs: 20,
456
510
  });
457
- const verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
511
+ let verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode,
458
512
  report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
513
+ evidenceReceipts.push(...invoked.evidenceReceipts);
514
+ const stranded = strandedSingleForkFiles(files, nonce, invoked);
515
+ if (stranded) {
516
+ const first = invoked.report;
517
+ const retryNonce = randomBytes(16).toString("hex");
518
+ recovery = { firstNonce: nonce, firstReportPath: reportPath, retryNonce, files: stranded };
519
+ nonce = retryNonce;
520
+ reportPath = join(dir, `test-manifest-report-${nonce}.json`);
521
+ // Positional filters are substring matches (and OR with existing filters). Exclude every
522
+ // completed file as well, then require discovery to prove the exact retry set before launch.
523
+ const excluded = files.filter(f => !stranded.includes(f)).map(f => `--exclude=${shq(f.replace(/[\\*?[\]{}()!+@]/g, "\\$&"))}`).join(" ");
524
+ const retryCommand = `${cmd}${invocation.separator} ${stranded.map(f => shq(join(cwd, f))).join(" ")} ${excluded}`;
525
+ const listed = await discoverTestManifest(retryCommand, cwd, { dir, nonce, env,
526
+ overallCeilingMs: opts.overallCeilingMs, evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir } });
527
+ evidenceReceipts.push(...listed.evidenceReceipts);
528
+ if (listed.files.length !== stranded.length || listed.files.some(f => !stranded.includes(f)))
529
+ throw new Error("single fork retry discovery does not match the stranded set");
530
+ writeFileSync(join(dir, `test-manifest-expected-${nonce}.json`), JSON.stringify({ nonce, files: stranded,
531
+ firstNonce: recovery.firstNonce, listingCommand: listed.listing, verification }, null, 2) + "\n");
532
+ spawnedCommand = `${retryCommand}${listed.separator} --reporter=${shq(reporterPath)} --outputFile=${shq(reportPath)}`;
533
+ invoked = await runManifestedTest(spawnedCommand, cwd, {
534
+ evidence: { ...opts.evidence, artifactDir: opts.evidence?.artifactDir ?? dir }, manifest: stranded, nonce, reportPath, env,
535
+ baselineDurations: opts.baselineDurations?.map(d => ({ ...d, file: toManifestPath(d.file, cwd) })),
536
+ longestFile: opts.longestFile, overallCeilingMs: opts.overallCeilingMs ?? DEFAULT_FILE_HANG_BUDGET_MS, pollMs: 20,
537
+ });
538
+ evidenceReceipts.push(...invoked.evidenceReceipts);
539
+ verdict = verifyManifestReport({ manifest: stranded, nonce, exitCode: invoked.exitCode,
540
+ report: invoked.report, killedFile: invoked.killedFile, hangBudgetMs: invoked.hangBudgetMs });
541
+ // An all-skipped retry may contribute lifecycle accounting, but only the combined manifest
542
+ // can establish executed success. Never rewrite either invocation's persisted certificate.
543
+ if (verdict.pass || verdict.meta.noExecutedModules) {
544
+ const retry = invoked.report;
545
+ if (retry.certificate?.errors !== 0 || !Array.isArray(retry.certificate.diagnostics) || retry.certificate.diagnostics.length)
546
+ verdict = { kind: "fail-closed", pass: false, details: "single fork retry has unknown or nonempty runner diagnostics", meta: { classification: "infra", infra: true } };
547
+ else
548
+ verdict = verifyManifestReport({ manifest: files, nonce, exitCode: invoked.exitCode, report: {
549
+ ...retry, requested: files, started: { ...first.started, ...retry.started }, completed: { ...first.completed, ...retry.completed },
550
+ } });
551
+ }
552
+ verdict.details = `single fork retry ${nonce} after worker RPC timeout in ${recovery.firstNonce}: ${verdict.details}`;
553
+ }
459
554
  // Preserve the validator's verdict and classification; runner evidence only explains it.
460
- const stdoutPath = join(dir, `test-runner-stdout-${nonce}.log`);
461
- const stderrPath = join(dir, `test-runner-stderr-${nonce}.log`);
462
- const stdoutTail = Buffer.from(invoked.stdout).subarray(-16 * 1024);
463
- const stderrTail = Buffer.from(invoked.stderr).subarray(-16 * 1024);
464
- writeFileSync(stdoutPath, stdoutTail);
465
- writeFileSync(stderrPath, stderrTail);
555
+ const evidenceReceipt = invoked.evidenceReceipt;
556
+ const evidenceRoot = opts.evidence?.artifactDir ?? dir;
557
+ const stdoutPath = join(evidenceRoot, evidenceReceipt.stdout.path);
558
+ const stderrPath = join(evidenceRoot, evidenceReceipt.stderr.path);
559
+ const stdoutTail = Buffer.from(redactGateOutput(invoked.stdout, env)).subarray(-16 * 1024);
560
+ const stderrTail = Buffer.from(redactGateOutput(invoked.stderr, env)).subarray(-16 * 1024);
466
561
  const report = invoked.report?.nonce === nonce ? invoked.report : undefined;
467
562
  const neverStarted = report ? files.filter(file => !(file in report.started)).length : "unknown";
468
563
  const errors = report?.certificate?.errors;
@@ -470,20 +565,25 @@ export async function evaluateManifestedTest(cmd, cwd, opts) {
470
565
  const reportedDiagnostics = report?.certificate?.diagnostics;
471
566
  const runnerErrors = Array.isArray(reportedDiagnostics) ? reportedDiagnostics.filter(error => typeof error === "string") : [];
472
567
  const diagnostics = !verdict.pass
473
- ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}`
568
+ ? `\nclassification: ${verdict.meta.classification ?? "unknown"}; runner-level diagnostic: never-started ${neverStarted}; reporter errors ${reporterErrors}; runner vitest`
474
569
  + [...runnerErrors,
475
570
  stdoutTail.toString(), stderrTail.toString()].filter(Boolean).map(text => `\n${text}`).join("")
476
571
  : "";
477
- return { pass: verdict.pass, kind: verdict.kind,
572
+ return { evidenceReceipt, evidenceReceipts, pass: verdict.pass, kind: verdict.kind,
478
573
  details: verdict.details + diagnostics,
479
574
  classification: verdict.meta.classification,
480
575
  meta: { ...verdict.meta, nonce, manifest: files, manifestPath, listingCommand: invocation.listing, verification,
576
+ ...(recovery ? { recovery, retryable: false } : {}),
481
577
  spawnedCommand, processExit: invoked.exitCode, pid: invoked.pid, stdoutPath, stderrPath },
482
578
  exitCode: invoked.exitCode ?? -1, reportPath };
483
579
  }
484
580
  catch (error) {
485
- return { pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
486
- details: error instanceof Error ? error.message : String(error),
487
- meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification, ...(manifestPath ? { manifestPath } : {}) } };
581
+ const evidenceReceipt = error?.evidenceReceipt;
582
+ if (evidenceReceipt)
583
+ evidenceReceipts.push(...(error.evidenceReceipts ?? [evidenceReceipt]));
584
+ return { evidenceReceipt, evidenceReceipts, pass: false, kind: "infra", classification: "infra", exitCode: -1, reportPath,
585
+ details: (recovery ? `single fork retry ${recovery.retryNonce} after ${recovery.firstNonce}: ` : "") + (error instanceof Error ? error.message : String(error)),
586
+ meta: { classification: "infra", infra: true, manifestDiscoveryFailed: true, spawnedCommand, verification,
587
+ ...(recovery ? { recovery, retryable: false } : {}), ...(manifestPath ? { manifestPath } : {}) } };
488
588
  }
489
589
  }
@@ -35,6 +35,10 @@ export default class TickmarkrReporter {
35
35
  }
36
36
  onTestRunStart(specifications) {
37
37
  this.report.requested = specifications.map(s => this.file(s));
38
+ // The files-only CLI listing has no scheduling information. These are the resolved
39
+ // specifications the pool will consume, including per-file pool overrides.
40
+ this.report.scheduling = Object.fromEntries(specifications.filter(s => s.project?.config && typeof s.pool === 'string')
41
+ .map(s => [this.file(s), { pool: s.pool, singleFork: s.project.config.poolOptions?.forks?.singleFork === true }]));
38
42
  this.save();
39
43
  }
40
44
  onTestModuleStart(module) {
@@ -24,7 +24,7 @@ export declare function onDiskSpecHash(_repoRoot: string, graph: RunGraph): OnDi
24
24
  /** A task's definition minus its files[] (and run state): what a scope amendment must NOT move (OBS-1073). */
25
25
  export declare function taskDefinitionFingerprint(task: Task): string;
26
26
  export declare function graphDefinitionHash(g: RunGraph): string;
27
- export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance">): string;
27
+ export declare function taskContentDigest(task: Pick<Task, "goal" | "files" | "acceptance" | "outOfScope">): string;
28
28
  export declare function tickmarkrDir(repoRoot: string): string;
29
29
  export declare function loadGraph(repoRoot: string): RunGraph;
30
30
  export declare function saveGraph(repoRoot: string, g: RunGraph): void;
@@ -94,9 +94,14 @@ export function graphDefinitionHash(g) {
94
94
  // finding; changing the goal, write surface or acceptance contract does. Keep the full digest here:
95
95
  // unlike graphDefinitionHash this value is persisted beside evidence and is the fail-closed join a
96
96
  // later run uses, so there is no benefit in making collision diagnosis less explicit.
97
+ // OBS-1126: outOfScope joins the identity only when present — an absent list keeps the historical
98
+ // bytes so every digest persisted before the field existed still joins.
97
99
  export function taskContentDigest(task) {
98
100
  return createHash("sha256")
99
- .update(JSON.stringify({ goal: task.goal, files: task.files, acceptance: task.acceptance }))
101
+ .update(JSON.stringify({
102
+ goal: task.goal, files: task.files, acceptance: task.acceptance,
103
+ ...(task.outOfScope ? { outOfScope: task.outOfScope } : {}),
104
+ }))
100
105
  .digest("hex");
101
106
  }
102
107
  export function tickmarkrDir(repoRoot) {
@@ -80,6 +80,7 @@ export declare const TaskSchema: z.ZodObject<{
80
80
  kind: z.ZodLiteral<"fixture">;
81
81
  paths: z.ZodArray<z.ZodString>;
82
82
  }, z.core.$strip>], "kind">>>;
83
+ outOfScope: z.ZodOptional<z.ZodArray<z.ZodString>>;
83
84
  gates: z.ZodDefault<z.ZodArray<z.ZodEnum<{
84
85
  build: "build";
85
86
  test: "test";
@@ -181,6 +182,7 @@ export declare const RunGraphSchema: z.ZodObject<{
181
182
  kind: z.ZodLiteral<"fixture">;
182
183
  paths: z.ZodArray<z.ZodString>;
183
184
  }, z.core.$strip>], "kind">>>;
185
+ outOfScope: z.ZodOptional<z.ZodArray<z.ZodString>>;
184
186
  gates: z.ZodDefault<z.ZodArray<z.ZodEnum<{
185
187
  build: "build";
186
188
  test: "test";
@@ -71,6 +71,8 @@ export const TaskSchema = z.object({
71
71
  }))
72
72
  .optional(),
73
73
  pins: z.array(PinSchema).optional(),
74
+ // OBS-1126: declared bounds the task must not cross; absent ⇒ no key (sealed graph bytes hold).
75
+ outOfScope: z.array(z.string().min(1)).optional(),
74
76
  gates: z
75
77
  .array(z.enum(GATE_NAMES))
76
78
  .default(["build", "test", "lint", "evidence", "scope", "acceptance", "review"])
@@ -1,5 +1,5 @@
1
1
  import { type AuthHealth } from "../adapters/types.js";
2
- import type { TickmarkrConfig } from "../config/config.js";
2
+ import { type TickmarkrConfig } from "../config/config.js";
3
3
  export interface Disallowed {
4
4
  by: "deny" | "allow";
5
5
  entry: string;
@@ -1,4 +1,5 @@
1
1
  import { channelKey, channelsFromConfig } from "../adapters/types.js";
2
+ import { DENY_SCOPES, denyEntriesAt } from "../config/config.js";
2
3
  import { validateGraph } from "../graph/schema.js";
3
4
  import { route, RoutingError } from "./router.js";
4
5
  const PREFERENCE_ROLES = ["worker", "judge", "review", "consult"];
@@ -121,47 +122,17 @@ export function entryMatchesChannel(entry, c, allowFamilyMatching = false) {
121
122
  export function exclusionCollector(c, routingOrCfg, role = "worker") {
122
123
  const routing = "routing" in routingOrCfg ? routingOrCfg.routing : routingOrCfg;
123
124
  const out = [];
124
- const { allow, deny } = routing ?? {};
125
- for (const entry of deny?.adapters ?? []) {
126
- if (entryMatchesChannel(entry, c, true)) {
127
- out.push({
128
- scope: "routing.deny.adapters",
129
- path: "routing.deny.adapters",
130
- configPath: "routing.deny.adapters",
131
- entry,
132
- by: "deny",
133
- });
134
- }
135
- }
136
- for (const entry of deny?.models ?? []) {
137
- if (entryMatchesChannel(entry, c, true)) {
138
- out.push({
139
- scope: "routing.deny.models",
140
- path: "routing.deny.models",
141
- configPath: "routing.deny.models",
142
- entry,
143
- by: "deny",
144
- });
145
- }
146
- }
147
- if (role === "worker") {
148
- for (const entry of deny?.workers?.adapters ?? []) {
149
- if (entryMatchesChannel(entry, c, true)) {
150
- out.push({
151
- scope: "routing.deny.workers.adapters",
152
- path: "routing.deny.workers.adapters",
153
- configPath: "routing.deny.workers.adapters",
154
- entry,
155
- by: "deny",
156
- });
157
- }
158
- }
159
- for (const entry of deny?.workers?.models ?? []) {
125
+ const { allow } = routing ?? {};
126
+ for (const scope of DENY_SCOPES) {
127
+ // Lists under workers apply only to worker seats; flat deny lists cover every role.
128
+ if (scope.path[2] === "workers" && role !== "worker")
129
+ continue;
130
+ for (const entry of denyEntriesAt(routing, scope) ?? []) {
160
131
  if (entryMatchesChannel(entry, c, true)) {
161
132
  out.push({
162
- scope: "routing.deny.workers.models",
163
- path: "routing.deny.workers.models",
164
- configPath: "routing.deny.workers.models",
133
+ scope: scope.dotted,
134
+ path: scope.dotted,
135
+ configPath: scope.dotted,
165
136
  entry,
166
137
  by: "deny",
167
138
  });
@@ -170,6 +170,11 @@ export declare const HARVESTED_RESULT_SUMMARY = "harvested: the worktree carries
170
170
  export declare function commandsHash(commands: Record<string, string>): string;
171
171
  export interface TipProof {
172
172
  kind: "fresh" | "reused" | "failed" | "incomplete";
173
+ /** Each completed gate keeps its own execution provenance, even in a mixed cycle. */
174
+ gates?: Array<{
175
+ gate: string;
176
+ kind: "fresh" | "reused";
177
+ }>;
173
178
  /** the commit the cycle's start row spoke for */
174
179
  tip?: string;
175
180
  }
@@ -177,7 +182,8 @@ export interface TipProof {
177
182
  * OBS-1077 close rider: what the engagement's LATEST verification cycle proved — its
178
183
  * `tip-verify-start` row and what followed, never a commit comparison. Exactly one kind per close:
179
184
  * fresh (every tip command ran AND passed), reused (an eligible cached cycle carried forward),
180
- * failed, or incomplete (cancelled, cut short, mixed or undelimited). Nothing before the start row
185
+ * failed, or incomplete (cancelled, cut short or undelimited). Completed mixed cycles retain each
186
+ * gate's kind beside the whole-cycle reused kind. Nothing before the start row
181
187
  * is read, so an unfinished cycle inherits nothing from an earlier green one.
182
188
  */
183
189
  export declare function runEndTipProof(events: readonly JournalEvent[]): TipProof;