tickmarkr 2.2.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/README.md +10 -9
  2. package/dist/adapters/model-lints.js +9 -0
  3. package/dist/adapters/pi.d.ts +1 -0
  4. package/dist/adapters/pi.js +15 -1
  5. package/dist/adapters/prompt.js +11 -3
  6. package/dist/adapters/registry.js +13 -4
  7. package/dist/adapters/types.d.ts +23 -1
  8. package/dist/adapters/types.js +43 -2
  9. package/dist/cli/commands/approve.d.ts +3 -7
  10. package/dist/cli/commands/approve.js +26 -20
  11. package/dist/cli/commands/beat.js +7 -4
  12. package/dist/cli/commands/doctor.d.ts +6 -2
  13. package/dist/cli/commands/doctor.js +79 -9
  14. package/dist/cli/commands/init.js +36 -21
  15. package/dist/cli/commands/plan.js +20 -3
  16. package/dist/cli/commands/report.js +37 -1
  17. package/dist/cli/commands/verify.d.ts +5 -0
  18. package/dist/cli/commands/verify.js +142 -25
  19. package/dist/cli/commands/version.d.ts +2 -1
  20. package/dist/cli/commands/version.js +25 -4
  21. package/dist/compile/collateral.js +15 -9
  22. package/dist/compile/native.js +5 -3
  23. package/dist/config/config.js +1 -1
  24. package/dist/drivers/index.d.ts +7 -0
  25. package/dist/drivers/index.js +40 -10
  26. package/dist/drivers/orca.d.ts +41 -1
  27. package/dist/drivers/orca.js +192 -15
  28. package/dist/drivers/subprocess.d.ts +3 -3
  29. package/dist/drivers/subprocess.js +16 -9
  30. package/dist/drivers/types.d.ts +2 -0
  31. package/dist/gates/baseline.d.ts +4 -0
  32. package/dist/gates/baseline.js +68 -16
  33. package/dist/gates/llm.d.ts +7 -1
  34. package/dist/gates/llm.js +66 -35
  35. package/dist/gates/review.d.ts +3 -1
  36. package/dist/gates/review.js +42 -12
  37. package/dist/gates/run-gates.d.ts +6 -0
  38. package/dist/gates/run-gates.js +25 -10
  39. package/dist/gates/verdict-cause.d.ts +6 -2
  40. package/dist/gates/verdict-cause.js +8 -4
  41. package/dist/run/consult.d.ts +7 -0
  42. package/dist/run/consult.js +21 -3
  43. package/dist/run/daemon.d.ts +14 -0
  44. package/dist/run/daemon.js +249 -27
  45. package/dist/run/git.d.ts +1 -0
  46. package/dist/run/git.js +4 -0
  47. package/dist/run/journal.d.ts +15 -2
  48. package/dist/run/journal.js +70 -12
  49. package/dist/run/supervision.d.ts +6 -0
  50. package/dist/run/supervision.js +29 -1
  51. package/dist/tui/ink/init-app.js +4 -4
  52. package/package.json +1 -1
  53. package/skills/tickmarkr-loop/SKILL.md +1 -0
  54. package/skills/tickmarkr-overseer/SKILL.md +88 -18
  55. package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
  56. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
  57. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
  58. package/skills/tickmarkr-overseer/scripts/watch-context.sh +35 -9
  59. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
@@ -54,10 +54,14 @@ export function hasVerdictParticipationWitness(raw, nonce, discriminator) {
54
54
  const discriminatorPattern = new RegExp(String.raw `"${discriminator}"\s*:\s*${valuePattern}\s*[,}]`);
55
55
  return objectPrefixes(raw).some((prefix) => noncePattern.test(prefix) && discriminatorPattern.test(prefix));
56
56
  }
57
- export function classifyVerdictCause(raw, nonce, discriminator) {
57
+ export function classifyVerdictCause(raw, nonce, discriminator, process = {}) {
58
+ if (process.timedOut)
59
+ return "timeout";
58
60
  if (raw.trim().length === 0)
59
61
  return "empty-output";
60
- return hasVerdictParticipationWitness(raw, nonce, discriminator)
61
- ? "malformed-verdict"
62
- : "no-verdict";
62
+ if (hasVerdictParticipationWitness(raw, nonce, discriminator))
63
+ return "malformed-verdict";
64
+ if (process.exitCode !== undefined && process.exitCode !== 0)
65
+ return "startup-failure";
66
+ return "no-verdict";
63
67
  }
@@ -9,6 +9,9 @@ export interface ConsultVerdict {
9
9
  reason?: string;
10
10
  guidance?: string;
11
11
  excludeAdapter?: string;
12
+ adapter?: string;
13
+ model?: string;
14
+ vendor?: string;
12
15
  }
13
16
  export declare function renderRetryGuidance(v: ConsultVerdict): string;
14
17
  export declare function augmentRetryBrief(feedback: string, opts: {
@@ -36,5 +39,9 @@ export declare function consult(d: Dossier, cfg: TickmarkrConfig, adapters: Work
36
39
  runId?: string;
37
40
  channels?: Array<{
38
41
  adapter: string;
42
+ model?: string;
43
+ vendor?: string;
44
+ channel?: "sub" | "api";
45
+ tier?: string;
39
46
  }>;
40
47
  }): Promise<ConsultVerdict>;
@@ -147,6 +147,10 @@ opts = {}) {
147
147
  out = r.stdout + r.stderr;
148
148
  }
149
149
  else {
150
+ // Preserve the adapter contract for CLIs that cannot seed their TUI: supported adapters use
151
+ // the interactive form, while a declared null keeps the existing visible print fallback.
152
+ const command = adapter.interactiveCommand(promptFile, seatModel)
153
+ ?? adapter.headlessCommand(promptFile, seatModel);
150
154
  // T8: role-first pane name for fleet visibility (consult · T2); consultSeq stays on the dossier artifact only
151
155
  const slot = await driver.slot(cwd, gatePaneName("consult", d.taskId), {
152
156
  label: `CONSULT ${d.taskId}`,
@@ -158,7 +162,7 @@ opts = {}) {
158
162
  writeFileSync(scriptPath, [
159
163
  "export BASH_SILENCE_DEPRECATION_WARNING=1",
160
164
  bannerShell(),
161
- adapter.headlessCommand(promptFile, seatModel),
165
+ command,
162
166
  gateExitTrailer(nonce),
163
167
  ].join("\n"));
164
168
  try {
@@ -205,6 +209,17 @@ opts = {}) {
205
209
  // the same rule as every prefer entry — and disallowedBy carries the full deny grammar (adapter,
206
210
  // model, or adapter:model), so a model-scoped deny cannot slip past an adapter-id-only read.
207
211
  const allowedSeats = seats.filter((s) => disallowedBy(s, cfg.routing, "consult") === null);
212
+ const seatIdentity = (seat) => {
213
+ let vendor = opts.channels?.find((candidate) => candidate.adapter === seat.adapter && candidate.model === seat.model)?.vendor;
214
+ if (!vendor) {
215
+ try {
216
+ const adapter = getAdapter(seat.adapter, adapters);
217
+ vendor = adapter.channels(cfg).find((channel) => channel.model === seat.model)?.vendor ?? adapter.vendor;
218
+ }
219
+ catch { /* an unknown adapter still gets explicit unknown provenance */ }
220
+ }
221
+ return { adapter: seat.adapter, model: seat.model, vendor: vendor ?? "unknown" };
222
+ };
208
223
  if (!allowedSeats.length) {
209
224
  const d = disallowedBy(seats[seats.length - 1], cfg.routing, "consult");
210
225
  return {
@@ -216,11 +231,14 @@ opts = {}) {
216
231
  try {
217
232
  const parsed = await invokeSeat(seat.adapter, seat.model, i);
218
233
  if (parsed.verdict)
219
- return parsed.verdict;
234
+ return { ...parsed.verdict, ...seatIdentity(seat) };
220
235
  }
221
236
  catch {
222
237
  // failed seat (unknown adapter, dead driver/pane, shell error) — fall to the next entry
223
238
  }
224
239
  }
225
- return { action: "human", notes: "consult verdict unparseable — failing safe to human" };
240
+ return {
241
+ action: "human",
242
+ notes: "consult verdict unparseable — failing safe to human",
243
+ };
226
244
  }
@@ -1,5 +1,6 @@
1
1
  import { type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type ModeResolution, type RoutingMode, type TickmarkrConfig } from "../config/config.js";
3
+ import { type DriverChoice } from "../drivers/index.js";
3
4
  import { type ExecutorDriver } from "../drivers/types.js";
4
5
  import { type Baseline } from "../gates/baseline.js";
5
6
  import type { GateResult } from "../gates/types.js";
@@ -13,6 +14,7 @@ export interface RunOptions {
13
14
  retryFailed?: boolean;
14
15
  concurrency?: number;
15
16
  driver?: ExecutorDriver;
17
+ driverOverride?: DriverChoice;
16
18
  adapters?: WorkerAdapter[];
17
19
  globalDir?: string;
18
20
  mode?: RoutingMode;
@@ -100,6 +102,8 @@ export declare const gateSatisfied: (g: GateResult) => boolean;
100
102
  * A round is the gate-result span opened by each `gates` phase-start, per task.
101
103
  */
102
104
  export declare function decisiveReviewRounds(events: JournalEvent[]): JournalEvent[];
105
+ export declare const SUITE_POLL_MS = 250;
106
+ export declare const APPROVAL_POLL_MS = 250;
103
107
  export declare const EARLY_LAUNCH_LIVENESS_MS = 60000;
104
108
  /** Test seam — lowers the empty-pane liveness window without sleeping 60s per case. */
105
109
  export declare function setEarlyLaunchLivenessMsForTests(ms: number): void;
@@ -139,6 +143,16 @@ export declare function verifyIntegrationTipCached(intWt: string, commands: Reco
139
143
  lastMergedTask?: string;
140
144
  baseline?: Baseline;
141
145
  }): Promise<boolean>;
146
+ type SuitePidProbe = (pid: number) => number | undefined;
147
+ /** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership
148
+ * rules remain testable on hosts that forbid process inspection; production supplies cwd and the
149
+ * inherited TICKMARKR_SUITE_PARENT marker from the process itself. */
150
+ export declare function countLiveSuites(snapshot: string, repoRoot: string, daemonPid?: number, cwdForPid?: (pid: number) => string | undefined, suiteParentForPid?: SuitePidProbe): number;
151
+ export declare const setLiveSuiteCountForTests: (probe: (repoRoot: string) => Promise<number>) => void;
152
+ export declare const resetLiveSuiteCountForTests: () => void;
153
+ /** Live full-suite roots attributable to this repository or this daemon. Ancestors are excluded so
154
+ * a daemon invoked by vitest does not wait on its own test harness forever. */
155
+ export declare function liveSuiteCount(repoRoot: string): Promise<number>;
142
156
  /** Test seam — exercise the production observer's total read bound with a small real tree. */
143
157
  export declare function setObserveBudgetBytesForTests(bytes: number): void;
144
158
  export declare function resetObserveBudgetBytesForTests(): void;
@@ -12,6 +12,7 @@ import { bannerShell, paneDispatchCommand } from "../brand.js";
12
12
  import { collateralHits } from "../compile/collateral.js";
13
13
  import { DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
14
14
  import { DeliveryReadinessError } from "../drivers/herdr.js";
15
+ import { driverEvidence } from "../drivers/index.js";
15
16
  import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
16
17
  import { formatOwnedName } from "../drivers/types.js";
17
18
  import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
@@ -19,11 +20,12 @@ import { runGates } from "../gates/run-gates.js";
19
20
  import { filesGlob } from "../graph/files-glob.js";
20
21
  import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
21
22
  import { GATE_NAMES } from "../graph/schema.js";
23
+ import { distFingerprint } from "../cli/commands/version.js";
22
24
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
23
25
  import { runEnvironment } from "./environment.js";
24
- import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
26
+ import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, SUITE_PARENT_ENV, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
25
27
  import { runInteractiveSeed } from "./interactive-seed.js";
26
- import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, renderStructuredReviewFinding, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
28
+ import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingConsultGuidance, outstandingReviewFindings, pendingRechecks, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, RECHECK_RELEASE, renderStructuredReviewFinding, repairReachSinceApproval, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
27
29
  import { isDiffCapPark } from "../gates/review.js";
28
30
  import { acquireApprovalSerialization, acquireRunLock, isPidLive, releaseRunLock } from "./lock.js";
29
31
  import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
@@ -67,12 +69,13 @@ export function resolveRunMode(repoRoot, opts = {}) {
67
69
  return { cfg: resolved.cfg, mode: resolved.mode, source, ...(conflict ? { conflict } : {}) };
68
70
  }
69
71
  // T14: the events that prove an approval was ENACTED — narrowly causal, never merely subsequent.
70
- // An ordinary approval (no release, attempt-cap, recheck, review-upheld) all buy a WORKER, so the
71
- // proof is a dispatch. Generic terminal events are deliberately NOT proof: an approved human-gate
72
+ // Ordinary, attempt-cap and review-upheld approvals buy a WORKER, so their proof is a dispatch;
73
+ // recheck has its own battery enactment below. Generic terminal events are deliberately NOT proof: an approved human-gate
72
74
  // task can fail in routing before task-dispatch, and the catch appends task-failed — treating that
73
75
  // as enactment reports "complete" over an approval that never ran (the exact silent-completion this
74
76
  // task exists to kill).
75
77
  const DISPATCH_ENACTMENT = new Set(["task-dispatch", "repair-dispatch"]);
78
+ const RECHECK_ENACTMENT = "recheck-battery";
76
79
  // The ONE approval that enacts without buying a worker: GATE_SATISFIED_RELEASE resumes from the
77
80
  // persisted task branch after the approved gate (execTask's satisfiedGate branch), whose first act
78
81
  // for the task is worktree-recreation. That event is causal for this path, not incidental.
@@ -95,9 +98,13 @@ export function outstandingApprovals(events) {
95
98
  newest.set(e.taskId, i); });
96
99
  return [...newest]
97
100
  .filter(([taskId, i]) => {
98
- const noWorker = events[i].data.release === GATE_SATISFIED_RELEASE;
101
+ const release = events[i].data.release;
102
+ const noWorker = release === GATE_SATISFIED_RELEASE;
103
+ const recheck = release === RECHECK_RELEASE;
99
104
  return !events.slice(i + 1).some((e) => e.taskId === taskId
100
- && (DISPATCH_ENACTMENT.has(e.event) || (noWorker && e.event === GATE_SATISFIED_ENACTMENT)));
105
+ && (DISPATCH_ENACTMENT.has(e.event)
106
+ || (noWorker && e.event === GATE_SATISFIED_ENACTMENT)
107
+ || (recheck && e.event === RECHECK_ENACTMENT)));
101
108
  })
102
109
  .map(([taskId]) => taskId)
103
110
  .sort();
@@ -163,6 +170,19 @@ const isOracleFailure = (g) => g.details.startsWith("oracle failed:");
163
170
  */
164
171
  export const gateSatisfied = (g) => (g.pass || g.meta?.skipped === true) && g.meta?.infra !== true;
165
172
  const gateFailed = (g) => !gateSatisfied(g);
173
+ const SIGNAL_EXIT_RE = /\b(?:SIGTERM|SIGKILL|signal\s+(?:9|15)|exit(?:s|ed|\s+code)?\s+(?:137|143))\b/i;
174
+ const FAILURE_IDENTITY_RE = /\b(?:AssertionError|FAIL\s+\S|Tests?\s+\d+\s+failed|expected\s+.+\s+to\s+)\b/i;
175
+ /** A signalled test runner with no failure identity produced no verdict about HEAD. Baseline owns the
176
+ * ordinary infra vocabulary; this daemon-only rider handles the signal-shaped non-verdict before its
177
+ * journal row and repair accounting are written. */
178
+ function classifySignalOnlyTest(g) {
179
+ if (g.gate !== "test" || g.pass || g.meta?.infra === true || !SIGNAL_EXIT_RE.test(g.details))
180
+ return;
181
+ const named = Array.isArray(g.meta?.failingTests) && g.meta.failingTests.length > 0;
182
+ if (named || FAILURE_IDENTITY_RE.test(g.details))
183
+ return;
184
+ g.meta = { ...g.meta, classification: "infra", infra: true, retryable: false, kind: "signal-exit" };
185
+ }
166
186
  // v1.85 T3: the gates whose failure IS a deterministic measurement — a machine re-ran a command over a
167
187
  // tree and printed the same bytes. Those are the failures the fingerprint cap governs (the ruling names
168
188
  // it a "deterministic-gate" cap): a third identical answer to a question already answered twice is the
@@ -268,6 +288,8 @@ function approvedReviewRoundCeiling(events, taskId) {
268
288
  return undefined;
269
289
  }
270
290
  const BLOCKED_POLL_MS = 30_000; // between trailer-wait slices, check whether the pane is blocked on a prompt
291
+ export const SUITE_POLL_MS = 250;
292
+ export const APPROVAL_POLL_MS = 250;
271
293
  const PROVIDER_DEATH_REQUEUE_CAP = 2; // v1.46 T1: requeue same assignment twice, then fall through to the normal ladder
272
294
  const PROVIDER_DEATH_BACKOFF_MS = 500; // short backoff before provider-death requeue
273
295
  const NO_TRAILER_DEMOTION_STREAK = 2; // OBS-57: consecutive no-trailer windows demote a channel for the rest of the run
@@ -556,6 +578,106 @@ async function observeWorkerProcessTree(marker, cwd) {
556
578
  }
557
579
  return tree.size === 0 ? "empty" : "running";
558
580
  }
581
+ const SUITE_COMMAND_RE = /(?:^|[\s/])(vitest(?:\.mjs)?|jest|mocha)(?:[\s/]|$)|\bnpm(?:\s+run)?\s+test\b/i;
582
+ function processCwd(pid) {
583
+ try {
584
+ return realpathSync(readlinkSync(`/proc/${pid}/cwd`));
585
+ }
586
+ catch { /* Darwin has no /proc */ }
587
+ try {
588
+ const out = execFileSync("lsof", ["-a", "-p", String(pid), "-d", "cwd", "-Fn"], {
589
+ encoding: "utf8", stdio: ["ignore", "pipe", "ignore"], timeout: 5_000,
590
+ });
591
+ const path = out.split("\n").find((line) => line.startsWith("n"))?.slice(1);
592
+ return path ? realpathSync(path) : undefined;
593
+ }
594
+ catch {
595
+ return undefined;
596
+ }
597
+ }
598
+ function processSuiteParent(pid) {
599
+ try {
600
+ const env = readFileSync(`/proc/${pid}/environ`, "utf8").split("\0");
601
+ const value = env.find((entry) => entry.startsWith(`${SUITE_PARENT_ENV}=`))?.slice(SUITE_PARENT_ENV.length + 1);
602
+ return value && /^\d+$/.test(value) ? Number(value) : undefined;
603
+ }
604
+ catch { /* Darwin has no /proc process environments */ }
605
+ try {
606
+ const out = execFileSync("ps", ["eww", "-p", String(pid), "-o", "command="], {
607
+ encoding: "utf8", stdio: ["ignore", "pipe", "ignore"], timeout: 5_000,
608
+ });
609
+ const value = new RegExp(`(?:^|\\s)${SUITE_PARENT_ENV}=(\\d+)(?:\\s|$)`).exec(out)?.[1];
610
+ return value ? Number(value) : undefined;
611
+ }
612
+ catch {
613
+ return undefined;
614
+ }
615
+ }
616
+ const pathAtOrBelow = (root, candidate) => {
617
+ const rel = relative(root, candidate);
618
+ return rel === "" || (rel !== ".." && !rel.startsWith(`..${sep}`) && !isAbsolute(rel));
619
+ };
620
+ /** Count full-suite roots in one process-table snapshot. The probes are arguments so the ownership
621
+ * rules remain testable on hosts that forbid process inspection; production supplies cwd and the
622
+ * inherited TICKMARKR_SUITE_PARENT marker from the process itself. */
623
+ export function countLiveSuites(snapshot, repoRoot, daemonPid = process.pid, cwdForPid = processCwd, suiteParentForPid = processSuiteParent) {
624
+ const rows = [];
625
+ for (const line of snapshot.split("\n")) {
626
+ const match = /^\s*(\d+)\s+(\d+)\s+(\S+)\s+(.*)$/.exec(line);
627
+ if (match && !match[3].startsWith("Z")) {
628
+ rows.push({ pid: Number(match[1]), ppid: Number(match[2]), command: match[4] });
629
+ }
630
+ }
631
+ const byPid = new Map(rows.map((row) => [row.pid, row]));
632
+ const ancestors = new Set();
633
+ for (let pid = daemonPid; pid && !ancestors.has(pid); pid = byPid.get(pid)?.ppid ?? 0)
634
+ ancestors.add(pid);
635
+ const descendants = new Set([daemonPid]);
636
+ for (let grew = true; grew;) {
637
+ grew = false;
638
+ for (const row of rows)
639
+ if (!descendants.has(row.pid) && descendants.has(row.ppid)) {
640
+ descendants.add(row.pid);
641
+ grew = true;
642
+ }
643
+ }
644
+ const root = realpathSync(repoRoot);
645
+ const candidates = rows.filter((row) => !ancestors.has(row.pid) && SUITE_COMMAND_RE.test(row.command));
646
+ const attributable = new Set(candidates.filter((row) => {
647
+ if (descendants.has(row.pid))
648
+ return true;
649
+ const cwd = cwdForPid(row.pid);
650
+ if (cwd !== undefined && pathAtOrBelow(root, cwd))
651
+ return true;
652
+ const suiteParent = suiteParentForPid(row.pid);
653
+ if (suiteParent === daemonPid)
654
+ return true;
655
+ const parentCwd = suiteParent === undefined ? undefined : cwdForPid(suiteParent);
656
+ return parentCwd !== undefined && pathAtOrBelow(root, parentCwd);
657
+ }).map((row) => row.pid));
658
+ // npm/npx + vitest + pool workers are one suite. Count only attributable suite processes with no
659
+ // attributable suite ancestor, while still following ordinary non-suite parents between them.
660
+ return [...attributable].filter((pid) => {
661
+ for (let parent = byPid.get(pid)?.ppid; parent; parent = byPid.get(parent)?.ppid) {
662
+ if (attributable.has(parent))
663
+ return false;
664
+ }
665
+ return true;
666
+ }).length;
667
+ }
668
+ let liveSuiteCountForTests;
669
+ export const setLiveSuiteCountForTests = (probe) => {
670
+ liveSuiteCountForTests = probe;
671
+ };
672
+ export const resetLiveSuiteCountForTests = () => { liveSuiteCountForTests = undefined; };
673
+ /** Live full-suite roots attributable to this repository or this daemon. Ancestors are excluded so
674
+ * a daemon invoked by vitest does not wait on its own test harness forever. */
675
+ export async function liveSuiteCount(repoRoot) {
676
+ if (liveSuiteCountForTests)
677
+ return liveSuiteCountForTests(repoRoot);
678
+ const snapshot = await shGit("ps -Aww -o pid=,ppid=,state=,command=", repoRoot, 15_000);
679
+ return snapshot.code === 0 ? countLiveSuites(snapshot.stdout, repoRoot) : 0;
680
+ }
559
681
  const OBSERVE_CHUNK_BYTES = 64 * 1024;
560
682
  const OBSERVE_BUDGET_BYTES = 256 * 1024 * 1024;
561
683
  let observeBudgetBytes = OBSERVE_BUDGET_BYTES;
@@ -1255,11 +1377,25 @@ export async function runDaemon(repoRoot, opts = {}) {
1255
1377
  baseRef = await gitHead(repoRoot);
1256
1378
  baseline = await captureBaseline(repoRoot, commands);
1257
1379
  writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(baseline, null, 2));
1380
+ writeFileSync(join(journal.dir, "graph.json"), readFileSync(join(tickmarkrDir(repoRoot), "graph.json")));
1258
1381
  // v1.70 T2: environment identity beside the graph/branch identity — running tickmarkr version,
1259
1382
  // loaded-config hash, and the probed CLI version of each adapter holding a channel in the run,
1260
1383
  // gathered through the existing probe/config-load paths (no second mechanism).
1261
1384
  const environment = runEnvironment(cfg, channels, health);
1262
- journal.append("run-start", undefined, { pid: process.pid, baseRef, commands, channels: channels.map(channelKey), branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source, environment, ...(prior ? { supersedes: prior.runId } : {}) }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
1385
+ journal.append("run-start", undefined, {
1386
+ pid: process.pid, baseRef, commands, channels: channels.map(channelKey),
1387
+ channelsByRole: {
1388
+ worker: pools.worker.map(channelKey),
1389
+ judge: pools.judge.map(channelKey),
1390
+ review: pools.review.map(channelKey),
1391
+ consult: pools.consult.map(channelKey),
1392
+ },
1393
+ driver: driver.id,
1394
+ driverEvidence: driverEvidence(cfg, driver, opts.driverOverride),
1395
+ distFingerprint: distFingerprint(),
1396
+ branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source,
1397
+ environment, ...(prior ? { supersedes: prior.runId } : {}),
1398
+ }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
1263
1399
  runStarted = true;
1264
1400
  // v1.53 T5: mark the prior run AFTER this run's run-start exists, so the prior journal never
1265
1401
  // names a successor that has no journal. Append-only — the prior journal is never rewritten.
@@ -1370,6 +1506,41 @@ export async function runDaemon(repoRoot, opts = {}) {
1370
1506
  mergeChain = next.catch(() => undefined);
1371
1507
  return next;
1372
1508
  };
1509
+ // OBS-829/OBS-854: one full-suite verdict round at a time in this run, and do not begin beside an
1510
+ // externally live suite attributable to this repository. The process scan catches nested scratch
1511
+ // suites through daemon parentage even after cwd stops naming a worktree.
1512
+ let suiteChain = Promise.resolve();
1513
+ let suitePending = 0;
1514
+ const withSuiteWindow = async (taskId, enabled, run) => {
1515
+ if (!enabled)
1516
+ return run();
1517
+ const previous = suiteChain;
1518
+ let release;
1519
+ suiteChain = new Promise((resolve) => { release = resolve; });
1520
+ const queued = suitePending++ > 0;
1521
+ if (queued) {
1522
+ const count = Math.max(1, await liveSuiteCount(repoRoot));
1523
+ journal.append("suite-wait", taskId, { count });
1524
+ }
1525
+ await previous;
1526
+ try {
1527
+ let lastCount = -1;
1528
+ for (;;) {
1529
+ const count = await liveSuiteCount(repoRoot);
1530
+ if (count === 0)
1531
+ break;
1532
+ if (count !== lastCount)
1533
+ journal.append("suite-wait", taskId, { count });
1534
+ lastCount = count;
1535
+ await new Promise((wake) => setTimeout(wake, SUITE_POLL_MS));
1536
+ }
1537
+ return await run();
1538
+ }
1539
+ finally {
1540
+ suitePending--;
1541
+ release();
1542
+ }
1543
+ };
1373
1544
  // gateFails/consults are execTask-scoped counters passed in so a park row is a rich verified-failure
1374
1545
  // observation (e.g. ladder-exhausted + gateFails:4); every task-human row has a closed kind, never prose alone.
1375
1546
  const gateFailApprovalReason = (taskId, identity, includeUphold = false) => {
@@ -1786,6 +1957,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1786
1957
  const applyVerdict = async (v, attempts, trigger) => {
1787
1958
  journal.append("consult-verdict", t.id, {
1788
1959
  action: v.action, notes: v.notes,
1960
+ adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
1789
1961
  ...(v.reason ? { reason: v.reason } : {}),
1790
1962
  ...(v.guidance ? { guidance: v.guidance } : {}),
1791
1963
  ...(v.excludeAdapter ? { excludeAdapter: v.excludeAdapter } : {}),
@@ -1836,21 +2008,24 @@ export async function runDaemon(repoRoot, opts = {}) {
1836
2008
  // contiguous green prefix whose recorded commit is still the task branch tip.
1837
2009
  const satisfiedGate = satisfiedGates.get(t.id);
1838
2010
  const replayedGates = replayedGateResults.get(t.id);
1839
- resumeGateReplay: if (satisfiedGate || replayedGates) {
2011
+ const recheck = pendingRechecks(journal.read()).has(t.id);
2012
+ resumeGateReplay: if (satisfiedGate || replayedGates || recheck) {
1840
2013
  const taskBase = await integrationHead(intWt);
1841
2014
  const taskBranch = `${branch}--${t.id}`;
1842
2015
  const priorWt = worktreePath(repoRoot, taskBranch);
1843
2016
  if (!existsSync(priorWt)) {
1844
- if (satisfiedGate)
1845
- throw new Error(`approved gate ${satisfiedGate} cannot resume: task worktree is missing`);
2017
+ if (satisfiedGate || recheck)
2018
+ throw new Error(`${recheck ? "recheck" : `approved gate ${satisfiedGate}`} cannot resume: task worktree is missing`);
1846
2019
  // Observed passes are an optimization, never authority: without the task worktree there is no
1847
2020
  // commit to compare and no landed work to gate, so fall through to the ordinary worker path.
1848
2021
  replayedGateResults.delete(t.id);
1849
2022
  break resumeGateReplay;
1850
2023
  }
1851
- const resumeReason = satisfiedGate
1852
- ? `approved gate ${satisfiedGate}`
1853
- : `recorded gates on ${replayedGates.commit.slice(0, 10)}`;
2024
+ const resumeReason = recheck
2025
+ ? "operator recheck"
2026
+ : satisfiedGate
2027
+ ? `approved gate ${satisfiedGate}`
2028
+ : `recorded gates on ${replayedGates.commit.slice(0, 10)}`;
1854
2029
  const priorTaskTip = await gitHead(priorWt);
1855
2030
  const priorTaskSubject = await gateCommitSubject(taskBase, priorTaskTip, priorWt);
1856
2031
  const commitsToCarry = await commitsAheadOf(taskBase, priorWt);
@@ -1902,7 +2077,10 @@ export async function runDaemon(repoRoot, opts = {}) {
1902
2077
  : [],
1903
2078
  raw: "",
1904
2079
  };
1905
- const gateAuthor = rs?.lastAssignment ?? assignment;
2080
+ const parkedAuthor = recheck
2081
+ ? [...journal.read()].reverse().find((e) => e.event === "task-dispatch" && e.taskId === t.id)?.data.assignment
2082
+ : undefined;
2083
+ const gateAuthor = parkedAuthor ?? rs?.lastAssignment ?? assignment;
1906
2084
  const satisfiedIndex = satisfiedGate ? GATE_NAMES.indexOf(satisfiedGate) : -1;
1907
2085
  // The serial pipeline could have at most one blocking result, so "everything after the
1908
2086
  // approved gate" was enough. v1.85 can record both verdict siblings red in one round, and a
@@ -1926,7 +2104,10 @@ export async function runDaemon(repoRoot, opts = {}) {
1926
2104
  priorResults.set(e.data.gate, e);
1927
2105
  }
1928
2106
  let remainingGates;
1929
- if (satisfiedGate) {
2107
+ if (recheck) {
2108
+ remainingGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
2109
+ }
2110
+ else if (satisfiedGate) {
1930
2111
  remainingGates = t.gates.filter((gate) => {
1931
2112
  if (gate === satisfiedGate)
1932
2113
  return false;
@@ -1984,10 +2165,10 @@ export async function runDaemon(repoRoot, opts = {}) {
1984
2165
  // This suffix is re-measured to decide whether resume may advance, but the interrupted
1985
2166
  // attempt already paid for its red result. The next worker-backed round remains the next
1986
2167
  // deterministic-fingerprint occurrence/review round for budget accounting.
1987
- ...(!satisfiedGate ? { replayMeasurement: true } : {}),
2168
+ ...(!satisfiedGate && !recheck ? { replayMeasurement: true } : {}),
1988
2169
  };
1989
2170
  journal.phaseStart(t.id, "gates");
1990
- const { results } = await runGates(resumedTask, {
2171
+ const { results } = await withSuiteWindow(t.id, resumedTask.gates.includes("test") && commands.test !== undefined, () => runGates(resumedTask, {
1991
2172
  worktree: wt, baseRef: taskBase, result: priorResult, author: gateAuthor,
1992
2173
  commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
1993
2174
  collateral: collateral.get(t.id) ?? [],
@@ -2009,7 +2190,12 @@ export async function runDaemon(repoRoot, opts = {}) {
2009
2190
  journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
2010
2191
  return;
2011
2192
  }
2193
+ if (e.phase === "note") {
2194
+ journal.append(e.name, t.id, e.payload);
2195
+ return;
2196
+ }
2012
2197
  const g = e.result;
2198
+ classifySignalOnlyTest(g);
2013
2199
  inParallelOrder(g.gate, () => {
2014
2200
  journalGateResult(g);
2015
2201
  noteReviewRetry(g);
@@ -2019,7 +2205,15 @@ export async function runDaemon(repoRoot, opts = {}) {
2019
2205
  }
2020
2206
  });
2021
2207
  },
2022
- });
2208
+ }));
2209
+ results.forEach(classifySignalOnlyTest);
2210
+ if (pendingRechecks(journal.read()).has(t.id)) {
2211
+ journal.append("recheck-battery", t.id, {
2212
+ commit: gateSubject.commit,
2213
+ gates: resumedTask.gates,
2214
+ pass: results.every(gateSatisfied),
2215
+ });
2216
+ }
2023
2217
  const approvedCommits = await commitsAheadOf(taskBase, wt);
2024
2218
  graph = addEvidence(graph, t.id, { commits: approvedCommits, gateResults: results });
2025
2219
  saveGraph(repoRoot, graph);
@@ -2354,7 +2548,14 @@ export async function runDaemon(repoRoot, opts = {}) {
2354
2548
  const adapter = getAdapter(assignment.adapter, adapters);
2355
2549
  // VIS-04: workers share one role tab. T2: `owned` names the pane canonically (ownership contract);
2356
2550
  // the legacy name stays the fallback for drivers without owned handling (subprocess spies).
2357
- const slot = await trackedDriver.slot(wt, `${t.id}-worker-${assignment.adapter}-a${attempt}-${runTag}`, { group: "workers", owned: { role: "worker", taskId: t.id, attempt, runId } });
2551
+ const workerSlotOpts = {
2552
+ group: "workers",
2553
+ owned: { role: "worker", taskId: t.id, attempt, runId },
2554
+ };
2555
+ // Keep the established enumerable slot-options shape consumed by legacy drivers while making
2556
+ // the adapter hook available as an own request field to execution surfaces such as Orca.
2557
+ Object.defineProperty(workerSlotOpts, "agent", { value: assignment.adapter });
2558
+ const slot = await trackedDriver.slot(wt, `${t.id}-worker-${assignment.adapter}-a${attempt}-${runTag}`, workerSlotOpts);
2358
2559
  const sessionId = retryMode === "resume" ? priorSession.id : slot.name;
2359
2560
  const icmd = retryMode === "resume"
2360
2561
  ? adapter.resumeCommand(sessionId, promptFile, assignment.model)
@@ -3452,7 +3653,12 @@ export async function runDaemon(repoRoot, opts = {}) {
3452
3653
  journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
3453
3654
  return;
3454
3655
  }
3656
+ if (e.phase === "note") {
3657
+ journal.append(e.name, t.id, e.payload);
3658
+ return;
3659
+ }
3455
3660
  const g = e.result;
3661
+ classifySignalOnlyTest(g);
3456
3662
  inParallelOrder(g.gate, () => {
3457
3663
  // GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
3458
3664
  // which channel flaked, which channel retried — so `tickmarkr journal`/report can distinguish "judge
@@ -3485,7 +3691,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3485
3691
  const gated = await gitHead(wt);
3486
3692
  gateSubject = { commit: await gateCommitSubject(taskBase, gated, wt), attempt };
3487
3693
  journal.phaseStart(t.id, "gates");
3488
- ({ results, commits } = await runGates(t, {
3694
+ ({ results, commits } = await withSuiteWindow(t.id, t.gates.includes("test") && commands.test !== undefined, () => runGates(t, {
3489
3695
  worktree: wt, baseRef: taskBase, result, author: assignment,
3490
3696
  commands, baseline, channels: pools.review, judgeChannels: pools.judge, adapters, cfg, artifactDir: journal.dir,
3491
3697
  collateral: collateral.get(t.id) ?? [],
@@ -3508,7 +3714,8 @@ export async function runDaemon(repoRoot, opts = {}) {
3508
3714
  : undefined,
3509
3715
  excludeReviewers: badReviewers,
3510
3716
  onGate,
3511
- }));
3717
+ })));
3718
+ results.forEach(classifySignalOnlyTest);
3512
3719
  graph = addEvidence(graph, t.id, { commits, gateResults: results, artifacts: [promptFile] });
3513
3720
  saveGraph(repoRoot, graph);
3514
3721
  if (results.some((g) => g.gate === "test" && !g.pass))
@@ -3605,6 +3812,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3605
3812
  // failing gate, and `feedback` still carries every one of them to the next attempt.
3606
3813
  const repairBattery = failing.some((g) => g.gate !== "review") ? failing.filter((g) => g.gate !== "review") : failing;
3607
3814
  const repairable = narrowRepairBattery(repairBattery) && lostCommits.length === 0 && landed.length > 0;
3815
+ const repairHistory = repairReachSinceApproval(journal.read(), t.id);
3608
3816
  const repairsDrawn = repairsSinceApproval(journal.read(), t.id);
3609
3817
  const repair = repairable && repairsDrawn < MAX_REPAIRS;
3610
3818
  // v1.85 T3: the fingerprint cap. Two normalized-identical failures of one DETERMINISTIC gate on
@@ -3654,7 +3862,11 @@ export async function runDaemon(repoRoot, opts = {}) {
3654
3862
  continue;
3655
3863
  return;
3656
3864
  }
3657
- journal.append("consult-verdict", t.id, { action: v.action, notes: v.notes, capAdvisory: true });
3865
+ journal.append("consult-verdict", t.id, {
3866
+ action: v.action, notes: v.notes,
3867
+ adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
3868
+ capAdvisory: true,
3869
+ });
3658
3870
  }
3659
3871
  // v1.85 T3: a narrow battery over fully carried commits earns a REPAIR (decided above) — the
3660
3872
  // next dispatch carries the findings verbatim and the diff content instead of re-onboarding a
@@ -3670,13 +3882,17 @@ export async function runDaemon(repoRoot, opts = {}) {
3670
3882
  const repairExhausted = repairable && !repair;
3671
3883
  if (repair && !capStep) {
3672
3884
  journal.append("repair-attempt", t.id, {
3673
- repair: repairsDrawn + 1, of: MAX_REPAIRS, gates: failing.map((g) => g.gate),
3885
+ repair: repairHistory.length + 1, charge: repairsDrawn + 1, of: MAX_REPAIRS,
3886
+ gates: failing.map((g) => g.gate),
3674
3887
  commits: landed.length,
3675
3888
  findings: feedback, // the failure bytes this repair must carry, replayable across a resume
3676
3889
  });
3677
3890
  }
3678
3891
  else if (repairExhausted && !capStep) {
3679
- journal.append("repair-exhausted", t.id, { repairs: repairsDrawn, of: MAX_REPAIRS, gates: failing.map((g) => g.gate) });
3892
+ journal.append("repair-exhausted", t.id, {
3893
+ repairs: repairsDrawn, of: MAX_REPAIRS, gates: failing.map((g) => g.gate),
3894
+ reached: repairHistory,
3895
+ });
3680
3896
  }
3681
3897
  const step = capStep ?? (repair || (reviewFixRetry && !repairExhausted)
3682
3898
  ? "retry"
@@ -3685,7 +3901,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3685
3901
  journal.append("escalation", t.id, {
3686
3902
  step, attempt: attempt + 1,
3687
3903
  ...(reviewFixRetry && !repairExhausted ? { reviewFix: true } : {}),
3688
- ...(repair ? { repair: repairsDrawn + 1 } : {}),
3904
+ ...(repair ? { repair: repairHistory.length + 1, repairCharge: repairsDrawn + 1 } : {}),
3689
3905
  });
3690
3906
  await driver.notify(`tickmarkr ${runId}: ${t.id} escalation: ${step}`, { tier: "attention" });
3691
3907
  }
@@ -3740,7 +3956,13 @@ export async function runDaemon(repoRoot, opts = {}) {
3740
3956
  }
3741
3957
  if (inflight.size === 0)
3742
3958
  break;
3743
- await Promise.race([...inflight.values(), aborted]); // aborted rejects on termination — unwinds the run
3959
+ const waiters = [...inflight.values(), aborted];
3960
+ // A free slot is itself a scheduling boundary: poll the append-only approval stream instead of
3961
+ // sleeping until an unrelated long-running task settles.
3962
+ if (inflight.size < concurrency) {
3963
+ waiters.push(new Promise((wake) => setTimeout(wake, APPROVAL_POLL_MS)));
3964
+ }
3965
+ await Promise.race(waiters); // aborted rejects on termination — unwinds the run
3744
3966
  }
3745
3967
  // D-07: the sweep now closes only what's LEFT in keptSlots — done-closed worker slots were removed
3746
3968
  // (no double-close) and self-cleaned LLM/consult panes were never added under keepLlm:false. This
@@ -3765,7 +3987,7 @@ export async function runDaemon(repoRoot, opts = {}) {
3765
3987
  // OBS-34: post-merge integration-tip verify — strict exit codes, no baseline forgiveness.
3766
3988
  const lastMergedTask = [...journal.read()].reverse().find((e) => e.event === "merge" && e.taskId)?.taskId;
3767
3989
  if (summary.done.length > 0 && Object.keys(commands).length > 0) {
3768
- const tipFailed = await verifyIntegrationTipCached(intWt, commands, journal, { lastMergedTask, baseline });
3990
+ const tipFailed = await withSuiteWindow(undefined, commands.test !== undefined, () => verifyIntegrationTipCached(intWt, commands, journal, { lastMergedTask, baseline }));
3769
3991
  summary.tipVerify = tipFailed ? "failed" : "passed";
3770
3992
  if (tipFailed && lastMergedTask)
3771
3993
  summary.lastMergedTask = lastMergedTask;
package/dist/run/git.d.ts CHANGED
@@ -2,6 +2,7 @@ import { spawn } from "node:child_process";
2
2
  import { ROUTING_ENV_SEAMS } from "../route/router.js";
3
3
  export { ROUTING_ENV_SEAMS };
4
4
  export declare const FORK_CAP_ENV = "VITEST_MAX_FORKS";
5
+ export declare const SUITE_PARENT_ENV = "TICKMARKR_SUITE_PARENT";
5
6
  export declare const DEFAULT_FORK_CAP = "6";
6
7
  /**
7
8
  * OBS-618: how many PROCESSES one vitest fork can hold at its peak — the fork, a daemon it spawns,