tickmarkr 2.3.0 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/dist/adapters/catalog-remote.d.ts +12 -4
  2. package/dist/adapters/catalog-remote.js +97 -45
  3. package/dist/adapters/catalog.js +5 -3
  4. package/dist/adapters/claude-code.d.ts +1 -1
  5. package/dist/adapters/claude-code.js +8 -5
  6. package/dist/adapters/codex.js +6 -7
  7. package/dist/adapters/model-lints.d.ts +9 -5
  8. package/dist/adapters/model-lints.js +56 -15
  9. package/dist/adapters/model-windows.js +11 -0
  10. package/dist/adapters/prompt.js +1 -0
  11. package/dist/adapters/qwen.d.ts +5 -0
  12. package/dist/adapters/qwen.js +153 -0
  13. package/dist/adapters/registry.js +13 -1
  14. package/dist/adapters/types.d.ts +1 -0
  15. package/dist/adapters/types.js +1 -0
  16. package/dist/cli/commands/compile.d.ts +3 -0
  17. package/dist/cli/commands/compile.js +91 -34
  18. package/dist/cli/commands/doctor.d.ts +4 -3
  19. package/dist/cli/commands/doctor.js +28 -9
  20. package/dist/cli/commands/fleet.d.ts +4 -0
  21. package/dist/cli/commands/fleet.js +60 -15
  22. package/dist/cli/commands/init.js +12 -13
  23. package/dist/cli/commands/plan.js +75 -11
  24. package/dist/cli/commands/run.js +20 -1
  25. package/dist/cli/commands/status.d.ts +1 -0
  26. package/dist/cli/commands/status.js +59 -16
  27. package/dist/cli/commands/verify.d.ts +1 -0
  28. package/dist/cli/commands/verify.js +5 -0
  29. package/dist/cli/commands/version.js +2 -2
  30. package/dist/cli/index.d.ts +1 -1
  31. package/dist/cli/index.js +2 -2
  32. package/dist/compile/collateral.d.ts +14 -12
  33. package/dist/compile/collateral.js +32 -33
  34. package/dist/compile/index.d.ts +4 -1
  35. package/dist/compile/index.js +53 -8
  36. package/dist/compile/native.d.ts +4 -2
  37. package/dist/compile/native.js +63 -6
  38. package/dist/compile/ownership.js +41 -10
  39. package/dist/config/config.d.ts +1 -0
  40. package/dist/config/config.js +51 -5
  41. package/dist/drivers/herdr.d.ts +2 -0
  42. package/dist/drivers/herdr.js +43 -4
  43. package/dist/drivers/orca.d.ts +18 -1
  44. package/dist/drivers/orca.js +163 -15
  45. package/dist/drivers/types.d.ts +10 -0
  46. package/dist/gates/baseline.d.ts +26 -2
  47. package/dist/gates/baseline.js +115 -13
  48. package/dist/gates/review.d.ts +6 -4
  49. package/dist/gates/review.js +26 -31
  50. package/dist/gates/run-gates.d.ts +5 -2
  51. package/dist/gates/run-gates.js +34 -17
  52. package/dist/graph/graph.d.ts +20 -0
  53. package/dist/graph/graph.js +66 -1
  54. package/dist/route/preference.d.ts +6 -0
  55. package/dist/route/preference.js +40 -0
  56. package/dist/route/router.js +15 -2
  57. package/dist/run/consult.d.ts +1 -0
  58. package/dist/run/consult.js +35 -7
  59. package/dist/run/daemon.d.ts +9 -0
  60. package/dist/run/daemon.js +266 -59
  61. package/dist/run/git.d.ts +4 -0
  62. package/dist/run/git.js +51 -6
  63. package/dist/run/journal.d.ts +1 -1
  64. package/dist/run/journal.js +5 -2
  65. package/dist/run/lock.d.ts +6 -0
  66. package/dist/run/lock.js +41 -1
  67. package/dist/tui/ink/fleet-app.d.ts +4 -0
  68. package/dist/tui/ink/fleet-app.js +45 -16
  69. package/package.json +59 -1
  70. package/skills/tickmarkr-overseer/SKILL.md +39 -4
  71. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
  72. package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
@@ -2,7 +2,7 @@ import { type ShResult } from "../run/git.js";
2
2
  import { type JournalEvent } from "../run/journal.js";
3
3
  import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "./types.js";
4
4
  /** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
5
- export declare const ORCA_RESPONSE_FAMILIES: readonly ["status", "create", "list", "read", "send", "wait", "show", "close", "worktree-current", "hooks-status"];
5
+ export declare const ORCA_RESPONSE_FAMILIES: readonly ["status", "create", "list", "read", "send", "wait", "show", "close", "worktree-current", "worktree-set", "hooks-status"];
6
6
  export type OrcaFamily = (typeof ORCA_RESPONSE_FAMILIES)[number];
7
7
  export declare const ORCA_FIXTURE_VERSION = "1.4.195";
8
8
  export declare const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
@@ -15,6 +15,8 @@ export declare const NOT_WRITABLE_CODE = "terminal_not_writable";
15
15
  export declare const RUNNING_STATUS = "running";
16
16
  export declare const STATUS_GOVERNED_METHODS: readonly ["read", "waitOutput", "status", "waitAgentStatus"];
17
17
  export declare const WORKTREE_ADOPTION_TIMEOUT_MS = 60000;
18
+ /** A missing slot gets the same bounded chance to appear as a reaped shell gets to settle. */
19
+ export declare const PENDING_PROJECT_GRACE_MS = 2000;
18
20
  export interface OrcaExec {
19
21
  (args: string[], cwd: string, timeoutMs?: number): Promise<ShResult>;
20
22
  }
@@ -118,6 +120,8 @@ export declare class OrcaDriver implements ExecutorDriver {
118
120
  private journalRoots;
119
121
  private narrate?;
120
122
  private hookCoverage?;
123
+ private taskWorktrees;
124
+ private pendingProjects;
121
125
  constructor(opts?: OrcaDriverOpts);
122
126
  private call;
123
127
  /** The live runtime's identity, or an explicit failure. A missing or unreachable runtime is a
@@ -131,7 +135,13 @@ export declare class OrcaDriver implements ExecutorDriver {
131
135
  private state;
132
136
  private latched;
133
137
  private assertAvailable;
138
+ describe(slot: Slot): {
139
+ surface?: string;
140
+ hostPlatform?: string;
141
+ } | undefined;
134
142
  run(slot: Slot, cmd: string): Promise<void>;
143
+ private sendText;
144
+ private sendReceipt;
135
145
  private create;
136
146
  /**
137
147
  * A freshly-created git checkout does not become a valid Orca selector atomically. Ask
@@ -188,6 +198,11 @@ export declare class OrcaDriver implements ExecutorDriver {
188
198
  * only after this method validates all four fields. Any malformed/refused wait remains explicit. */
189
199
  private waitCondition;
190
200
  waitAgentStatus(slot: Slot, status: string, timeoutMs: number): Promise<boolean>;
201
+ sendKey(slot: Slot, key: string): Promise<void>;
202
+ nudge(slot: Slot, message: string): Promise<boolean>;
203
+ narrator(cwd: string, command: string, runId?: string): Promise<Slot>;
204
+ project(taskId: string, state: "in-progress" | "in-review" | "completed"): Promise<void>;
205
+ private setWorkspaceStatus;
191
206
  notify(msg: string, opts?: NotifyOpts): Promise<void>;
192
207
  narrateWith(narrate: (event: JournalEvent) => void): void;
193
208
  close(slot: Slot): Promise<void>;
@@ -208,6 +223,8 @@ export declare class OrcaDriver implements ExecutorDriver {
208
223
  * that still reports truncated at totalCount rows is a listing this sweep declines to judge on.
209
224
  */
210
225
  private listAll;
226
+ private openRunJournal;
227
+ private dropExpiredProjects;
211
228
  /**
212
229
  * Sweep tickmarkr-owned terminals down to `desired`. Ownership is decided ONLY by parseOwnedName
213
230
  * over the owned TAB title, through the same panesToClose fold herdr uses (drivers/types.ts): an
@@ -34,7 +34,7 @@ import { formatOwnedName, panesToClose, parseOwnedName } from "./types.js";
34
34
  /** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
35
35
  export const ORCA_RESPONSE_FAMILIES = [
36
36
  "status", "create", "list", "read", "send", "wait", "show", "close",
37
- "worktree-current", "hooks-status",
37
+ "worktree-current", "worktree-set", "hooks-status",
38
38
  ];
39
39
  export const ORCA_FIXTURE_VERSION = "1.4.195";
40
40
  export const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
@@ -57,6 +57,9 @@ const POLL_MS = 200;
57
57
  export const WORKTREE_ADOPTION_TIMEOUT_MS = 60_000;
58
58
  const WORKTREE_ADOPTION_POLL_MS = 1_000;
59
59
  const WORKTREE_ADOPTION_JOURNAL_MS = 2_000;
60
+ const NUDGE_ECHO_TIMEOUT_MS = 2_000;
61
+ /** A missing slot gets the same bounded chance to appear as a reaped shell gets to settle. */
62
+ export const PENDING_PROJECT_GRACE_MS = 2_000;
60
63
  const SYSTEM_TIME = {
61
64
  now: () => Date.now(),
62
65
  sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
@@ -262,6 +265,8 @@ export class OrcaDriver {
262
265
  journalRoots = new Map();
263
266
  narrate;
264
267
  hookCoverage;
268
+ taskWorktrees = new Map();
269
+ pendingProjects = new Map();
265
270
  constructor(opts = {}) {
266
271
  this.bin = opts.bin ?? resolveOrcaCliBinary(process.cwd(), { env: opts.env, platform: opts.platform }) ?? "orca";
267
272
  // Config values flow into a shell here: every argv element is quoted, always.
@@ -358,6 +363,15 @@ export class OrcaDriver {
358
363
  // receipt, relist, reconcile — is against THIS value.
359
364
  const worktree = canonicalWorktreePath(cwd);
360
365
  this.slots.set(id, { title, cwd: worktree, agent: opts?.agent, buf: "", recoveries: 0, recovering: false });
366
+ const owned = parseOwnedName(title);
367
+ if (owned?.role === "worker") {
368
+ this.taskWorktrees.set(owned.taskId, worktree);
369
+ const pending = this.pendingProjects.get(owned.taskId);
370
+ if (pending) {
371
+ await this.setWorkspaceStatus(worktree, pending.state);
372
+ this.pendingProjects.delete(owned.taskId);
373
+ }
374
+ }
361
375
  return { id, name: title, cwd: worktree, group: opts?.group };
362
376
  }
363
377
  /** Where to invoke the CLI for this slot's calls (see OrcaSlotState.dir). */
@@ -378,6 +392,17 @@ export class OrcaDriver {
378
392
  if (st.unavailable)
379
393
  throw new OrcaUnavailableError(family, st.unavailable, "");
380
394
  }
395
+ describe(slot) {
396
+ const st = this.state(slot);
397
+ // The daemon asks only after run(), but callers probing a lazy slot must see absence rather
398
+ // than an invented placement. The interface's object spread accepts this runtime absence.
399
+ if (!st.handle)
400
+ return undefined;
401
+ return {
402
+ ...(st.surface === undefined ? {} : { surface: st.surface }),
403
+ ...(st.hostPlatform === undefined ? {} : { hostPlatform: st.hostPlatform }),
404
+ };
405
+ }
381
406
  async run(slot, cmd) {
382
407
  const st = this.state(slot);
383
408
  if (!st.handle) {
@@ -387,22 +412,28 @@ export class OrcaDriver {
387
412
  this.assertAvailable("send", st);
388
413
  // Later deliveries go into the terminal this slot already owns — never a second create.
389
414
  // terminalOp proves the runtime binding before the handle goes on the wire.
390
- const env = await this.terminalOp("send", st, (h) => this.call("send", ["terminal", "send", "--terminal", h, "--text", cmd, "--enter"], this.cliCwd(st)), { mutating: true });
391
- // Recorded 1.4.186 send receipt: result.send = {handle, accepted, bytesWritten} — there is no
392
- // `delivered`. `ok:true` alone is not a receipt: only an accepted receipt naming THIS handle
393
- // proves the text was submitted, so anything else is a failed send, never a silent no-op.
415
+ await this.sendText(st, cmd);
416
+ }
417
+ async sendText(st, text) {
418
+ const env = await this.terminalOp("send", st, (h) => this.call("send", ["terminal", "send", "--terminal", h, "--text", text, "--enter"], this.cliCwd(st)), { mutating: true });
419
+ // Recorded 1.4.195 send receipt: result.send = {handle, accepted, bytesWritten}. `ok:true`
420
+ // alone is not delivery: the accepted receipt must name this handle and account for the bytes.
421
+ const receipt = this.sendReceipt(env, st);
422
+ const expectedBytes = Buffer.byteLength(text, "utf8") + 1;
423
+ if (receipt.accepted !== true || typeof receipt.bytesWritten !== "number" || receipt.bytesWritten !== expectedBytes) {
424
+ throw new OrcaError("send", `send receipt does not report expected byte delivery (accepted: ${JSON.stringify(receipt.accepted)}, bytesWritten: ${receipt.bytesWritten}, expected: ${expectedBytes})`, env.raw);
425
+ }
426
+ }
427
+ sendReceipt(env, st) {
394
428
  const receipt = env.result.send;
395
429
  if (typeof receipt !== "object" || receipt === null || Array.isArray(receipt)) {
396
430
  throw new OrcaError("send", "send response carries no send receipt", env.raw);
397
431
  }
398
- const s = receipt;
399
- if (str(s.handle) !== st.handle) {
400
- throw new OrcaError("send", `send receipt names terminal ${str(s.handle) ?? "none"}, not the addressed ${st.handle}`, env.raw);
401
- }
402
- const expectedBytes = Buffer.byteLength(cmd, "utf8") + 1; // 1 for the --enter newline
403
- if (s.accepted !== true || typeof s.bytesWritten !== "number" || s.bytesWritten !== expectedBytes) {
404
- throw new OrcaError("send", `send receipt does not report expected byte delivery (accepted: ${JSON.stringify(s.accepted)}, bytesWritten: ${s.bytesWritten}, expected: ${expectedBytes})`, env.raw);
432
+ const parsed = receipt;
433
+ if (str(parsed.handle) !== st.handle) {
434
+ throw new OrcaError("send", `send receipt names terminal ${str(parsed.handle) ?? "none"}, not the addressed ${st.handle}`, env.raw);
405
435
  }
436
+ return parsed;
406
437
  }
407
438
  async create(st, cmd) {
408
439
  await this.probeRuntime(this.cliCwd(st));
@@ -435,9 +466,10 @@ export class OrcaDriver {
435
466
  st.handle = handle;
436
467
  // The handle is bound to the runtime identity that ANSWERED its create.
437
468
  st.runtimeId = env.runtimeId;
438
- const surface = str(term.surface);
439
- if (surface !== undefined && surface !== "visible") {
440
- await this.notify(`tickmarkr orca terminal created on ${surface} surface`, { tier: "attention" });
469
+ st.surface = str(term.surface);
470
+ st.hostPlatform = str(term.hostPlatform);
471
+ if (st.surface !== undefined && st.surface !== "visible") {
472
+ await this.notify(`tickmarkr orca terminal created on ${st.surface} surface`, { tier: "attention" });
441
473
  }
442
474
  }
443
475
  /**
@@ -920,6 +952,75 @@ export class OrcaDriver {
920
952
  await this.time.sleep(Math.min(this.pollMs, left2));
921
953
  }
922
954
  }
955
+ async sendKey(slot, key) {
956
+ const st = this.state(slot);
957
+ this.assertAvailable("send", st);
958
+ if (key === "enter") {
959
+ await this.sendText(st, "");
960
+ return;
961
+ }
962
+ if (key === "ctrl+c") {
963
+ const env = await this.terminalOp("send", st, (h) => this.call("send", ["terminal", "send", "--terminal", h, "--interrupt"], this.cliCwd(st)), { mutating: true });
964
+ // UNRECORDED SHAPE: Orca 1.4.195 was not captured for `send --interrupt`. Until a live
965
+ // receipt exists, validate only the shared envelope and its handle binding; do not invent
966
+ // accepted/bytesWritten semantics for an interrupt.
967
+ this.sendReceipt(env, st);
968
+ return;
969
+ }
970
+ throw new OrcaError("send", `Orca has no terminal key verb for ${JSON.stringify(key)}`, "");
971
+ }
972
+ async nudge(slot, message) {
973
+ if (!message)
974
+ return false;
975
+ try {
976
+ const st = this.state(slot);
977
+ if (!await this.waitCondition(st, "tui-idle", 1))
978
+ return false;
979
+ // Drain to the current stream cursor before sending. A message already present makes the
980
+ // proof ambiguous, so decline rather than reporting delivery from old scrollback.
981
+ const before = await this.sweep(st);
982
+ if (before.includes(message) || joinWrapped(before).includes(message))
983
+ return false;
984
+ await this.sendText(st, message);
985
+ const deadline = this.time.now() + NUDGE_ECHO_TIMEOUT_MS;
986
+ for (;;) {
987
+ const after = await this.sweep(st);
988
+ if (after.includes(message) || joinWrapped(after).includes(message))
989
+ return true;
990
+ const left = deadline - this.time.now();
991
+ if (left <= 0)
992
+ return false;
993
+ await this.time.sleep(Math.min(this.pollMs, left));
994
+ }
995
+ }
996
+ catch {
997
+ return false;
998
+ }
999
+ }
1000
+ async narrator(cwd, command, runId) {
1001
+ if (!runId)
1002
+ throw new OrcaError("create", "Orca narrator requires a run identity", "");
1003
+ const slot = await this.slot(cwd, formatOwnedName({ role: "watch", taskId: "run", attempt: 0, runId }), { owned: { role: "watch", taskId: "run", attempt: 0, runId } });
1004
+ await this.run(slot, command);
1005
+ return slot;
1006
+ }
1007
+ async project(taskId, state) {
1008
+ const worktree = this.taskWorktrees.get(taskId);
1009
+ if (!worktree) {
1010
+ // The daemon projects in-progress immediately before it creates the task checkout/slot.
1011
+ // Hold only that latest state; slot() applies it once the task's own path is known.
1012
+ this.pendingProjects.set(taskId, { state, since: this.time.now() });
1013
+ return;
1014
+ }
1015
+ await this.setWorkspaceStatus(worktree, state);
1016
+ }
1017
+ async setWorkspaceStatus(worktree, state) {
1018
+ // UNRECORDED SHAPE: no Orca 1.4.195 `worktree set` receipt was captured. The shared envelope
1019
+ // parser is the complete success proof here; no result payload is assumed or fabricated.
1020
+ await this.call("worktree-set", [
1021
+ "worktree", "set", "--worktree", `path:${worktree}`, "--workspace-status", state,
1022
+ ], worktree);
1023
+ }
923
1024
  async notify(msg, opts) {
924
1025
  if (opts?.tier === "routine")
925
1026
  return;
@@ -977,6 +1078,52 @@ export class OrcaDriver {
977
1078
  throw new OrcaError("list", "terminal list is still truncated at totalCount rows", whole.raw);
978
1079
  return whole;
979
1080
  }
1081
+ openRunJournal(runId) {
1082
+ const roots = new Set(this.journalRoots.values());
1083
+ roots.add(process.cwd());
1084
+ for (const repoRoot of roots) {
1085
+ try {
1086
+ return Journal.open(repoRoot, runId, this.narrate);
1087
+ }
1088
+ catch {
1089
+ /* try the next daemon-bound root */
1090
+ }
1091
+ }
1092
+ return undefined;
1093
+ }
1094
+ // A projection exists only to bridge project() to the worker slot that follows it. After the
1095
+ // grace, desired remains the dispatch oracle: a declared worker is still being placed and must
1096
+ // keep its projection. Make genuine absence durable by appending first and deleting second; with
1097
+ // no writable run journal the entry remains eligible for a later reconcile.
1098
+ dropExpiredProjects(desired, runId) {
1099
+ const now = this.time.now();
1100
+ const desiredTasks = new Set();
1101
+ for (const name of desired) {
1102
+ const owned = parseOwnedName(name);
1103
+ if (owned?.role === "worker")
1104
+ desiredTasks.add(owned.taskId);
1105
+ }
1106
+ const expired = [...this.pendingProjects].filter(([, pending]) => now - pending.since > PENDING_PROJECT_GRACE_MS).filter(([taskId]) => !desiredTasks.has(taskId));
1107
+ if (expired.length === 0)
1108
+ return;
1109
+ const journal = this.openRunJournal(runId);
1110
+ if (!journal)
1111
+ return;
1112
+ for (const [taskId, pending] of expired) {
1113
+ try {
1114
+ const pendingMs = Math.max(0, now - pending.since);
1115
+ journal.append("project-unplaced", taskId, {
1116
+ state: pending.state,
1117
+ pendingMs,
1118
+ graceMs: PENDING_PROJECT_GRACE_MS,
1119
+ });
1120
+ this.pendingProjects.delete(taskId);
1121
+ }
1122
+ catch {
1123
+ /* reconcile is cosmetic; preserve the projection until a later journalled drop */
1124
+ }
1125
+ }
1126
+ }
980
1127
  /**
981
1128
  * Sweep tickmarkr-owned terminals down to `desired`. Ownership is decided ONLY by parseOwnedName
982
1129
  * over the owned TAB title, through the same panesToClose fold herdr uses (drivers/types.ts): an
@@ -992,6 +1139,7 @@ export class OrcaDriver {
992
1139
  * Cosmetic by contract: every failure is swallowed, per candidate and overall.
993
1140
  */
994
1141
  async reconcile(desired, runId, opts) {
1142
+ this.dropExpiredProjects(desired, runId);
995
1143
  try {
996
1144
  // Every call below is handle-addressed or explicitly selectored, so the CLI's own cwd selects
997
1145
  // nothing — it only has to exist, which the checkouts being swept no longer need to.
@@ -66,9 +66,17 @@ export declare function panesToClose(agents: FleetAgent[], desired: Set<string>,
66
66
  tabId?: string;
67
67
  }[];
68
68
  export declare function canonicalizeLegacyName(name: string, runId: string): OwnedName;
69
+ export interface SlotPlacement {
70
+ surface?: string;
71
+ hostPlatform?: string;
72
+ }
69
73
  export interface ExecutorDriver {
70
74
  id: string;
71
75
  interactive: boolean;
76
+ /** The exact terminal read surface used for liveness evidence. */
77
+ readSource?: string;
78
+ /** Placement facts returned by drivers whose terminal host exposes them. */
79
+ describe?(slot: Slot): SlotPlacement | Promise<SlotPlacement> | undefined;
72
80
  slot(cwd: string, name: string, opts?: SlotOpts): Promise<Slot>;
73
81
  run(slot: Slot, cmd: string): Promise<void>;
74
82
  waitOutput(slot: Slot, pattern: string, timeoutMs: number, opts?: {
@@ -84,5 +92,7 @@ export interface ExecutorDriver {
84
92
  narrateWith?(narrate: (event: JournalEvent) => void): void;
85
93
  worktree(repo: string, branch: string, baseRef: string): Promise<string>;
86
94
  narrator?: (cwd: string, command: string, runId?: string) => Promise<Slot>;
95
+ /** Best-effort projection of a task's lifecycle onto the execution host. */
96
+ project?: (taskId: string, state: "in-progress" | "in-review" | "completed") => Promise<void>;
87
97
  reconcile?: (desired: Set<string>, runId: string, opts?: PanesToCloseOpts) => Promise<void>;
88
98
  }
@@ -24,7 +24,9 @@ export interface BaselineCommand {
24
24
  fingerprints: string[];
25
25
  missingCommand?: boolean;
26
26
  /** Why a capture returned no verdict. */
27
- invalidCause?: "ceiling-kill" | "resource-exhaustion";
27
+ invalidCause?: "ceiling-kill" | "resource-exhaustion" | "infra";
28
+ /** The runner's summary was green and only its teardown fingerprint followed. */
29
+ teardownFingerprint?: true;
28
30
  /** What this command actually took at capture, on a pristine tree. Absent in pre-v1.90 baselines. */
29
31
  durationMs?: number;
30
32
  /** Sum of the per-file durations named by the runner; null when its output names none. */
@@ -154,4 +156,26 @@ export declare function detectVacuousOracles(cwd: string, tasks: ReadonlyArray<{
154
156
  id: string;
155
157
  acceptance: AcceptanceItem[];
156
158
  }>): Promise<VacuousOracleWarning[]>;
157
- export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[]): Promise<GateResult[]>;
159
+ export declare function compareToBaseline(cwd: string, commands: Record<string, string>, baseline: Baseline, enabled: string[], opts?: {
160
+ rerunOf?: HostStarvedRerun;
161
+ }): Promise<GateResult[]>;
162
+ export interface HostStarvedRerun {
163
+ durationMs: number;
164
+ referenceMs: number;
165
+ waitedMs: number;
166
+ }
167
+ export type RunnerVerdict = FailureClassification | "green-teardown" | undefined;
168
+ /** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
169
+ export declare function classifyRunnerOutput(raw: string, code: number): RunnerVerdict;
170
+ export declare const HOST_STARVED_DURATION_FACTOR = 2;
171
+ /** Every fresh failure head is timeout-class and the suite took twice its own baseline measurement. */
172
+ export declare function hostStarved(fresh: string, durationMs: number, referenceMs: number | undefined): boolean;
173
+ interface CalmWindow {
174
+ pollMs: number;
175
+ maxWaitMs: number;
176
+ loadProvider: () => number;
177
+ calmLoad: () => number;
178
+ }
179
+ export declare function setCalmWindowForTests(over: Partial<CalmWindow>): void;
180
+ export declare function resetCalmWindowForTests(): void;
181
+ export {};
@@ -1,14 +1,24 @@
1
1
  import { existsSync, readFileSync } from "node:fs";
2
+ import { availableParallelism, loadavg } from "node:os";
2
3
  import { join } from "node:path";
3
4
  import { DEFAULT_SHELL_TIMEOUT_MS, describeCapacity, sameCapacity, sh } from "../run/git.js";
4
5
  // incident #2 (run-20260709-104447): a vitest ✓ PASS line with "error" in the test NAME, wrapped in ANSI
5
6
  // codes that varied between baseline and worktree runs, was reported as a "new failure". Strip ANSI first;
6
7
  // a pass-marker line is never a failure. [\d;#] covers raw ANSI and digit-normalized ANSI ("\x1b[#m") from
7
8
  // baselines stored by pre-hardening code.
8
- const ANSI_RE = /\x1b\[[\d;#]*[A-Za-z]/g;
9
+ // OBS-891 (run 3372): vitest toggles the cursor (`\x1b[?25l` / `\x1b[?25h`) around its progress
10
+ // output, and the private-mode parameter byte `?` never matched [\d;#], so an echo-block HEADER glued
11
+ // to a cursor-show sequence stayed invisible to withoutVitestEchoBlocks and its whole block leaked as
12
+ // runner evidence — seven prose-only "infra" parks in one night. Full CSI grammar: parameter bytes
13
+ // 0x30–0x3F (plus `#` for digit-normalized stored baselines), intermediates 0x20–0x2F, final 0x40–0x7E.
14
+ const ANSI_RE = /\x1b\[[0-?#]*[ -/]*[@-~]/g;
9
15
  // ponytail: only leading ✓/✔ after optional "label:" prefixes (turbo/vitest), or tickmarkr's own run
10
16
  // summary, counts as a pass line — other runners' pass markers (PASS, ok) stay fingerprintable
11
17
  const PASS_LINE_RE = /^\s*(?:(?:[\w@./-]+:\s*)*[✓✔]|(?:\[tickmarkr\]\s+)?(?:tickmarkr\s+[\w.-]+:\s+)?(?:\d+|#)\s+done,\s+(?:\d+|#)\s+failed(?:,\s+(?:\d+|#)\s+awaiting human)?\b)/;
18
+ // OBS-888: tickmarkr's own operator lines (`tickmarkr: baseline capture for "test" … spawn EAGAIN`)
19
+ // are printed by this product, never by a runner about the work. When this repository's tests exercise
20
+ // the capture path they print them too, carrying errno tokens INFRA_RE would read as host evidence.
21
+ const OPERATOR_LINE_RE = /^\s*tickmarkr: /;
12
22
  // HYG-08 (D-01, incident run-20260711-154920): a failing test went unnamed for 3 attempts because details
13
23
  // headlined benign fingerprint-diff noise. These anchors harvest the runner's OWN failure naming from fresh
14
24
  // output to headline it. \s is fine in a TS regex — the BSD [[:space:]] rule binds shell grep only.
@@ -149,7 +159,10 @@ const isFingerprintShaped = (l) => isFailureShaped(l) || INFRA_RE.test(l);
149
159
  * unreadable-runner case the existing fail-closed path already owns.
150
160
  */
151
161
  export function classifyFailureOutput(output) {
152
- const lines = output.split("\n").map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l));
162
+ // OBS-891: the gate reads the WHOLE output here when the fresh-fingerprint diff is empty, so an errno
163
+ // token inside a test's echoed stdout/stderr block must be as invisible to this classifier as it is
164
+ // to fingerprint(): test-owned output is never runner evidence about the work.
165
+ const lines = withoutVitestEchoBlocks(output).map((l) => l.replace(ANSI_RE, "")).filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
153
166
  if (lines.some(namesRegression))
154
167
  return "regression";
155
168
  return lines.some(isInfraLine) ? "infra" : undefined;
@@ -161,7 +174,11 @@ export function classifyFailureOutput(output) {
161
174
  * no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
162
175
  * separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
163
176
  */
164
- const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+\S+\.(?:test|spec)\.[cm]?[jt]sx?\s+>\s+\S/;
177
+ // OBS-888 row 1: vitest 3.2.7 heads an echo block `std{out,err} | <file> > <test>` when the log is
178
+ // attributed to a test, `stderr | unknown test` when it is not, `stderr | <file>` for file-level output
179
+ // and `stderr | <task id>` (digits and underscores) when the reporter no longer knows the task
180
+ // (dist/chunks/index.*.js, `headerText`). The stripper knew only the first form.
181
+ const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+(?:unknown test\s*$|\d+_[\d_]+\s*$|\S+\.(?:test|spec)\.[cm]?[jt]sx?(?:\s*$|\s+>\s+\S))/;
165
182
  /** Keep runner output while omitting Vitest's echoed test-owned stdout/stderr diagnostic blocks. */
166
183
  const withoutVitestEchoBlocks = (output) => {
167
184
  const outside = [];
@@ -186,7 +203,7 @@ const captureInvalidatingLines = (output) => {
186
203
  const invalidating = [];
187
204
  for (const line of withoutVitestEchoBlocks(output)) {
188
205
  const clean = line.replace(ANSI_RE, "");
189
- if (!PASS_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
206
+ if (!PASS_LINE_RE.test(clean) && !OPERATOR_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
190
207
  invalidating.push(line);
191
208
  }
192
209
  return invalidating;
@@ -230,16 +247,18 @@ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
230
247
  export function fingerprint(output) {
231
248
  const lines = withoutVitestEchoBlocks(output)
232
249
  .map((l) => l.replace(ANSI_RE, ""))
233
- .filter((l) => !PASS_LINE_RE.test(l));
250
+ .filter((l) => !PASS_LINE_RE.test(l) && !OPERATOR_LINE_RE.test(l));
234
251
  // GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
235
252
  // prefix removed. A recognized stripped line fingerprints as its STRIPPED text, so the same
236
253
  // failure fingerprints identically whether turbo prefixed it or a bare runner printed it; the
237
254
  // prefixed form keeps fingerprinting too (baseline-recorded package-level reds stay forgivable).
238
255
  const shaped = [];
239
256
  for (const l of lines) {
257
+ const stripped = stripTurboPrefix(l);
258
+ if (stripped !== undefined && OPERATOR_LINE_RE.test(stripped))
259
+ continue; // OBS-888: operator prose under a turbo prefix
240
260
  if (isFingerprintShaped(l))
241
261
  shaped.push(l);
242
- const stripped = stripTurboPrefix(l);
243
262
  if (stripped !== undefined && !PASS_LINE_RE.test(stripped) && isFingerprintShaped(stripped))
244
263
  shaped.push(stripped);
245
264
  }
@@ -483,11 +502,20 @@ export async function captureBaseline(cwd, commands) {
483
502
  base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
484
503
  continue;
485
504
  }
505
+ // OBS-885/887: capture and gate ask the same classifier. A green summary followed only by the
506
+ // teardown fingerprint is a pass; infrastructure without a summary is no verdict to forgive.
507
+ const runnerVerdict = classifyRunnerOutput(raw, r.code);
508
+ if (runnerVerdict === "infra") {
509
+ console.error(`tickmarkr: baseline capture for "${name}" carries runner-infrastructure evidence and no green summary — it recorded NO verdict; nothing is forgiven for this command`);
510
+ base.commands[name] = invalidCaptureEntry(durationMs, "infra");
511
+ continue;
512
+ }
486
513
  base.commands[name] = {
487
- exitCode: r.code,
514
+ exitCode: runnerVerdict === "green-teardown" ? 0 : r.code,
515
+ ...(runnerVerdict === "green-teardown" ? { teardownFingerprint: true } : {}),
488
516
  // a command that exits 0 has no failures to fingerprint — recording any would be a lie the
489
517
  // compare step then has to forgive
490
- fingerprints: r.code === 0 ? [] : fingerprint(raw),
518
+ fingerprints: r.code === 0 || runnerVerdict === "green-teardown" ? [] : fingerprint(raw),
491
519
  missingCommand: missingConfiguredCommand(cmd, r),
492
520
  durationMs,
493
521
  ...fileTiming(raw, durationMs),
@@ -552,8 +580,9 @@ function headlineDetails(raw, fresh) {
552
580
  meta: { failingTests: headlines.filter(namesFailureEitherForm) },
553
581
  };
554
582
  }
555
- export async function compareToBaseline(cwd, commands, baseline, enabled) {
583
+ export async function compareToBaseline(cwd, commands, baseline, enabled, opts = {}) {
556
584
  const results = [];
585
+ const rerunOf = opts.rerunOf;
557
586
  for (const name of enabled) {
558
587
  const cmd = commands[name];
559
588
  if (!cmd) {
@@ -576,7 +605,14 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
576
605
  // result rather than re-derived after the fact. The skip row above ran no command and therefore
577
606
  // states no capacity — a row that never divided the machine must not claim that it did.
578
607
  const record = (g) => {
579
- results.push(r.capacity ? { ...g, capacity: r.capacity } : g);
608
+ const withReap = r.reapedGroup ? { ...g, meta: { ...g.meta, reapedGroup: true } } : g;
609
+ const withReapError = r.reapError ? { ...withReap, meta: { ...withReap.meta, reapError: r.reapError } } : withReap;
610
+ const withRerun = rerunOf ? {
611
+ ...withReapError,
612
+ details: `host-starved rerun after waiting ${rerunOf.waitedMs}ms for a calm load window: ${withReapError.details}`,
613
+ meta: { ...withReapError.meta, hostStarvedRerun: rerunOf },
614
+ } : withReapError;
615
+ results.push(r.capacity ? { ...withRerun, capacity: r.capacity } : withRerun);
580
616
  };
581
617
  // …and whether the entry that would forgive this command was measured in the same world. A
582
618
  // baseline captured under a different fork cap forgives nothing: its fingerprints describe a
@@ -610,6 +646,12 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
610
646
  continue;
611
647
  }
612
648
  const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
649
+ // OBS-885/887: the same classifier the capture applied names a completed green suite on both sides.
650
+ const runnerVerdict = classifyRunnerOutput(raw, r.code);
651
+ if (runnerVerdict === "green-teardown") {
652
+ record({ gate: name, pass: true, details: `exit ${r.code} after a green suite summary; only the runner's teardown fingerprint followed it`, meta: { teardownFingerprint: true } });
653
+ continue;
654
+ }
613
655
  // OBS-278: only a failure SHAPE is a verdict — everything fingerprint() keeps is one, except the
614
656
  // unrecognized-output marker, which is evidence for the operator and never grounds to reject.
615
657
  // ponytail: ceiling — a runner whose failure output holds no shape above and whose baseline is
@@ -621,10 +663,20 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
621
663
  // T9: classify the FRESH diff before charging it. The complete runner output can legitimately
622
664
  // contain a baseline-recorded assertion beside a newly introduced infrastructure death; letting
623
665
  // that known assertion outvote the fresh birpc line turns machine failure into a worker defect.
624
- // `classifyFailureOutput` remains the single discriminator. When there is no fresh fingerprint,
666
+ // `classifyRunnerOutput` remains the single discriminator. When there is no fresh fingerprint,
625
667
  // retain the whole-output read so a repeated infra abort can never be baseline-forgiven as green.
626
- const freshClassification = failing.length ? classifyFailureOutput(failing.join("\n")) : undefined;
627
- const classification = freshClassification ?? (!failing.length ? classifyFailureOutput(raw) : undefined);
668
+ const freshVerdict = failing.length ? classifyRunnerOutput(failing.join("\n"), r.code) : undefined;
669
+ const freshClassification = freshVerdict === "infra" || freshVerdict === "regression" ? freshVerdict : undefined;
670
+ const classification = freshClassification ?? (!failing.length && (runnerVerdict === "infra" || runnerVerdict === "regression") ? runnerVerdict : undefined);
671
+ // OBS-896: every fresh failure must be timeout-class, and the suite must take at least twice its
672
+ // own baseline measurement. The first read buys one calm rerun here, never a worker repair.
673
+ if (name === "test" && classification !== "infra" && failing.length && !rerunOf
674
+ && hostStarved(failing.join("\n"), r.durationMs ?? 0, entry?.durationMs)) {
675
+ const waitedMs = await waitForCalmWindow();
676
+ const provenance = { durationMs: r.durationMs ?? 0, referenceMs: entry?.durationMs ?? 0, waitedMs };
677
+ results.push(...await compareToBaseline(cwd, { [name]: cmd }, baseline, [name], { rerunOf: provenance }));
678
+ continue;
679
+ }
628
680
  if (classification === "infra") {
629
681
  const evidence = failing.length
630
682
  ? failing.slice(0, 10).join("\n")
@@ -689,3 +741,53 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
689
741
  }
690
742
  return results;
691
743
  }
744
+ const SUMMARY_LINE_RE = /^[^\S\n]*Test Files[^\S\n]+(.+)$/m;
745
+ const TEARDOWN_RE = /\[vitest-worker\]: Timeout calling\b|\[birpc\] rpc is closed, cannot call\b/;
746
+ const UNHANDLED_HEADER_RE = /^\s*[^\w]*(?:Unhandled Errors|Uncaught Exception)\b/;
747
+ /** One interpretation of runner bytes, shared by baseline capture and every gate consumer. */
748
+ export function classifyRunnerOutput(raw, code) {
749
+ if (code === 0)
750
+ return undefined;
751
+ const lines = withoutVitestEchoBlocks(raw).map((l) => l.replace(ANSI_RE, ""));
752
+ const text = lines.join("\n");
753
+ const summary = SUMMARY_LINE_RE.exec(text);
754
+ if (summary) {
755
+ const failed = [...summary[1].matchAll(/\b(\d+)\s+failed\b/g)].map((match) => Number(match[1]));
756
+ const summaryGreen = failed.every((count) => count === 0);
757
+ const summaryLine = text.slice(0, summary.index).split("\n").length - 1;
758
+ const teardownLine = lines.findIndex((line, index) => index > summaryLine && TEARDOWN_RE.test(line));
759
+ const otherFailure = lines.some((line, index) => {
760
+ if (index === summaryLine || TEARDOWN_RE.test(line))
761
+ return false;
762
+ if (PASS_LINE_RE.test(line) || OPERATOR_LINE_RE.test(line))
763
+ return false;
764
+ if (index > summaryLine && UNHANDLED_HEADER_RE.test(line))
765
+ return false;
766
+ return namesRegression(line);
767
+ });
768
+ if (summaryGreen && teardownLine > summaryLine && !otherFailure)
769
+ return "green-teardown";
770
+ }
771
+ return classifyFailureOutput(text);
772
+ }
773
+ const TIMEOUT_CLASS_RE = /\b(?:Test|Hook) timed out in (?:\d+|#) ?ms\b|\bexceeded (?:\d+|#) ?(?:ms|s|seconds)\b|\btimed out after (?:\d+|#)/i;
774
+ const ERROR_HEAD_RE = /^\s*(?:[A-Za-z][A-Za-z0-9]*Error|Error):\s/;
775
+ export const HOST_STARVED_DURATION_FACTOR = 2;
776
+ /** Every fresh failure head is timeout-class and the suite took twice its own baseline measurement. */
777
+ export function hostStarved(fresh, durationMs, referenceMs) {
778
+ if (!referenceMs || durationMs < HOST_STARVED_DURATION_FACTOR * referenceMs)
779
+ return false;
780
+ const heads = fresh.split("\n").map((line) => line.replace(ANSI_RE, "")).filter((line) => ERROR_HEAD_RE.test(line));
781
+ return heads.length > 0 && heads.every((line) => TIMEOUT_CLASS_RE.test(line));
782
+ }
783
+ const DEFAULT_CALM = { pollMs: 5_000, maxWaitMs: 600_000, loadProvider: () => loadavg()[0] ?? 0, calmLoad: () => availableParallelism() / 2 };
784
+ let calm = DEFAULT_CALM;
785
+ export function setCalmWindowForTests(over) { calm = { ...calm, ...over }; }
786
+ export function resetCalmWindowForTests() { calm = DEFAULT_CALM; }
787
+ async function waitForCalmWindow() {
788
+ const started = Date.now();
789
+ while (calm.loadProvider() > calm.calmLoad() && Date.now() - started < calm.maxWaitMs) {
790
+ await new Promise((resolve) => setTimeout(resolve, calm.pollMs));
791
+ }
792
+ return Date.now() - started;
793
+ }
@@ -1,6 +1,7 @@
1
1
  import { type Assignment, type BillingChannel, type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type TickmarkrConfig, type Tier } from "../config/config.js";
3
3
  import { type Task } from "../graph/schema.js";
4
+ import { modelProvider } from "../route/preference.js";
4
5
  import { type GateVia } from "./llm.js";
5
6
  import type { GateResult } from "./types.js";
6
7
  import { type VerdictUnparseableCause } from "./verdict-cause.js";
@@ -48,11 +49,12 @@ export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffM
48
49
  export declare function isDiffCapPark(result: GateResult): boolean;
49
50
  export declare function diffCapParkReason(results: GateResult[]): string | null;
50
51
  export declare function modelId(model: string): string;
51
- /** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
52
- export declare function modelProvider(model: string, fallback?: string): string;
52
+ export { modelProvider };
53
53
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
54
54
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
55
- floor?: Tier): BillingChannel | null;
55
+ floor?: Tier, // task-declared only; config floors govern workers and must not silently move review seats
56
+ history?: string[], // run-scoped picks, oldest to newest; empty preserves the established ranking
57
+ onSeat?: (seat: number) => void): BillingChannel | null;
56
58
  export type ReviewUnparseableCause = VerdictUnparseableCause;
57
59
  /**
58
60
  * This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
@@ -60,4 +62,4 @@ export type ReviewUnparseableCause = VerdictUnparseableCause;
60
62
  * judgement rather than a guarantee made by this renderer.
61
63
  */
62
64
  export declare function renderDeclaredWriteScope(files: ReadonlyArray<string>): string;
63
- export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string): Promise<GateResult>;
65
+ export declare function reviewGate(task: Task, worktree: string, baseRef: string, author: Assignment, channels: BillingChannel[], adapters: WorkerAdapter[], cfg: TickmarkrConfig, via?: GateVia, excludeReviewers?: string[], artifactDir?: string, reviewHistory?: string[]): Promise<GateResult>;