tickmarkr 2.2.1 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +10 -9
  2. package/dist/adapters/types.d.ts +20 -1
  3. package/dist/adapters/types.js +42 -2
  4. package/dist/cli/commands/approve.js +5 -4
  5. package/dist/cli/commands/beat.js +7 -4
  6. package/dist/cli/commands/doctor.d.ts +5 -1
  7. package/dist/cli/commands/doctor.js +67 -7
  8. package/dist/cli/commands/init.js +36 -21
  9. package/dist/cli/commands/plan.js +16 -2
  10. package/dist/cli/commands/report.js +37 -1
  11. package/dist/cli/commands/verify.d.ts +5 -0
  12. package/dist/cli/commands/verify.js +140 -25
  13. package/dist/compile/collateral.js +15 -9
  14. package/dist/compile/native.js +3 -3
  15. package/dist/config/config.js +1 -1
  16. package/dist/drivers/index.d.ts +6 -0
  17. package/dist/drivers/index.js +19 -4
  18. package/dist/drivers/orca.d.ts +22 -1
  19. package/dist/drivers/orca.js +147 -6
  20. package/dist/drivers/subprocess.d.ts +3 -3
  21. package/dist/drivers/subprocess.js +16 -9
  22. package/dist/drivers/types.d.ts +2 -0
  23. package/dist/gates/baseline.d.ts +2 -0
  24. package/dist/gates/baseline.js +21 -4
  25. package/dist/gates/llm.d.ts +6 -0
  26. package/dist/gates/llm.js +25 -9
  27. package/dist/gates/review.d.ts +3 -1
  28. package/dist/gates/review.js +40 -12
  29. package/dist/gates/run-gates.js +17 -10
  30. package/dist/gates/verdict-cause.d.ts +6 -2
  31. package/dist/gates/verdict-cause.js +8 -4
  32. package/dist/run/consult.js +5 -1
  33. package/dist/run/daemon.d.ts +12 -0
  34. package/dist/run/daemon.js +218 -25
  35. package/dist/run/git.d.ts +1 -0
  36. package/dist/run/git.js +4 -0
  37. package/dist/run/journal.d.ts +15 -2
  38. package/dist/run/journal.js +70 -12
  39. package/dist/run/supervision.d.ts +6 -0
  40. package/dist/run/supervision.js +29 -1
  41. package/dist/tui/ink/init-app.js +4 -4
  42. package/package.json +1 -1
  43. package/skills/tickmarkr-overseer/SKILL.md +77 -18
  44. package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
  45. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
  46. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
  47. package/skills/tickmarkr-overseer/scripts/watch-context.sh +33 -7
  48. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
@@ -2,8 +2,9 @@ import { realpathSync } from "node:fs";
2
2
  import { resolve } from "node:path";
3
3
  import { shq } from "../adapters/types.js";
4
4
  import { createWorktree, sh } from "../run/git.js";
5
+ import { Journal } from "../run/journal.js";
5
6
  import { MAX_BUF } from "./subprocess.js";
6
- import { formatOwnedName, panesToClose } from "./types.js";
7
+ import { formatOwnedName, panesToClose, parseOwnedName } from "./types.js";
7
8
  // Orca (onorca.dev) as a third execution surface beside herdr and subprocess. tickmarkr keeps
8
9
  // worktrees, routing, gates, journal and merges; orca supplies visible terminals only. Everything
9
10
  // here is bound by the 1.4.186 conformance spike
@@ -31,7 +32,10 @@ import { formatOwnedName, panesToClose } from "./types.js";
31
32
  // - `terminal list`'s `--worktree` is OPTIONAL (same help): the reconcile sweep omits it, because
32
33
  // an older run's leftover sits in a checkout this run never knew (T2).
33
34
  /** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
34
- export const ORCA_RESPONSE_FAMILIES = ["status", "create", "list", "read", "send", "wait", "show", "close"];
35
+ export const ORCA_RESPONSE_FAMILIES = [
36
+ "status", "create", "list", "read", "send", "wait", "show", "close",
37
+ "worktree-current", "hooks-status",
38
+ ];
35
39
  export const ORCA_FIXTURE_VERSION = "1.4.195";
36
40
  export const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
37
41
  export const STALE_HANDLE_CODE = "terminal_handle_stale";
@@ -50,6 +54,9 @@ const PAGE_LINES = 500; // per-page ask; orca caps server-side and reports `limi
50
54
  const LIST_LIMIT = 10000; // well past orca's own row default; `truncated` still decides (listAll)
51
55
  const MAX_PAGES = 400; // runaway guard: a cursor that stops advancing ends the sweep, never loops
52
56
  const POLL_MS = 200;
57
+ export const WORKTREE_ADOPTION_TIMEOUT_MS = 60_000;
58
+ const WORKTREE_ADOPTION_POLL_MS = 1_000;
59
+ const WORKTREE_ADOPTION_JOURNAL_MS = 2_000;
53
60
  const SYSTEM_TIME = {
54
61
  now: () => Date.now(),
55
62
  sleep: (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
@@ -252,6 +259,9 @@ export class OrcaDriver {
252
259
  pageLines;
253
260
  pollMs;
254
261
  probeStalenessMs;
262
+ journalRoots = new Map();
263
+ narrate;
264
+ hookCoverage;
255
265
  constructor(opts = {}) {
256
266
  this.bin = opts.bin ?? resolveOrcaCliBinary(process.cwd(), { env: opts.env, platform: opts.platform }) ?? "orca";
257
267
  // Config values flow into a shell here: every argv element is quoted, always.
@@ -347,7 +357,7 @@ export class OrcaDriver {
347
357
  // Canonical: git and Orca can spell one checkout two ways, and every later comparison — create
348
358
  // receipt, relist, reconcile — is against THIS value.
349
359
  const worktree = canonicalWorktreePath(cwd);
350
- this.slots.set(id, { title, cwd: worktree, buf: "", recoveries: 0, recovering: false });
360
+ this.slots.set(id, { title, cwd: worktree, agent: opts?.agent, buf: "", recoveries: 0, recovering: false });
351
361
  return { id, name: title, cwd: worktree, group: opts?.group };
352
362
  }
353
363
  /** Where to invoke the CLI for this slot's calls (see OrcaSlotState.dir). */
@@ -396,6 +406,7 @@ export class OrcaDriver {
396
406
  }
397
407
  async create(st, cmd) {
398
408
  await this.probeRuntime(this.cliCwd(st));
409
+ await this.awaitWorktreeAdoption(st);
399
410
  // The selector names THIS slot's checkout outright, and the CLI child is bound to it too, so
400
411
  // neither the UI's active worktree (`active`/`current`) nor the daemon's cwd can place it.
401
412
  const env = await this.call("create", [
@@ -429,6 +440,64 @@ export class OrcaDriver {
429
440
  await this.notify(`tickmarkr orca terminal created on ${surface} surface`, { tier: "attention" });
430
441
  }
431
442
  }
443
+ /**
444
+ * A freshly-created git checkout does not become a valid Orca selector atomically. Ask
445
+ * `worktree current` FROM that checkout until Orca itself resolves the exact filesystem identity;
446
+ * an enclosing checkout is still not adoption. Only selector_not_found is a retryable refusal —
447
+ * malformed envelopes and every other refusal remain explicit driver failures.
448
+ */
449
+ async awaitWorktreeAdoption(st) {
450
+ const started = this.time.now();
451
+ const deadline = started + WORKTREE_ADOPTION_TIMEOUT_MS;
452
+ let lastReported;
453
+ for (;;) {
454
+ const left = deadline - this.time.now();
455
+ if (left < 0)
456
+ break;
457
+ try {
458
+ const env = await this.call("worktree-current", ["worktree", "current"], st.cwd, Math.max(1, left));
459
+ const worktree = env.result.worktree;
460
+ if (typeof worktree !== "object" || worktree === null || Array.isArray(worktree)) {
461
+ throw new OrcaError("worktree-current", "response carries no worktree record", env.raw, { runtimeId: env.runtimeId });
462
+ }
463
+ const reported = terminalWorktree(worktree);
464
+ if (!reported) {
465
+ throw new OrcaError("worktree-current", "worktree record carries no path", env.raw, { runtimeId: env.runtimeId });
466
+ }
467
+ lastReported = canonicalWorktreePath(reported);
468
+ if (lastReported === st.cwd) {
469
+ const waitedMs = this.time.now() - started;
470
+ if (waitedMs > WORKTREE_ADOPTION_JOURNAL_MS)
471
+ this.appendAdoptionWait(st, waitedMs);
472
+ return;
473
+ }
474
+ }
475
+ catch (error) {
476
+ if (!(error instanceof OrcaError) || error.code !== "selector_not_found")
477
+ throw error;
478
+ lastReported = undefined;
479
+ }
480
+ const remaining = deadline - this.time.now();
481
+ if (remaining <= 0)
482
+ break;
483
+ await this.time.sleep(Math.min(WORKTREE_ADOPTION_POLL_MS, remaining));
484
+ }
485
+ const waitedMs = this.time.now() - started;
486
+ throw new OrcaUnavailableError("worktree-current", `Orca did not adopt ${st.cwd} within ${WORKTREE_ADOPTION_TIMEOUT_MS}ms${lastReported ? ` (last answered ${lastReported})` : ""}`, `waitedMs=${waitedMs}`);
487
+ }
488
+ /** Same repo/run/narration path Herdr uses for its driver-owned dispatch-retry row. */
489
+ appendAdoptionWait(st, waitedMs) {
490
+ const owned = parseOwnedName(st.title);
491
+ if (!owned)
492
+ throw new Error(`cannot journal worktree-adoption-wait: slot ${st.title} carries no run identity`);
493
+ const repoRoot = this.journalRoots.get(st.cwd);
494
+ if (!repoRoot) {
495
+ throw new Error(`cannot journal worktree-adoption-wait: slot ${st.title} has no daemon repo binding for ${st.cwd}`);
496
+ }
497
+ Journal.open(repoRoot, owned.runId, this.narrate).append("worktree-adoption-wait", owned.taskId, {
498
+ milliseconds: waitedMs,
499
+ });
500
+ }
432
501
  // ---- handle identity and restart recovery ----------------------------------------------------
433
502
  /**
434
503
  * Every terminal-addressed call — read AND write — goes through here, and the runtime identity is
@@ -628,6 +697,18 @@ export class OrcaDriver {
628
697
  ], this.cliCwd(st)), { onRecovered: () => { recovered = true; } });
629
698
  return { term: this.validated(family, st, env), raw: env.raw, recovered };
630
699
  }
700
+ /** A rendered-frame liveness read. `--screen` and `--cursor` are mutually exclusive in Orca. */
701
+ async readScreen(st) {
702
+ const env = await this.terminalOp("status", st, (h) => this.call("read", [
703
+ "terminal", "read", "--terminal", h, "--screen",
704
+ ], this.cliCwd(st)));
705
+ const term = this.validated("status", st, env);
706
+ const source = str(term.source);
707
+ if (source !== "screen" && source !== "screen-unavailable") {
708
+ throw new OrcaError("read", `screen read reports source ${source ?? "absent"}, not screen or screen-unavailable`, env.raw);
709
+ }
710
+ return { term, source };
711
+ }
631
712
  /** A single UNPAGED tail read — exactly what the caller asked for and nothing more. Markers split
632
713
  * across cursor pages are not reassembled here; that is waitOutput's job. */
633
714
  async read(slot, lines) {
@@ -704,10 +785,14 @@ export class OrcaDriver {
704
785
  // reporting even when the show record alone would still look connected (show carries no
705
786
  // status field of its own, so a terminal can report "unknown"/"exited" on read while its
706
787
  // show row still says connected — the read leg is the only place that catches that).
707
- await this.readPage("status", st, undefined, 1);
788
+ const screen = await this.readScreen(st);
708
789
  if (`${st.runtimeId}:${st.handle}:${st.recoveries}` !== gen) {
709
790
  continue;
710
791
  }
792
+ // No rendered frame means no trustworthy TUI state. In particular, the stream fragments this
793
+ // call replaced cannot license an idle verdict or a wait probe.
794
+ if (screen.source === "screen-unavailable")
795
+ return "unknown";
711
796
  const env = await this.terminalOp("show", st, (h) => this.call("show", ["terminal", "show", "--terminal", h], this.cliCwd(st)));
712
797
  if (`${st.runtimeId}:${st.handle}:${st.recoveries}` !== gen) {
713
798
  continue;
@@ -715,6 +800,11 @@ export class OrcaDriver {
715
800
  const term = this.liveShowTerm("status", st, env);
716
801
  if (term.agentWait === true)
717
802
  return "blocked";
803
+ // Orca can only report tui-idle/agentWait for agents whose managed hook is installed. A
804
+ // definitively unhooked adapter is unknown; an agent absent from Orca's table keeps the
805
+ // legacy probe because absence is not proof that the CLI has no compatible status surface.
806
+ if (await this.agentHookAvailable(st) === false)
807
+ return "unknown";
718
808
  // `idle` is proven only by orca's own tui-idle condition: a 1ms wait is a point-in-time probe —
719
809
  // satisfied now → idle; elapsed (the recorded `timeout` refusal) → not idle.
720
810
  const isIdle = await this.waitCondition(st, "tui-idle", 1);
@@ -724,6 +814,50 @@ export class OrcaDriver {
724
814
  return mapAgentState(term, isIdle);
725
815
  }
726
816
  }
817
+ hookAgent(adapter) {
818
+ if (adapter === "claude-code")
819
+ return "claude";
820
+ if (adapter === "cursor-agent")
821
+ return "cursor";
822
+ return adapter;
823
+ }
824
+ /** true = hooked, false = definitively unhooked, undefined = agent absent from Orca's table. */
825
+ async agentHookAvailable(st) {
826
+ if (!st.agent)
827
+ return undefined;
828
+ this.hookCoverage ??= this.loadHookCoverage(st.cwd);
829
+ const coverage = await this.hookCoverage;
830
+ const state = coverage.states.get(this.hookAgent(st.agent));
831
+ // The table's omission is deliberately inconclusive even when managed hooks are disabled:
832
+ // Orca may not know this agent, so preserve the legacy probe exactly as an unlisted row does.
833
+ if (state === undefined)
834
+ return undefined;
835
+ return coverage.enabled && state === "installed";
836
+ }
837
+ async loadHookCoverage(cwd) {
838
+ const env = await this.call("hooks-status", ["agent", "hooks", "status"], cwd);
839
+ if (typeof env.result.enabled !== "boolean") {
840
+ throw new OrcaError("hooks-status", "response carries no boolean enabled", env.raw, { runtimeId: env.runtimeId });
841
+ }
842
+ const statuses = env.result.statuses;
843
+ if (!Array.isArray(statuses)) {
844
+ throw new OrcaError("hooks-status", "response carries no statuses array", env.raw, { runtimeId: env.runtimeId });
845
+ }
846
+ const states = new Map();
847
+ const allowed = new Set(["installed", "not_installed", "partial", "error"]);
848
+ for (const row of statuses) {
849
+ if (typeof row !== "object" || row === null || Array.isArray(row)) {
850
+ throw new OrcaError("hooks-status", "statuses carries a non-object row", env.raw, { runtimeId: env.runtimeId });
851
+ }
852
+ const agent = str(row.agent);
853
+ const state = str(row.state);
854
+ if (!agent || !state || !allowed.has(state)) {
855
+ throw new OrcaError("hooks-status", "status row carries no valid agent/state pair", env.raw, { runtimeId: env.runtimeId });
856
+ }
857
+ states.set(agent, state);
858
+ }
859
+ return { enabled: env.result.enabled, states };
860
+ }
727
861
  /** One `terminal wait` through the full identity machinery. The recorded 1.4.186 elapsed answer
728
862
  * is rc 1 + ok:true + {handle, condition, satisfied:false, status:"running"}; it is "not yet"
729
863
  * only after this method validates all four fields. Any malformed/refused wait remains explicit. */
@@ -791,6 +925,9 @@ export class OrcaDriver {
791
925
  return;
792
926
  console.log(`[tickmarkr] ${msg}`); // console fallback only — notification injection is out of scope
793
927
  }
928
+ narrateWith(narrate) {
929
+ this.narrate = narrate;
930
+ }
794
931
  async close(slot) {
795
932
  const st = this.slots.get(slot.id);
796
933
  if (!st)
@@ -909,7 +1046,11 @@ export class OrcaDriver {
909
1046
  catch { /* cosmetic — visibility hygiene never fails the run */ }
910
1047
  }
911
1048
  // tickmarkr's own createWorktree stays the sole checkout authority — orca never makes worktrees.
912
- worktree(repo, branch, baseRef) {
913
- return createWorktree(repo, branch, baseRef);
1049
+ async worktree(repo, branch, baseRef) {
1050
+ const worktree = await createWorktree(repo, branch, baseRef);
1051
+ const repoRoot = canonicalWorktreePath(repo);
1052
+ this.journalRoots.set(repoRoot, repoRoot);
1053
+ this.journalRoots.set(canonicalWorktreePath(worktree), repoRoot);
1054
+ return worktree;
914
1055
  }
915
1056
  }
@@ -1,14 +1,14 @@
1
1
  import type { ExecutorDriver, NotifyOpts, Slot } from "./types.js";
2
2
  export declare const MAX_BUF: number;
3
- export declare const HERDR_CONTROL_VARS: readonly ["HERDR_ENV", "HERDR_SOCKET_PATH"];
3
+ export declare const HERDR_CONTROL_VARS: readonly ["HERDR_ENV", "HERDR_SOCKET_PATH", "ORCA_TERMINAL_HANDLE", "ORCA_PANE_KEY", "ORCA_TAB_ID"];
4
4
  /**
5
- * Copy of worker env with the fork cap applied and herdr control-plane vars stripped.
5
+ * Copy of worker env with the fork cap applied and host control-plane vars stripped.
6
6
  * The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
7
7
  * same machine the gate shells do, so both seams have to read the same run-owned number rather
8
8
  * than a flat constant. The operator's own export still wins.
9
9
  */
10
10
  export declare function sealHerdrEnv(env?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
11
- /** Pane/login-shell form of the same worker env seal (herdr seed + daemon setup). */
11
+ /** Pane/login-shell form of the same host-neutral worker env seal (herdr seed + daemon setup). */
12
12
  export declare function herdrSealShellPrefix(env?: NodeJS.ProcessEnv): string;
13
13
  export declare class SubprocessDriver implements ExecutorDriver {
14
14
  id: string;
@@ -8,13 +8,20 @@ import { createWorktree, FORK_CAP_ENV, resolvedForkCap } from "../run/git.js";
8
8
  // waitOutput poll (~10MB/s sustained, far beyond agent-CLI rates). Tail-truncate, never head:
9
9
  // consumers only ever tail-read.
10
10
  export const MAX_BUF = 2 * 1024 * 1024;
11
- // OBS-17 / v1.22 T3: control-plane vars that let a process talk to the operator's herdr.
12
- // Workers, judges, reviewers, and consults must never inherit them — only the daemon process
13
- // (and its own herdr driver CLI calls) keep the live session. Socket path is the wire; HERDR_ENV
14
- // is the "I am inside herdr" gate every agent skill checks before mutating panes.
15
- export const HERDR_CONTROL_VARS = ["HERDR_ENV", "HERDR_SOCKET_PATH"];
11
+ // OBS-17 / v1.22 T3 and OBS-843: host control-plane vars that let a process address the operator's
12
+ // herdr or Orca UI. Workers, judges, reviewers, and consults must never inherit them — only the
13
+ // daemon process and its driver calls keep the live session. TERM_PROGRAM is host description, not
14
+ // an addressing capability, and ORCA_AGENT_HOOK_* belongs to agent status transport, so both stay.
15
+ // Keep the exported name for compatibility even though the list is now host-neutral.
16
+ export const HERDR_CONTROL_VARS = [
17
+ "HERDR_ENV",
18
+ "HERDR_SOCKET_PATH",
19
+ "ORCA_TERMINAL_HANDLE",
20
+ "ORCA_PANE_KEY",
21
+ "ORCA_TAB_ID",
22
+ ];
16
23
  /**
17
- * Copy of worker env with the fork cap applied and herdr control-plane vars stripped.
24
+ * Copy of worker env with the fork cap applied and host control-plane vars stripped.
18
25
  * The cap is the one the enclosing run resolved (resolvedForkCap) — a worker's suites divide the
19
26
  * same machine the gate shells do, so both seams have to read the same run-owned number rather
20
27
  * than a flat constant. The operator's own export still wins.
@@ -27,7 +34,7 @@ export function sealHerdrEnv(env = process.env) {
27
34
  delete out[k];
28
35
  return out;
29
36
  }
30
- /** Pane/login-shell form of the same worker env seal (herdr seed + daemon setup). */
37
+ /** Pane/login-shell form of the same host-neutral worker env seal (herdr seed + daemon setup). */
31
38
  export function herdrSealShellPrefix(env = process.env) {
32
39
  const forkCap = sealHerdrEnv(env)[FORK_CAP_ENV] ?? resolvedForkCap();
33
40
  return `export ${FORK_CAP_ENV}=${shq(forkCap)}; ` +
@@ -56,8 +63,8 @@ export class SubprocessDriver {
56
63
  // HARD-05: interactive=false — no operator, so an open stdin pipe is a promise tickmarkr can never
57
64
  // keep; codex exec appends a piped stdin as a <stdin> block (`codex exec --help`) and blocks on a
58
65
  // read that never EOFs. One spawn site covers every adapter (D-06).
59
- // v1.22 T3: seal herdr control vars so worker/judge/review/consult children cannot reach the
60
- // operator's herdr (OBS-17 watch-tab leak class). process.env of the daemon is untouched.
66
+ // v1.22 T3 / OBS-843: seal host control vars so worker/judge/review/consult children cannot
67
+ // address the operator's herdr or Orca UI. process.env of the daemon is untouched.
61
68
  const p = spawn("bash", ["-lc", cmd], {
62
69
  cwd: slot.cwd,
63
70
  stdio: ["ignore", "pipe", "pipe"],
@@ -36,6 +36,8 @@ export interface SlotOpts {
36
36
  group?: string;
37
37
  label?: string;
38
38
  owned?: OwnedName;
39
+ /** Adapter id for execution surfaces whose liveness support is agent-specific. */
40
+ agent?: string;
39
41
  }
40
42
  export declare const OWNED_ROLES: readonly ["worker", "judge", "review", "consult", "watch", "other"];
41
43
  export type OwnedRole = (typeof OWNED_ROLES)[number];
@@ -23,6 +23,8 @@ export interface BaselineCommand {
23
23
  exitCode?: number;
24
24
  fingerprints: string[];
25
25
  missingCommand?: boolean;
26
+ /** Why a capture returned no verdict. */
27
+ invalidCause?: "ceiling-kill" | "resource-exhaustion";
26
28
  /** What this command actually took at capture, on a pristine tree. Absent in pre-v1.90 baselines. */
27
29
  durationMs?: number;
28
30
  /** Sum of the per-file durations named by the runner; null when its output names none. */
@@ -427,8 +427,9 @@ export function ceilingKillResult(gate, r, ceilingMs) {
427
427
  * shell asks it to finish faster than the thing it is measuring.
428
428
  */
429
429
  export const CAPTURE_CEILING_MS = 1_800_000;
430
- const invalidCaptureEntry = (durationMs, invalidatingLines = []) => ({
430
+ const invalidCaptureEntry = (durationMs, invalidCause, invalidatingLines = []) => ({
431
431
  infra: true,
432
+ invalidCause,
432
433
  fingerprints: [],
433
434
  durationMs,
434
435
  fileDurationSumMs: null,
@@ -463,7 +464,7 @@ export async function captureBaseline(cwd, commands) {
463
464
  console.error(`tickmarkr: baseline capture for "${name}" was killed at its ${CAPTURE_CEILING_MS}ms ceiling — `
464
465
  + `it recorded NO fingerprints, so nothing is forgiven and every gate will treat a pre-existing `
465
466
  + `failure as a fresh one. Raise the ceiling or shorten the command.`);
466
- base.commands[name] = invalidCaptureEntry(durationMs);
467
+ base.commands[name] = invalidCaptureEntry(durationMs, "ceiling-kill");
467
468
  continue;
468
469
  }
469
470
  const combinedOutput = r.stdout + "\n" + r.stderr;
@@ -479,7 +480,7 @@ export async function captureBaseline(cwd, commands) {
479
480
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
480
481
  + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
481
482
  + `First invalidating line: ${invalidatingLines[0]}`);
482
- base.commands[name] = invalidCaptureEntry(durationMs, invalidatingLines);
483
+ base.commands[name] = invalidCaptureEntry(durationMs, "resource-exhaustion", invalidatingLines);
483
484
  continue;
484
485
  }
485
486
  base.commands[name] = {
@@ -591,6 +592,19 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
591
592
  record(killed);
592
593
  continue;
593
594
  }
595
+ const runner = shellToken(cmd) ?? name;
596
+ if (entry?.missingCommand === true || r.code === 127) {
597
+ const cause = entry?.missingCommand === true && r.code !== 127
598
+ ? "the baseline capture recorded this runner as missing"
599
+ : `the head command exited 127${/\bENOENT\b/i.test(`${r.stdout}\n${r.stderr}`) ? " (spawn ENOENT)" : ""}`;
600
+ record({
601
+ gate: name,
602
+ pass: false,
603
+ details: `unreadable — ${cause}; runner ${JSON.stringify(runner)} did not produce a trustworthy verdict`,
604
+ meta: { unreadable: true, runner },
605
+ });
606
+ continue;
607
+ }
594
608
  if (r.code === 0) {
595
609
  record({ gate: name, pass: true, details: "exit 0" });
596
610
  continue;
@@ -629,8 +643,11 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
629
643
  // entries with no exitCode keep the `?? 1` red default both readers share (merge.ts:131).
630
644
  const baselineRed = entry?.infra !== true && (entry?.exitCode ?? 1) !== 0;
631
645
  if (!failing.length && !baselineRed) {
646
+ const recordedCause = entry?.invalidCause === "resource-exhaustion" || entry?.invalidatingLines?.length
647
+ ? `was invalidated by its recorded process/resource-exhaustion cause${entry.invalidatingLines?.[0] ? ` (${entry.invalidatingLines[0]})` : ""}`
648
+ : "was killed at its ceiling";
632
649
  const closed = entry?.infra === true
633
- ? `the baseline capture for this command was killed at its ceiling and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
650
+ ? `the baseline capture for this command ${recordedCause} and recorded no verdict, so nothing here is forgivable — it now exits ${r.code} with no recognizable failure lines — failing closed`
634
651
  : `command was green at baseline but now exits ${r.code} with no recognizable failure lines — failing closed`;
635
652
  const evidence = unrecognizedEvidence(raw);
636
653
  record({
@@ -57,9 +57,15 @@ export declare function captureLlmOutput<T>(run: () => Promise<T>): Promise<{
57
57
  value: T;
58
58
  outputs: string[];
59
59
  }>;
60
+ export interface LlmRunResult {
61
+ output: string;
62
+ exitCode?: number;
63
+ timedOut: boolean;
64
+ }
60
65
  export declare function runHeadless(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, timeoutMs?: number): Promise<string>;
61
66
  export declare function runViaDriver(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via: LlmVia, timeoutMs?: number): Promise<string>;
62
67
  export declare function dewrapPaneVerdict(out: string, nonce: string): string;
68
+ export declare function runLlmDetailed(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<LlmRunResult>;
63
69
  export declare function runLlm(adapter: WorkerAdapter, model: string, prompt: string, cwd: string, via?: LlmVia, timeoutMs?: number): Promise<string>;
64
70
  export declare function extractJson<T>(raw: string): T | null;
65
71
  /** Fable F3: verdict JSON must echo the call nonce — skip unbound or mismatched objects. */
package/dist/gates/llm.js CHANGED
@@ -129,21 +129,24 @@ export async function captureLlmOutput(run) {
129
129
  const value = await llmOutputCapture.run(outputs, run);
130
130
  return { value, outputs };
131
131
  }
132
- export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
132
+ async function runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs = 300000) {
133
133
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
134
134
  try {
135
135
  const pf = join(dir, "prompt.md");
136
136
  writeFileSync(pf, prompt);
137
137
  const r = await sh(adapter.headlessCommand(pf, model), cwd, timeoutMs);
138
- return r.stdout + "\n" + r.stderr;
138
+ return { output: r.stdout + "\n" + r.stderr, exitCode: r.code, timedOut: r.timedOut === true };
139
139
  }
140
140
  finally {
141
141
  rmSync(dir, { recursive: true, force: true });
142
142
  }
143
143
  }
144
+ export async function runHeadless(adapter, model, prompt, cwd, timeoutMs = 300000) {
145
+ return (await runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs)).output;
146
+ }
144
147
  // v1.1 default path: the same headless CLI call, but dispatched through the driver
145
148
  // as a visible named agent (herdr pane), with the quote-split completion wrapper.
146
- export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
149
+ async function runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
147
150
  const dir = mkdtempSync(join(tmpdir(), "tickmarkr-llm-"));
148
151
  let slot;
149
152
  let accountant;
@@ -176,6 +179,7 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
176
179
  // false-complete — same guard the worker path uses (daemon.ts:330-331).
177
180
  const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
178
181
  let out;
182
+ let timedOut = false;
179
183
  const gatePrompt = prompt.startsWith("TICKMARKR-JUDGE") || prompt.startsWith("TICKMARKR-REVIEW");
180
184
  if (!gatePrompt) {
181
185
  await via.driver.waitOutput(slot, exitPattern, timeoutMs, { regex: true });
@@ -241,8 +245,14 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
241
245
  break;
242
246
  }
243
247
  }
248
+ timedOut = Date.now() - startedAt >= timeoutMs && !new RegExp(exitPattern).test(out);
244
249
  }
245
- return dewrapPaneVerdict(out, nonce);
250
+ const exitCode = Number(new RegExp(`TICKMARKR_EXIT_${nonce}:(\\d+)`).exec(out)?.[1]);
251
+ return {
252
+ output: dewrapPaneVerdict(out, nonce),
253
+ ...(Number.isFinite(exitCode) ? { exitCode } : {}),
254
+ timedOut,
255
+ };
246
256
  }
247
257
  finally {
248
258
  try {
@@ -256,6 +266,9 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
256
266
  }
257
267
  }
258
268
  }
269
+ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
270
+ return (await runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
271
+ }
259
272
  // OBS-155: a TUI renders the verdict as a bullet and HARD-wraps it at pane width with a 2-space
260
273
  // continuation indent, splitting words mid-token — so literal newlines land inside JSON string
261
274
  // literals and ZERO lines begin with `{`. `--source recent-unwrapped` cannot undo it: the wrap is
@@ -315,12 +328,15 @@ export function dewrapPaneVerdict(out, nonce) {
315
328
  }
316
329
  return out;
317
330
  }
331
+ export async function runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
332
+ const result = await (via
333
+ ? runViaDriverDetailed(adapter, model, prompt, cwd, via, timeoutMs)
334
+ : runHeadlessDetailed(adapter, model, prompt, cwd, timeoutMs));
335
+ llmOutputCapture.getStore()?.push(result.output);
336
+ return result;
337
+ }
318
338
  export async function runLlm(adapter, model, prompt, cwd, via, timeoutMs = 300000) {
319
- const out = await (via
320
- ? runViaDriver(adapter, model, prompt, cwd, via, timeoutMs)
321
- : runHeadless(adapter, model, prompt, cwd, timeoutMs));
322
- llmOutputCapture.getStore()?.push(out);
323
- return out;
339
+ return (await runLlmDetailed(adapter, model, prompt, cwd, via, timeoutMs)).output;
324
340
  }
325
341
  export function extractJson(raw) {
326
342
  const fenced = [...raw.matchAll(/```json\s*\n([\s\S]*?)```/g)].at(-1);
@@ -41,13 +41,15 @@ export type TaskDiffMeasurement = {
41
41
  readonly fullMeasurement: ArtifactDiffMeasurement;
42
42
  readonly capMeasurement: ArtifactDiffMeasurement;
43
43
  };
44
- export declare function fetchTaskDiff(worktree: string, baseRef: string): Promise<TaskDiffMeasurement>;
44
+ export declare function fetchTaskDiff(worktree: string, baseRef: string, files?: readonly string[]): Promise<TaskDiffMeasurement>;
45
45
  export declare function checkDiffCap(gate: string, measured: number, cap: number, prefix?: string): GateResult | null;
46
46
  /** Apply the strict reviewable-logic cap and the finite, larger capture cap independently. */
47
47
  export declare function checkTaskDiffCaps(gate: string, measured: Pick<TaskDiffMeasurement, "logicBytes" | "captureBytes">, logicCap: number, prefix?: string): GateResult | null;
48
48
  export declare function isDiffCapPark(result: GateResult): boolean;
49
49
  export declare function diffCapParkReason(results: GateResult[]): string | null;
50
50
  export declare function modelId(model: string): string;
51
+ /** Provider identity comes from the served model, not a gateway adapter's stamped vendor. */
52
+ export declare function modelProvider(model: string, fallback?: string): string;
51
53
  export declare function pickReviewer(author: Assignment, channels: BillingChannel[], exclude?: string[], // v1.1 failover: reviewer channels that already produced garbage for this task
52
54
  prefer?: string[], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
53
55
  floor?: Tier): BillingChannel | null;