tickmarkr 2.2.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/README.md +10 -9
  2. package/dist/adapters/model-lints.js +9 -0
  3. package/dist/adapters/pi.d.ts +1 -0
  4. package/dist/adapters/pi.js +15 -1
  5. package/dist/adapters/prompt.js +11 -3
  6. package/dist/adapters/registry.js +13 -4
  7. package/dist/adapters/types.d.ts +23 -1
  8. package/dist/adapters/types.js +43 -2
  9. package/dist/cli/commands/approve.d.ts +3 -7
  10. package/dist/cli/commands/approve.js +26 -20
  11. package/dist/cli/commands/beat.js +7 -4
  12. package/dist/cli/commands/doctor.d.ts +6 -2
  13. package/dist/cli/commands/doctor.js +79 -9
  14. package/dist/cli/commands/init.js +36 -21
  15. package/dist/cli/commands/plan.js +20 -3
  16. package/dist/cli/commands/report.js +37 -1
  17. package/dist/cli/commands/verify.d.ts +5 -0
  18. package/dist/cli/commands/verify.js +142 -25
  19. package/dist/cli/commands/version.d.ts +2 -1
  20. package/dist/cli/commands/version.js +25 -4
  21. package/dist/compile/collateral.js +15 -9
  22. package/dist/compile/native.js +5 -3
  23. package/dist/config/config.js +1 -1
  24. package/dist/drivers/index.d.ts +7 -0
  25. package/dist/drivers/index.js +40 -10
  26. package/dist/drivers/orca.d.ts +41 -1
  27. package/dist/drivers/orca.js +192 -15
  28. package/dist/drivers/subprocess.d.ts +3 -3
  29. package/dist/drivers/subprocess.js +16 -9
  30. package/dist/drivers/types.d.ts +2 -0
  31. package/dist/gates/baseline.d.ts +4 -0
  32. package/dist/gates/baseline.js +68 -16
  33. package/dist/gates/llm.d.ts +7 -1
  34. package/dist/gates/llm.js +66 -35
  35. package/dist/gates/review.d.ts +3 -1
  36. package/dist/gates/review.js +42 -12
  37. package/dist/gates/run-gates.d.ts +6 -0
  38. package/dist/gates/run-gates.js +25 -10
  39. package/dist/gates/verdict-cause.d.ts +6 -2
  40. package/dist/gates/verdict-cause.js +8 -4
  41. package/dist/run/consult.d.ts +7 -0
  42. package/dist/run/consult.js +21 -3
  43. package/dist/run/daemon.d.ts +14 -0
  44. package/dist/run/daemon.js +249 -27
  45. package/dist/run/git.d.ts +1 -0
  46. package/dist/run/git.js +4 -0
  47. package/dist/run/journal.d.ts +15 -2
  48. package/dist/run/journal.js +70 -12
  49. package/dist/run/supervision.d.ts +6 -0
  50. package/dist/run/supervision.js +29 -1
  51. package/dist/tui/ink/init-app.js +4 -4
  52. package/package.json +1 -1
  53. package/skills/tickmarkr-loop/SKILL.md +1 -0
  54. package/skills/tickmarkr-overseer/SKILL.md +88 -18
  55. package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
  56. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
  57. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
  58. package/skills/tickmarkr-overseer/scripts/watch-context.sh +35 -9
  59. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
package/README.md CHANGED
@@ -13,8 +13,9 @@ tickmarkr is a spec-driven orchestration harness for AI coding agent CLIs. You w
13
13
  acceptance criteria; the engine routes tasks to the best installed agent CLI (claude-code, codex,
14
14
  cursor-agent, opencode, grok, pi, kimi) by cost and capability, dispatches work in git worktrees for
15
15
  change isolation — as interactive TUIs when running under [herdr](https://herdr.dev), headless
16
- subprocesses otherwise, or in [Orca](https://onorca.dev) terminals when you name that driver
17
- yourself — and independently verifies each committed result by checking for no new
16
+ subprocesses otherwise, or in [Orca](https://onorca.dev) terminals when auto detects both Orca
17
+ markers (name that driver explicitly outside one) — and independently verifies each committed
18
+ result by checking for no new
18
19
  baseline failures per task, then strictly verifying the integration tip. Green tasks consolidate onto a
19
20
  `tickmarkr/<runId>` branch; merging to your mainline is always your call, never automated. Engage
20
21
  with full visibility into routing decisions, worker progress, and gate verdicts — or run headless
@@ -251,14 +252,14 @@ and first-attempt success rate. Cost reporting follows strict honesty rules and
251
252
  When running under [herdr](https://herdr.dev), tickmarkr creates a labeled pane-and-tab workspace
252
253
  for real-time visibility (optional — omit `--driver herdr` or run headless if preferred).
253
254
 
254
- ### Orca: an explicit-selection execution surface
255
+ ### Orca: a detected-or-named execution surface
255
256
 
256
- [Orca](https://onorca.dev) is the third execution surface, and the only one you must ask for by
257
- name: `--driver orca` or `driver: orca` in config. `--driver auto` never selects it — auto picks
258
- herdr when a herdr session is live and subprocess otherwise — so Orca is never inherited from an
259
- ambient environment variable, and an Orca that is installed but unreachable is not silently
260
- downgraded to a hidden subprocess worker either. Naming it is the whole gate; its runtime failures
261
- stay Orca's, reported as failures.
257
+ [Orca](https://onorca.dev) is the third execution surface. `auto` resolves herdr first when
258
+ `HERDR_ENV=1`, then Orca only when both Orca-authored markers `TERM_PROGRAM=Orca` and
259
+ `ORCA_TERMINAL_HANDLE` are present, then subprocess. This environment-only choice executes no
260
+ binary or runtime probe. Outside an Orca terminal, name it explicitly with `--driver orca` or
261
+ `driver: orca` in config. Once selected either way, an unreachable Orca stays a loud Orca driver
262
+ failure and is never silently replaced by a hidden subprocess worker.
262
263
 
263
264
  What Orca supplies is terminals. What tickmarkr keeps is everything that decides whether work
264
265
  ships: **it creates and owns the git worktree** for every task (Orca is told which checkout to bind
@@ -3,6 +3,7 @@ import { existsSync } from "node:fs";
3
3
  import { join } from "node:path";
4
4
  import { DEFAULT_CONFIG, TIER_RANK } from "../config/config.js";
5
5
  import { filesGlob } from "../graph/files-glob.js";
6
+ import { piModelVendor } from "./pi.js";
6
7
  import { buildTaskPrompt } from "./prompt.js";
7
8
  import { channelKey, MODEL_ID_RE } from "./types.js";
8
9
  import { resolveCatalogModel } from "./catalog-remote.js";
@@ -446,6 +447,14 @@ export function modelLints(cfg, health, adapters, opts) {
446
447
  const lints = [];
447
448
  for (const adapter of adapters) {
448
449
  const id = adapter.id;
450
+ if (id === "pi" && adapter.channels) {
451
+ for (const channel of adapter.channels(cfg)) {
452
+ const expected = cfg.tiers.pi?.modelOverrides?.[channel.model]?.vendor ?? piModelVendor(channel.model);
453
+ if (expected && channel.vendor !== expected) {
454
+ lints.push(`pi: ${channel.model} channel vendor ${channel.vendor} disagrees with provider vendor ${expected} — set tiers.pi.modelOverrides.${channel.model}.vendor to ${expected}`);
455
+ }
456
+ }
457
+ }
449
458
  if (!adapter.listModels) {
450
459
  // v1.90 / OBS-504: the seeds-stamped wording presumes a seeded tier table. agy ships routable
451
460
  // but UNCLASSIFIED (no listModels, no seed models) — for that shape the honest sentence names
@@ -6,4 +6,5 @@ export interface ServedModelDrift {
6
6
  }
7
7
  export declare function readPiServedModels(): ServedModelDrift[];
8
8
  export declare function servedModelNote(drifts?: ServedModelDrift[]): string;
9
+ export declare function piModelVendor(model: string): string | undefined;
9
10
  export declare const pi: WorkerAdapter;
@@ -118,6 +118,17 @@ export function servedModelNote(drifts = readPiServedModels()) {
118
118
  return "";
119
119
  return `served-model drift: ${drifts.map((d) => `pinned ${d.pinned} served ${d.served}`).join(", ")}`;
120
120
  }
121
+ const PI_PROVIDER_VENDORS = {
122
+ anthropic: "anthropic",
123
+ google: "google",
124
+ openai: "openai",
125
+ "openai-codex": "openai",
126
+ xai: "xai",
127
+ zai: "zhipu",
128
+ };
129
+ export function piModelVendor(model) {
130
+ return PI_PROVIDER_VENDORS[model.split("/", 1)[0]];
131
+ }
121
132
  export const pi = {
122
133
  id: "pi",
123
134
  // FLEET-04: cross-vendor review honesty — GLM's provider (pi's own label is "zai"; either is
@@ -140,7 +151,10 @@ export const pi = {
140
151
  const note = `auth verified via pi --list-models (free; auth-filtered by pi)${drift ? `; ${drift}` : ""}`;
141
152
  return { ...h, servable: parsePiModels(r.stdout || ""), note };
142
153
  },
143
- channels: (cfg) => channelsFromConfig("pi", cfg),
154
+ channels: (cfg) => channelsFromConfig("pi", cfg).map((channel) => ({
155
+ ...channel,
156
+ vendor: cfg.tiers.pi?.modelOverrides?.[channel.model]?.vendor ?? piModelVendor(channel.model) ?? channel.vendor,
157
+ })),
144
158
  // v1.65 T3: every flag the command builders below hardcode — verified in `pi --help` 2026-07-22.
145
159
  hardcodedFlags: { binary: "pi", flags: ["-p", "--approve", "--model"] },
146
160
  // --approve: pi's per-directory trust prompt would stall fresh worktrees (herdr scrapes the dialog
@@ -1,4 +1,4 @@
1
- import { mkdirSync, writeFileSync } from "node:fs";
1
+ import { copyFileSync, existsSync, mkdirSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { renderAcceptanceItem } from "../graph/schema.js";
4
4
  import { classifyVerdictCause } from "../gates/verdict-cause.js";
@@ -30,8 +30,16 @@ TICKMARKR_RESULT_${nonce} {"ok":true|false,"summary":"<one sentence>","deviation
30
30
  `;
31
31
  }
32
32
  export function writePrompt(dir, task, attempt, feedback = "", nonce = "") {
33
- const p = join(dir, "prompts", `${task.id}-a${attempt}.md`);
34
- mkdirSync(join(dir, "prompts"), { recursive: true });
33
+ const prompts = join(dir, "prompts");
34
+ const p = join(prompts, `${task.id}-a${attempt}.md`);
35
+ mkdirSync(prompts, { recursive: true });
36
+ if (existsSync(p)) {
37
+ let engagement = 0;
38
+ let archive = join(prompts, `${task.id}-a${attempt}-engagement-${engagement}.md`);
39
+ while (existsSync(archive))
40
+ archive = join(prompts, `${task.id}-a${attempt}-engagement-${++engagement}.md`);
41
+ copyFileSync(p, archive);
42
+ }
35
43
  writeFileSync(p, buildTaskPrompt(task, feedback, nonce));
36
44
  return p;
37
45
  }
@@ -11,7 +11,7 @@ import { FakeAdapter } from "./fake.js";
11
11
  import { parseWorkerResult } from "./prompt.js";
12
12
  import { sealHerdrEnv } from "../drivers/subprocess.js";
13
13
  import { catalogEntries, isNativeCliDrive, projectCliEntries, SHIPPED_CLI_CATALOG, } from "./catalog.js";
14
- import { channelKey, channelsFromConfig, modelAuthed, MODEL_ID_RE, QUOTA_RE, shq, } from "./types.js";
14
+ import { channelKey, channelsFromConfig, modelAuthed, MODEL_ID_RE, MODEL_PROBE_ERRORS, QUOTA_RE, shq, } from "./types.js";
15
15
  // Compatibility projection for callers/tests that only need shipped advisory names. The literal
16
16
  // list is gone: both advisory and routable catalog views derive from SHIPPED_CLI_CATALOG. Native
17
17
  // definitions (claudeCode, codex, cursorAgent, opencode, pi, grok, kimi) are owned by catalog.ts;
@@ -410,6 +410,12 @@ function probeFailure(code, stdout, stderr, timedOut, timeoutMs = MODEL_PROBE_TI
410
410
  ? reasonTail(output) || `probe exited ${code}`
411
411
  : undefined;
412
412
  }
413
+ const PROBE_ERROR_RE = new RegExp(`\\b(${MODEL_PROBE_ERRORS.join("|")})\\b`);
414
+ function probeError(code, stdout, stderr, timedOut) {
415
+ if (timedOut || code === 0)
416
+ return undefined;
417
+ return PROBE_ERROR_RE.exec(`${stderr}\n${stdout}`)?.[1];
418
+ }
413
419
  const probeModelStatus = (v) => v.authed ? "ok" : v.reason?.includes("timed out") ? "timeout" : "failed";
414
420
  // v1.21: one bounded, headless call per configured model; detected-but-unclassified models never enter this loop.
415
421
  export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
@@ -436,7 +442,7 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
436
442
  const attempt = async (model, retry) => {
437
443
  const t0 = Date.now();
438
444
  const probedAt = new Date().toISOString();
439
- const v = (authed, reason) => ({ authed, ...(reason !== undefined ? { reason } : {}), probedAt, durationMs: Date.now() - t0 });
445
+ const v = (authed, reason, error) => ({ authed, ...(reason !== undefined ? { reason } : {}), ...(error ? { probeError: error } : {}), probedAt, durationMs: Date.now() - t0 });
440
446
  try {
441
447
  if (typeof a.headlessCommand !== "function")
442
448
  return { verdict: v(false, "headless probe unavailable"), timedOut: false };
@@ -455,8 +461,11 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
455
461
  return { verdict: v(true), timedOut: false };
456
462
  if (!retry)
457
463
  return { verdict: null, timedOut: r.timedOut === true };
464
+ const error = probeError(r.code, r.stdout, r.stderr, r.timedOut);
458
465
  return {
459
- verdict: r.timedOut && retry.firstTimedOut ? v(false, `probe timed out twice (${MODEL_PROBE_TIMEOUT_MS}ms)`) : v(false, reason),
466
+ verdict: error
467
+ ? v(priorModelAuth?.[model]?.authed === true, undefined, error)
468
+ : r.timedOut && retry.firstTimedOut ? v(false, `probe timed out twice (${MODEL_PROBE_TIMEOUT_MS}ms)`) : v(false, reason),
460
469
  timedOut: r.timedOut === true,
461
470
  };
462
471
  }
@@ -605,7 +614,7 @@ export function modelAuthExclusions(cfg, adapters, health) {
605
614
  if (modelAuthed(h, c.model, cfg.routing.allowUnverifiedModels))
606
615
  continue;
607
616
  if (v?.authed === false)
608
- out.push({ key: channelKey(c), adapter: a.id, reason: v.reason ?? "probe failed", probedAt: v.probedAt });
617
+ out.push({ key: channelKey(c), adapter: a.id, reason: v.probeError ? `probe-error (${v.probeError})` : v.reason ?? "probe failed", probedAt: v.probedAt });
609
618
  else
610
619
  out.push({ key: channelKey(c), adapter: a.id, reason: "no model auth verdict — run tickmarkr doctor", probedAt: "not recorded" });
611
620
  }
@@ -23,9 +23,12 @@ export interface BillingChannel {
23
23
  channel: "sub" | "api";
24
24
  tier: Tier;
25
25
  }
26
+ export declare const MODEL_PROBE_ERRORS: readonly ["EMFILE", "EAGAIN", "ENFILE", "ENOMEM", "ENOSPC"];
27
+ export type ModelProbeError = typeof MODEL_PROBE_ERRORS[number];
26
28
  export interface ModelAuth {
27
29
  authed: boolean;
28
30
  reason?: string;
31
+ probeError?: ModelProbeError;
29
32
  probedAt: string;
30
33
  }
31
34
  export interface AuthHealth {
@@ -113,7 +116,23 @@ export declare const TrustDialogSchema: z.ZodUnion<readonly [z.ZodObject<{
113
116
  reason: z.ZodString;
114
117
  }, z.core.$strict>]>;
115
118
  export declare function matchesTrustDialog(paneText: string, dialog: TrustDialog): dialog is CapturedTrustDialog;
119
+ export declare const ADAPTER_PROMPT_GLYPHS: {
120
+ readonly "claude-code": "❯";
121
+ readonly codex: "›";
122
+ readonly "cursor-agent": ">";
123
+ readonly opencode: ">";
124
+ readonly pi: ">";
125
+ readonly grok: ">";
126
+ readonly kimi: ">";
127
+ readonly omp: ">";
128
+ readonly agy: ">";
129
+ readonly "prime-agent": ">";
130
+ readonly fake: ">";
131
+ };
132
+ export type PromptGlyph = (typeof ADAPTER_PROMPT_GLYPHS)[keyof typeof ADAPTER_PROMPT_GLYPHS];
133
+ export declare function declaredPromptGlyphForAdapter(adapterId: string): PromptGlyph | undefined;
116
134
  export interface InputBox {
135
+ promptGlyph?: PromptGlyph;
117
136
  fingerprint: string;
118
137
  match?(paneText: string): boolean;
119
138
  emptyMatch?(paneText: string): boolean;
@@ -121,8 +140,11 @@ export interface InputBox {
121
140
  launchCommand?(command: string): boolean;
122
141
  readinessTimeoutMs?: number;
123
142
  }
124
- export declare function declareInputBox(adapterId: string, inputBox: InputBox): InputBox;
143
+ export declare function declareInputBox(adapterId: string, inputBox: Omit<InputBox, "promptGlyph"> & {
144
+ promptGlyph?: PromptGlyph;
145
+ }): InputBox;
125
146
  export declare function declaredInputBoxForWorkerName(workerName: string): InputBox | undefined;
147
+ export declare function declaredPromptGlyphForWorkerName(workerName: string): PromptGlyph | undefined;
126
148
  export declare function matchesInputBox(paneText: string, inputBox: InputBox): boolean;
127
149
  export declare function matchesEmptyInputBox(paneText: string, inputBox: InputBox): boolean;
128
150
  export declare function matchesOccupiedInputBox(paneText: string, inputBox: InputBox): boolean;
@@ -23,6 +23,7 @@ export function addUsage(a, b) {
23
23
  reasoning: add(a.reasoning, b.reasoning),
24
24
  };
25
25
  }
26
+ export const MODEL_PROBE_ERRORS = ["EMFILE", "EAGAIN", "ENFILE", "ENOMEM", "ENOSPC"];
26
27
  export function modelAuthed(health, model, allowUnverifiedModels = false) {
27
28
  const authed = health?.modelAuth?.[model]?.authed;
28
29
  return authed === true || (authed === undefined && allowUnverifiedModels);
@@ -128,15 +129,55 @@ export function matchesTrustDialog(paneText, dialog) {
128
129
  return false;
129
130
  return dialog.fingerprint.trim().length > 0 && paneText.includes(dialog.fingerprint);
130
131
  }
132
+ // v1.75 T1 / OBS-136: an adapter whose steady-state TUI presents a bordered input box declares
133
+ // one distinctive pane-text fingerprint. The herdr driver associates declarations with worker
134
+ // slots by the existing adapter-bearing dispatch name before that name is canonicalized.
135
+ // v1.77 / OBS-142: launchCommand identifies the adapter-owned bootstrap command that necessarily
136
+ // precedes the box; that command settles on a clean shell line, while every other delivery waits
137
+ // for the declared box itself. readinessTimeoutMs bounds that evidence loop per adapter.
138
+ // v1.85 T5 / OBS-140: typed delivery is licensed by DECLARED states, never by transcript shape.
139
+ // `match` is the box painted, `emptyMatch` the box carrying nothing, `occupiedMatch` the box still
140
+ // holding a prompt. An adapter may pin `occupiedMatch` directly, or leave it derived from the other
141
+ // two — but an adapter that declares neither cannot acknowledge a submission and is refused.
142
+ // OBS-620: prompt glyphs are adapter facts, just like the input-box matchers below. Keep the shipped
143
+ // set in one shell-readable table: the overseer receipt cannot import TypeScript, but it reads these
144
+ // exact declarations from src/ in a checkout or dist/ in an installed package. A newly shipped
145
+ // adapter therefore has one visible place where omission can be enumerated and refused.
146
+ export const ADAPTER_PROMPT_GLYPHS = {
147
+ "claude-code": "❯",
148
+ "codex": "›",
149
+ "cursor-agent": ">",
150
+ "opencode": ">",
151
+ "pi": ">",
152
+ "grok": ">",
153
+ "kimi": ">",
154
+ "omp": ">",
155
+ "agy": ">",
156
+ "prime-agent": ">",
157
+ // Test-only, but declared so a fake interactive slot exercises the same fail-closed contract.
158
+ "fake": ">",
159
+ };
160
+ export function declaredPromptGlyphForAdapter(adapterId) {
161
+ return ADAPTER_PROMPT_GLYPHS[adapterId];
162
+ }
131
163
  const inputBoxes = new Map();
132
164
  export function declareInputBox(adapterId, inputBox) {
133
- inputBoxes.set(adapterId, inputBox);
134
- return inputBox;
165
+ const promptGlyph = inputBox.promptGlyph ?? declaredPromptGlyphForAdapter(adapterId);
166
+ // The shipped-adapter registry is audited separately against ADAPTER_PROMPT_GLYPHS. Keep this
167
+ // declaration helper open to synthetic and extension adapters: Herdr's input-state tests register
168
+ // those at module load, and prompt glyphs are not part of its typed-delivery decision.
169
+ const declared = promptGlyph === undefined ? inputBox : { ...inputBox, promptGlyph };
170
+ inputBoxes.set(adapterId, declared);
171
+ return declared;
135
172
  }
136
173
  export function declaredInputBoxForWorkerName(workerName) {
137
174
  const adapterId = /^.+-worker-(.+)-a\d+-.+$/.exec(workerName)?.[1];
138
175
  return adapterId === undefined ? undefined : inputBoxes.get(adapterId);
139
176
  }
177
+ export function declaredPromptGlyphForWorkerName(workerName) {
178
+ const adapterId = /^.+-worker-(.+)-a\d+-.+$/.exec(workerName)?.[1];
179
+ return adapterId === undefined ? undefined : declaredPromptGlyphForAdapter(adapterId);
180
+ }
140
181
  export function matchesInputBox(paneText, inputBox) {
141
182
  return inputBox.match?.(paneText) ?? paneText.includes(inputBox.fingerprint);
142
183
  }
@@ -9,19 +9,15 @@ export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
9
9
  export declare const APPROVAL_ENACTS: Record<ApprovalDisposition, string>;
10
10
  export declare function approvalDispositionForRelease(release: unknown): ApprovalDisposition;
11
11
  export type ApprovalStatus = "deferred-live" | "recorded-no-owner";
12
- /** Which run, and whether a LIVE daemon owns THAT run — the two facts an enactment sentence needs. */
12
+ /** The requested run plus the different live run currently blocking its repository, when present. */
13
13
  export interface ApprovalRunOwner {
14
14
  runId: string;
15
15
  live: boolean;
16
+ blockingRunId?: string;
16
17
  }
17
18
  /** The same read `approve` performs, for surfaces that must predict an enactment before writing. */
18
19
  export declare function approvalRunOwner(cwd: string, runId: string): ApprovalRunOwner;
19
- /**
20
- * The one sentence that says who enacts this release and what it buys. A live owner's approval is
21
- * already scheduled — it rides that daemon's next task boundary — so it must NOT be told to resume:
22
- * a second run in the same repository is forbidden, and it would contend for the live daemon's
23
- * graph.lock over an approval that has already dispatched.
24
- */
20
+ /** The one sentence that says who enacts this release and what it buys. */
25
21
  export declare function approvalEnactment(token: ApprovalDisposition, run: ApprovalRunOwner): string;
26
22
  /** The production command registered in COMMANDS; its returned bytes are what the CLI prints. */
27
23
  export declare function approve(argv: string[], cwd?: string): Promise<string>;
@@ -11,7 +11,7 @@ export const APPROVAL_DISPOSITIONS = ["dispatch", "waive-gate", "re-dispatch", "
11
11
  export const APPROVAL_ENACTS = {
12
12
  dispatch: "dispatch it",
13
13
  "waive-gate": "continue past the approved gate",
14
- "re-dispatch": "re-dispatch against the full gate suite",
14
+ "re-dispatch": "re-dispatch against the full gate suite only if re-running the whole declared battery on the parked commit before any worker is red",
15
15
  "fund-fixed-attempt": "dispatch a fixed attempt carrying the findings",
16
16
  "fresh-budget": "dispatch it on a fresh attempt budget",
17
17
  };
@@ -32,21 +32,27 @@ import { acquireApprovalSerialization, runLockOwner } from "../../run/lock.js";
32
32
  // falsehood in a new shape. Liveness itself comes from lock.ts's runLockOwner (the same inspect() the
33
33
  // acquire/unlock decision table uses), never a second `process.kill(pid, 0)`, and never the lock
34
34
  // FILE's presence: a stale lock whose recorded pid is dead is not a live run.
35
- const ownedByLiveDaemon = (owner, runId) => ({ runId, live: owner?.live === true && owner.runId === runId });
35
+ const ownedByLiveDaemon = (owner, runId) => {
36
+ const run = { runId, live: owner?.live === true && owner.runId === runId };
37
+ if (owner?.live === true && owner.runId !== undefined && owner.runId !== runId) {
38
+ // Preserve the shipped enumerable { runId, live } shape while carrying the third state.
39
+ Object.defineProperty(run, "blockingRunId", { value: owner.runId });
40
+ }
41
+ return run;
42
+ };
36
43
  /** The same read `approve` performs, for surfaces that must predict an enactment before writing. */
37
44
  export function approvalRunOwner(cwd, runId) {
38
45
  return ownedByLiveDaemon(runLockOwner(cwd), runId);
39
46
  }
40
- /**
41
- * The one sentence that says who enacts this release and what it buys. A live owner's approval is
42
- * already scheduled — it rides that daemon's next task boundary — so it must NOT be told to resume:
43
- * a second run in the same repository is forbidden, and it would contend for the live daemon's
44
- * graph.lock over an approval that has already dispatched.
45
- */
47
+ /** The one sentence that says who enacts this release and what it buys. */
46
48
  export function approvalEnactment(token, run) {
47
- return run.live
48
- ? `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}`
49
- : `run \`tickmarkr resume ${run.runId}\` to ${APPROVAL_ENACTS[token]}`;
49
+ if (run.live) {
50
+ return `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}`;
51
+ }
52
+ if (run.blockingRunId) {
53
+ return `release recorded; live run \`${run.blockingRunId}\` holds the repository lock, so resume \`${run.runId}\` after it ends to ${APPROVAL_ENACTS[token]}`;
54
+ }
55
+ return `run \`tickmarkr resume ${run.runId}\` to ${APPROVAL_ENACTS[token]}`;
50
56
  }
51
57
  /** The production command registered in COMMANDS; its returned bytes are what the CLI prints. */
52
58
  export async function approve(argv, cwd = process.cwd()) {
@@ -79,6 +85,7 @@ export async function approve(argv, cwd = process.cwd()) {
79
85
  const lastHuman = events[lastHumanIndex];
80
86
  const capPark = lastHuman?.data.kind === ATTEMPT_CAP_RELEASE;
81
87
  const gateFailPark = lastHuman?.data.kind === "gate-fail";
88
+ const infraPark = lastHuman?.data.kind === "infra";
82
89
  const failedGate = gateFailPark ? failedGateForNewestPark(events, taskId, lastHumanIndex) : undefined;
83
90
  if (gateFailPark && !failedGate) {
84
91
  throw new Error(`task ${taskId} is parked on gate-fail but has no failed gate result on the newest park — refusing to infer one`);
@@ -98,8 +105,8 @@ export async function approve(argv, cwd = process.cwd()) {
98
105
  return disposition(cwd, runId, "fund-fixed-attempt", `upheld the reviewer for ${taskId} in ${runId} — by ${by}`, serialization.contended);
99
106
  }
100
107
  if (recheck) {
101
- if (!gateFailPark || !failedGate) {
102
- throw new Error(`--recheck applies to a gate-fail park; ${taskId}'s newest park is ${String(lastHuman?.data.kind ?? "none")} with failed gate ${failedGate ?? "none"} — refusing`);
108
+ if ((!gateFailPark || !failedGate) && !infraPark) {
109
+ throw new Error(`--recheck applies to a gate-fail or infra park; ${taskId}'s newest park is ${String(lastHuman?.data.kind ?? "none")} with failed gate ${failedGate ?? "none"} — refusing`);
103
110
  }
104
111
  journal.append("task-approved", taskId, {
105
112
  by,
@@ -108,7 +115,7 @@ export async function approve(argv, cwd = process.cwd()) {
108
115
  release: RECHECK_RELEASE,
109
116
  ...(reviewRoundCeiling === undefined ? {} : { reviewRoundCeiling }),
110
117
  });
111
- return disposition(cwd, runId, "re-dispatch", `re-checking ${taskId} in ${runId} — by ${by}; failed gate ${failedGate}; no gate marked satisfied`, serialization.contended);
118
+ return disposition(cwd, runId, "re-dispatch", `re-checking ${taskId} in ${runId} — by ${by}; ${failedGate ? `failed gate ${failedGate}` : "infra park"}; no gate marked satisfied`, serialization.contended);
112
119
  }
113
120
  if (waive) {
114
121
  if (!gateFailPark || !failedGate) {
@@ -153,12 +160,12 @@ export async function approve(argv, cwd = process.cwd()) {
153
160
  //
154
161
  // The ENACTMENT half of the message is completed here because only here is liveness known: the call
155
162
  // sites carry the decision, not the answer to who will act on it. `deferred-live` keeps its v1.89
156
- // token — machine consumers parse it — while its TEXT now names the boundary sweep, and the `resume`
157
- // field it used to carry unconditionally is withheld from the live branch it would misdirect.
163
+ // token — machine consumers parse it — while its TEXT now names the boundary sweep. A recovery
164
+ // command is emitted only with no live repository owner; a different run's live owner instead names
165
+ // the blocker and waits until it ends.
158
166
  function disposition(cwd, runId, token, message, contended) {
159
167
  const owner = runLockOwner(cwd);
160
168
  const run = ownedByLiveDaemon(owner, runId);
161
- const resume = `tickmarkr resume ${runId}`;
162
169
  const out = `approval disposition ${token}: ${message}; ${approvalEnactment(token, run)}`;
163
170
  if (!owner && !contended)
164
171
  return out;
@@ -166,9 +173,8 @@ function disposition(cwd, runId, token, message, contended) {
166
173
  const record = {
167
174
  status,
168
175
  disposition: token,
169
- // The recovery command is claimed only where it IS the enactment. A live owner's approval is
170
- // already scheduled, and a resume would contend for that daemon's graph.lock.
171
- ...(run.live ? {} : { resume }),
176
+ // Resume is safe to prescribe only when no live repository owner would contend with it.
177
+ ...(!run.live && !run.blockingRunId ? { resume: `tickmarkr resume ${runId}` } : {}),
172
178
  ...(owner?.pid === undefined ? {} : { ownerPid: owner.pid }),
173
179
  ...(owner?.runId === undefined ? {} : { ownerRunId: owner.runId }),
174
180
  };
@@ -1,7 +1,7 @@
1
1
  import { mkdirSync, renameSync, rmSync, writeFileSync } from "node:fs";
2
2
  import { dirname } from "node:path";
3
3
  import { tickmarkrDir } from "../../graph/graph.js";
4
- import { SUPERVISION_BEAT_MS, SUPERVISION_DEFAULT_THRESHOLD_PCT, SUPERVISION_STALE_MS, SUPERVISION_TIERS, beatSupervision, supervisionStandDownPath, } from "../../run/supervision.js";
4
+ import { SUPERVISION_BEAT_MS, SUPERVISION_DEFAULT_THRESHOLD_PCT, SUPERVISION_STALE_MS, SUPERVISION_TIERS, beatSupervision, resolveSupervisionRoot, supervisionStandDownPath, } from "../../run/supervision.js";
5
5
  // SUP-04: the writer side of supervision, as a VERB. `beatSupervision` and `SUPERVISION_BEAT_MS` shipped
6
6
  // with exactly one in-repo caller — the daemon, on one tier — so `status` printed
7
7
  // `orchestrator ARMED / overseer ABSENT / watch ABSENT` while a real overseer worked the run: two thirds
@@ -41,14 +41,17 @@ export async function beat(argv, cwd = process.cwd()) {
41
41
  throw new Error(`${named} needs --seat <identity> — a beat that names no seat arms a tier nobody occupies` +
42
42
  " (pass the seat's own pane id or agent name)");
43
43
  }
44
+ // A beat names repository state, never the caller's incidental directory. Resolution is read-only
45
+ // and happens before every write, so a non-repository invocation cannot create the state it claims.
46
+ const repoRoot = resolveSupervisionRoot(cwd);
44
47
  if (standDown)
45
- return standDownTier(cwd, named, seat);
48
+ return standDownTier(repoRoot, named, seat);
46
49
  // Clear a stand-down marker left by an earlier session BEFORE beating: a valid marker outranks every
47
50
  // beat that does not strictly follow it, and two writes landing in the same millisecond do not. An
48
51
  // uncleared marker would render DISARMED while this verb claimed ARMED, so the removal is unguarded
49
52
  // too — `force` makes the ordinary "no marker" case a no-op, and anything else is a real failure.
50
- rmSync(supervisionStandDownPath(cwd, named), { force: true, recursive: true });
51
- beatSupervision(cwd, named, seat, pct === undefined ? undefined : { armId: armId ?? seat, pct, thresholdPct });
53
+ rmSync(supervisionStandDownPath(repoRoot, named), { force: true, recursive: true });
54
+ beatSupervision(repoRoot, named, seat, pct === undefined ? undefined : { armId: armId ?? seat, pct, thresholdPct });
52
55
  return `${named} ARMED as ${seat} — beat again every ${SUPERVISION_BEAT_MS / 1_000}s; the tier reads STALE ${SUPERVISION_STALE_MS / 1_000}s after the last beat`;
53
56
  }
54
57
  /** `--seat <identity>` or `--seat=<identity>`; blank and missing are the same answer — none. */
@@ -19,13 +19,17 @@ export type DoctorOpts = {
19
19
  compact?: boolean;
20
20
  /** Test seam for the same `orca status --json` transport production invokes. */
21
21
  orcaStatusProbe?: (cwd: string, binary: string) => Promise<ShResult>;
22
+ /** Test seam for Orca's authoritative per-agent hook coverage listing. */
23
+ orcaHooksStatusProbe?: (cwd: string, binary: string) => Promise<ShResult>;
24
+ /** Environment used only to explain whether auto-selection applies in this terminal. */
25
+ orcaEnv?: NodeJS.ProcessEnv | Record<string, string | undefined>;
22
26
  /** Test seam for shell-path discovery; absence remains a normal doctor row, never an exception. */
23
27
  resolveOrcaBinary?: (cwd: string) => string | undefined;
24
28
  /** Test seam for the runner-owned JSON listing used by the acceptance-oracle report row. */
25
29
  listTests?: (cwd: string) => Promise<VitestListResult>;
26
30
  };
27
31
  type OrcaCapability = {
28
- verdict: "pass" | "fail";
32
+ verdict: "pass" | "fail" | "warn";
29
33
  detail: string;
30
34
  };
31
35
  /**
@@ -36,7 +40,7 @@ type OrcaCapability = {
36
40
  * Capability-row `detail` stays hermetic — never `OrcaError.message`, which embeds volatile CLI
37
41
  * stderr (Electron timestamps), so the row is byte-stable across runs.
38
42
  */
39
- export declare function probeOrcaCapability(cwd: string, opts?: Pick<DoctorOpts, "orcaStatusProbe" | "resolveOrcaBinary">): Promise<OrcaCapability>;
43
+ export declare function probeOrcaCapability(cwd: string, opts?: Pick<DoctorOpts, "orcaStatusProbe" | "orcaHooksStatusProbe" | "resolveOrcaBinary" | "orcaEnv">): Promise<OrcaCapability>;
40
44
  export declare function runnerIgnoreFinding(cwd: string): {
41
45
  verdict: "pass" | "warn";
42
46
  detail: string;
@@ -12,7 +12,7 @@ import { graphPath, loadGraph, tickmarkrDir, stateDirName } from "../../graph/gr
12
12
  import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
13
13
  import { loadConfig, overlayPreferShapes } from "../../config/config.js";
14
14
  import { HerdrDriver } from "../../drivers/herdr.js";
15
- import { parseEnvelope } from "../../drivers/orca.js";
15
+ import { ORCA_FIXTURE_VERSION, parseEnvelope, resolveOrcaCliBinary } from "../../drivers/orca.js";
16
16
  import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
17
17
  import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
18
18
  import { LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
@@ -25,6 +25,62 @@ export const LIVEBENCH_TABLE_MAX_AGE_DAYS = 90;
25
25
  const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
26
26
  const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
27
27
  const attentionRow = (text) => ` ${statusRow("warn", text)}`;
28
+ const ORCA_HOOK_ADAPTERS = {
29
+ claude: "claude-code",
30
+ "claude-code": "claude-code",
31
+ codex: "codex",
32
+ cursor: "cursor-agent",
33
+ "cursor-agent": "cursor-agent",
34
+ grok: "grok",
35
+ kimi: "kimi",
36
+ opencode: "opencode",
37
+ pi: "pi",
38
+ };
39
+ function orcaSelectionDetail(env) {
40
+ const inside = env.TERM_PROGRAM === "Orca" && !!env.ORCA_TERMINAL_HANDLE?.trim();
41
+ return inside
42
+ ? "auto picks orca in this Orca terminal"
43
+ : "not an Orca terminal; auto picks orca only inside one; use --driver orca";
44
+ }
45
+ async function orcaHookCoverage(cwd, binary, opts) {
46
+ // A supplied status transport is a hermetic boundary: do not escape it to a real binary for the
47
+ // second command. Tests (and embedders) that want coverage supply the matching hook transport.
48
+ if (opts.orcaHooksStatusProbe === undefined && opts.orcaStatusProbe !== undefined) {
49
+ return "hooks status unavailable";
50
+ }
51
+ let response;
52
+ try {
53
+ response = await (opts.orcaHooksStatusProbe ?? ((probeCwd, executable) => sh(`${shq(executable)} agent hooks status --json`, probeCwd, 10_000)))(cwd, binary);
54
+ if (response.code !== 0 || response.timedOut)
55
+ return "hooks status unavailable";
56
+ // Until the driver's response-family expansion lands, this still uses its one strict envelope
57
+ // parser. Coverage comes only from this command's `statuses[].state`; managedHooksPresent and
58
+ // local agent config files are deliberately not alternative oracles.
59
+ const envelope = parseEnvelope("status", response.stdout);
60
+ const statuses = envelope.result.statuses;
61
+ if (!Array.isArray(statuses))
62
+ return "hooks status unavailable";
63
+ const coverage = new Map();
64
+ for (const raw of statuses) {
65
+ if (typeof raw !== "object" || raw === null || Array.isArray(raw))
66
+ continue;
67
+ const row = raw;
68
+ const adapter = typeof row.agent === "string" ? ORCA_HOOK_ADAPTERS[row.agent] : undefined;
69
+ if (!adapter)
70
+ continue;
71
+ if (row.state === "installed")
72
+ coverage.set(adapter, "hooked");
73
+ else if (row.state === "not_installed")
74
+ coverage.set(adapter, "unhooked");
75
+ }
76
+ return coverage.size
77
+ ? `hooks: ${[...coverage].map(([adapter, state]) => `${adapter} ${state}`).join(", ")}`
78
+ : "hooks: no tickmarkr adapter status reported";
79
+ }
80
+ catch {
81
+ return "hooks status unavailable";
82
+ }
83
+ }
28
84
  /**
29
85
  * Orca's status body is deliberately interpreted by T1's one shared envelope parser. Doctor owns
30
86
  * only capability presentation: it may classify an absent executable, but it never invents a second
@@ -34,7 +90,9 @@ const attentionRow = (text) => ` ${statusRow("warn", text)}`;
34
90
  * stderr (Electron timestamps), so the row is byte-stable across runs.
35
91
  */
36
92
  export async function probeOrcaCapability(cwd, opts = {}) {
37
- const binary = opts.resolveOrcaBinary ? opts.resolveOrcaBinary(cwd) : resolveShellBinary("orca", cwd).resolved;
93
+ const binary = opts.resolveOrcaBinary
94
+ ? opts.resolveOrcaBinary(cwd)
95
+ : resolveOrcaCliBinary(cwd, { resolve: (bin, dir) => resolveShellBinary(bin, dir) });
38
96
  if (!binary)
39
97
  return { verdict: "fail", detail: "CLI not installed" };
40
98
  let response;
@@ -60,8 +118,18 @@ export async function probeOrcaCapability(cwd, opts = {}) {
60
118
  const reachable = typeof runtime === "object" && runtime !== null && !Array.isArray(runtime)
61
119
  ? runtime.reachable
62
120
  : undefined;
63
- if (reachable === true)
64
- return { verdict: "pass", detail: `runtime reachable (${envelope.runtimeId})` };
121
+ if (reachable === true) {
122
+ const hooks = await orcaHookCoverage(cwd, binary, opts);
123
+ const selection = orcaSelectionDetail(opts.orcaEnv ?? process.env);
124
+ const appVersion = typeof runtime === "object" && runtime !== null && !Array.isArray(runtime)
125
+ ? runtime.appVersion
126
+ : undefined;
127
+ if (typeof appVersion !== "string" || !appVersion) {
128
+ return { verdict: "warn", detail: `runtime reachable (${envelope.runtimeId}, appVersion absent; fixture pin ${ORCA_FIXTURE_VERSION}); ${hooks} — ${selection}` };
129
+ }
130
+ const detail = `runtime reachable (${envelope.runtimeId}, appVersion ${appVersion}; fixture pin ${ORCA_FIXTURE_VERSION}); ${hooks} — ${selection}`;
131
+ return { verdict: appVersion === ORCA_FIXTURE_VERSION ? "pass" : "warn", detail };
132
+ }
65
133
  if (reachable === false)
66
134
  return { verdict: "fail", detail: "CLI installed but runtime unreachable" };
67
135
  return { verdict: "fail", detail: "CLI installed but runtime probe failed — status carries no reachability proof" };
@@ -449,8 +517,8 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
449
517
  }
450
518
  if (trustNa.length)
451
519
  rows.push(` ${dim("=")} ${dim(`n/a (${trustNa.length}): ${trustNa.join(", ")}`)}`);
452
- // Orca is an explicit choice, so its health is capability information rather than an auto-routing
453
- // input. A failed probe never changes pickDriver's auto ordering or substitutes subprocess.
520
+ // Capability is diagnostic only: auto-selection is decided from terminal identity markers, never
521
+ // from this runtime probe, and a failed probe cannot silently substitute another driver.
454
522
  const orca = await probeOrcaCapability(cwd, opts);
455
523
  rows.push(legend("execution runtime:"));
456
524
  rows.push(alignedStatusRow(orca.verdict, "orca", orca.detail));
@@ -567,9 +635,11 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
567
635
  const probed = probedMs !== undefined ? dim(` ${(probedMs / 1000).toFixed(1)}s`) : "";
568
636
  const auth = !v
569
637
  ? dim("unknown")
570
- : v.authed
571
- ? `${ok("authed")}${probed}`
572
- : `${fail("unauthed:")} ${trunc(v.reason ?? "probe failed", 40)} (${dateOf(v.probedAt)})`;
638
+ : v.probeError
639
+ ? `${fail(`probe error (${v.probeError})`)}${probed}`
640
+ : v.authed
641
+ ? `${ok("authed")}${probed}`
642
+ : `${fail("unauthed:")} ${trunc(v.reason ?? "probe failed", 40)} (${dateOf(v.probedAt)})`;
573
643
  const d = disallowedBy({ adapter: a.id, model: m }, cfg.routing);
574
644
  const denied = d?.by === "deny" ? d.entry : "—";
575
645
  const pref = preferRanks({ adapter: a.id, model: m }, cfg).map((p) => `${p.shape}#${p.rank}`).join(",") || "—";