tickmarkr 2.2.0 → 2.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,6 +3,7 @@ import { existsSync } from "node:fs";
3
3
  import { join } from "node:path";
4
4
  import { DEFAULT_CONFIG, TIER_RANK } from "../config/config.js";
5
5
  import { filesGlob } from "../graph/files-glob.js";
6
+ import { piModelVendor } from "./pi.js";
6
7
  import { buildTaskPrompt } from "./prompt.js";
7
8
  import { channelKey, MODEL_ID_RE } from "./types.js";
8
9
  import { resolveCatalogModel } from "./catalog-remote.js";
@@ -446,6 +447,14 @@ export function modelLints(cfg, health, adapters, opts) {
446
447
  const lints = [];
447
448
  for (const adapter of adapters) {
448
449
  const id = adapter.id;
450
+ if (id === "pi" && adapter.channels) {
451
+ for (const channel of adapter.channels(cfg)) {
452
+ const expected = cfg.tiers.pi?.modelOverrides?.[channel.model]?.vendor ?? piModelVendor(channel.model);
453
+ if (expected && channel.vendor !== expected) {
454
+ lints.push(`pi: ${channel.model} channel vendor ${channel.vendor} disagrees with provider vendor ${expected} — set tiers.pi.modelOverrides.${channel.model}.vendor to ${expected}`);
455
+ }
456
+ }
457
+ }
449
458
  if (!adapter.listModels) {
450
459
  // v1.90 / OBS-504: the seeds-stamped wording presumes a seeded tier table. agy ships routable
451
460
  // but UNCLASSIFIED (no listModels, no seed models) — for that shape the honest sentence names
@@ -6,4 +6,5 @@ export interface ServedModelDrift {
6
6
  }
7
7
  export declare function readPiServedModels(): ServedModelDrift[];
8
8
  export declare function servedModelNote(drifts?: ServedModelDrift[]): string;
9
+ export declare function piModelVendor(model: string): string | undefined;
9
10
  export declare const pi: WorkerAdapter;
@@ -118,6 +118,17 @@ export function servedModelNote(drifts = readPiServedModels()) {
118
118
  return "";
119
119
  return `served-model drift: ${drifts.map((d) => `pinned ${d.pinned} served ${d.served}`).join(", ")}`;
120
120
  }
121
+ const PI_PROVIDER_VENDORS = {
122
+ anthropic: "anthropic",
123
+ google: "google",
124
+ openai: "openai",
125
+ "openai-codex": "openai",
126
+ xai: "xai",
127
+ zai: "zhipu",
128
+ };
129
+ export function piModelVendor(model) {
130
+ return PI_PROVIDER_VENDORS[model.split("/", 1)[0]];
131
+ }
121
132
  export const pi = {
122
133
  id: "pi",
123
134
  // FLEET-04: cross-vendor review honesty — GLM's provider (pi's own label is "zai"; either is
@@ -140,7 +151,10 @@ export const pi = {
140
151
  const note = `auth verified via pi --list-models (free; auth-filtered by pi)${drift ? `; ${drift}` : ""}`;
141
152
  return { ...h, servable: parsePiModels(r.stdout || ""), note };
142
153
  },
143
- channels: (cfg) => channelsFromConfig("pi", cfg),
154
+ channels: (cfg) => channelsFromConfig("pi", cfg).map((channel) => ({
155
+ ...channel,
156
+ vendor: cfg.tiers.pi?.modelOverrides?.[channel.model]?.vendor ?? piModelVendor(channel.model) ?? channel.vendor,
157
+ })),
144
158
  // v1.65 T3: every flag the command builders below hardcode — verified in `pi --help` 2026-07-22.
145
159
  hardcodedFlags: { binary: "pi", flags: ["-p", "--approve", "--model"] },
146
160
  // --approve: pi's per-directory trust prompt would stall fresh worktrees (herdr scrapes the dialog
@@ -1,4 +1,4 @@
1
- import { mkdirSync, writeFileSync } from "node:fs";
1
+ import { copyFileSync, existsSync, mkdirSync, writeFileSync } from "node:fs";
2
2
  import { join } from "node:path";
3
3
  import { renderAcceptanceItem } from "../graph/schema.js";
4
4
  import { classifyVerdictCause } from "../gates/verdict-cause.js";
@@ -30,8 +30,16 @@ TICKMARKR_RESULT_${nonce} {"ok":true|false,"summary":"<one sentence>","deviation
30
30
  `;
31
31
  }
32
32
  export function writePrompt(dir, task, attempt, feedback = "", nonce = "") {
33
- const p = join(dir, "prompts", `${task.id}-a${attempt}.md`);
34
- mkdirSync(join(dir, "prompts"), { recursive: true });
33
+ const prompts = join(dir, "prompts");
34
+ const p = join(prompts, `${task.id}-a${attempt}.md`);
35
+ mkdirSync(prompts, { recursive: true });
36
+ if (existsSync(p)) {
37
+ let engagement = 0;
38
+ let archive = join(prompts, `${task.id}-a${attempt}-engagement-${engagement}.md`);
39
+ while (existsSync(archive))
40
+ archive = join(prompts, `${task.id}-a${attempt}-engagement-${++engagement}.md`);
41
+ copyFileSync(p, archive);
42
+ }
35
43
  writeFileSync(p, buildTaskPrompt(task, feedback, nonce));
36
44
  return p;
37
45
  }
@@ -11,7 +11,7 @@ import { FakeAdapter } from "./fake.js";
11
11
  import { parseWorkerResult } from "./prompt.js";
12
12
  import { sealHerdrEnv } from "../drivers/subprocess.js";
13
13
  import { catalogEntries, isNativeCliDrive, projectCliEntries, SHIPPED_CLI_CATALOG, } from "./catalog.js";
14
- import { channelKey, channelsFromConfig, modelAuthed, MODEL_ID_RE, QUOTA_RE, shq, } from "./types.js";
14
+ import { channelKey, channelsFromConfig, modelAuthed, MODEL_ID_RE, MODEL_PROBE_ERRORS, QUOTA_RE, shq, } from "./types.js";
15
15
  // Compatibility projection for callers/tests that only need shipped advisory names. The literal
16
16
  // list is gone: both advisory and routable catalog views derive from SHIPPED_CLI_CATALOG. Native
17
17
  // definitions (claudeCode, codex, cursorAgent, opencode, pi, grok, kimi) are owned by catalog.ts;
@@ -410,6 +410,12 @@ function probeFailure(code, stdout, stderr, timedOut, timeoutMs = MODEL_PROBE_TI
410
410
  ? reasonTail(output) || `probe exited ${code}`
411
411
  : undefined;
412
412
  }
413
+ const PROBE_ERROR_RE = new RegExp(`\\b(${MODEL_PROBE_ERRORS.join("|")})\\b`);
414
+ function probeError(code, stdout, stderr, timedOut) {
415
+ if (timedOut || code === 0)
416
+ return undefined;
417
+ return PROBE_ERROR_RE.exec(`${stderr}\n${stdout}`)?.[1];
418
+ }
413
419
  const probeModelStatus = (v) => v.authed ? "ok" : v.reason?.includes("timed out") ? "timeout" : "failed";
414
420
  // v1.21: one bounded, headless call per configured model; detected-but-unclassified models never enter this loop.
415
421
  export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
@@ -436,7 +442,7 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
436
442
  const attempt = async (model, retry) => {
437
443
  const t0 = Date.now();
438
444
  const probedAt = new Date().toISOString();
439
- const v = (authed, reason) => ({ authed, ...(reason !== undefined ? { reason } : {}), probedAt, durationMs: Date.now() - t0 });
445
+ const v = (authed, reason, error) => ({ authed, ...(reason !== undefined ? { reason } : {}), ...(error ? { probeError: error } : {}), probedAt, durationMs: Date.now() - t0 });
440
446
  try {
441
447
  if (typeof a.headlessCommand !== "function")
442
448
  return { verdict: v(false, "headless probe unavailable"), timedOut: false };
@@ -455,8 +461,11 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
455
461
  return { verdict: v(true), timedOut: false };
456
462
  if (!retry)
457
463
  return { verdict: null, timedOut: r.timedOut === true };
464
+ const error = probeError(r.code, r.stdout, r.stderr, r.timedOut);
458
465
  return {
459
- verdict: r.timedOut && retry.firstTimedOut ? v(false, `probe timed out twice (${MODEL_PROBE_TIMEOUT_MS}ms)`) : v(false, reason),
466
+ verdict: error
467
+ ? v(priorModelAuth?.[model]?.authed === true, undefined, error)
468
+ : r.timedOut && retry.firstTimedOut ? v(false, `probe timed out twice (${MODEL_PROBE_TIMEOUT_MS}ms)`) : v(false, reason),
460
469
  timedOut: r.timedOut === true,
461
470
  };
462
471
  }
@@ -605,7 +614,7 @@ export function modelAuthExclusions(cfg, adapters, health) {
605
614
  if (modelAuthed(h, c.model, cfg.routing.allowUnverifiedModels))
606
615
  continue;
607
616
  if (v?.authed === false)
608
- out.push({ key: channelKey(c), adapter: a.id, reason: v.reason ?? "probe failed", probedAt: v.probedAt });
617
+ out.push({ key: channelKey(c), adapter: a.id, reason: v.probeError ? `probe-error (${v.probeError})` : v.reason ?? "probe failed", probedAt: v.probedAt });
609
618
  else
610
619
  out.push({ key: channelKey(c), adapter: a.id, reason: "no model auth verdict — run tickmarkr doctor", probedAt: "not recorded" });
611
620
  }
@@ -23,9 +23,12 @@ export interface BillingChannel {
23
23
  channel: "sub" | "api";
24
24
  tier: Tier;
25
25
  }
26
+ export declare const MODEL_PROBE_ERRORS: readonly ["EMFILE", "EAGAIN", "ENFILE", "ENOMEM", "ENOSPC"];
27
+ export type ModelProbeError = typeof MODEL_PROBE_ERRORS[number];
26
28
  export interface ModelAuth {
27
29
  authed: boolean;
28
30
  reason?: string;
31
+ probeError?: ModelProbeError;
29
32
  probedAt: string;
30
33
  }
31
34
  export interface AuthHealth {
@@ -23,6 +23,7 @@ export function addUsage(a, b) {
23
23
  reasoning: add(a.reasoning, b.reasoning),
24
24
  };
25
25
  }
26
+ export const MODEL_PROBE_ERRORS = ["EMFILE", "EAGAIN", "ENFILE", "ENOMEM", "ENOSPC"];
26
27
  export function modelAuthed(health, model, allowUnverifiedModels = false) {
27
28
  const authed = health?.modelAuth?.[model]?.authed;
28
29
  return authed === true || (authed === undefined && allowUnverifiedModels);
@@ -9,19 +9,15 @@ export type ApprovalDisposition = (typeof APPROVAL_DISPOSITIONS)[number];
9
9
  export declare const APPROVAL_ENACTS: Record<ApprovalDisposition, string>;
10
10
  export declare function approvalDispositionForRelease(release: unknown): ApprovalDisposition;
11
11
  export type ApprovalStatus = "deferred-live" | "recorded-no-owner";
12
- /** Which run, and whether a LIVE daemon owns THAT run — the two facts an enactment sentence needs. */
12
+ /** The requested run plus the different live run currently blocking its repository, when present. */
13
13
  export interface ApprovalRunOwner {
14
14
  runId: string;
15
15
  live: boolean;
16
+ blockingRunId?: string;
16
17
  }
17
18
  /** The same read `approve` performs, for surfaces that must predict an enactment before writing. */
18
19
  export declare function approvalRunOwner(cwd: string, runId: string): ApprovalRunOwner;
19
- /**
20
- * The one sentence that says who enacts this release and what it buys. A live owner's approval is
21
- * already scheduled — it rides that daemon's next task boundary — so it must NOT be told to resume:
22
- * a second run in the same repository is forbidden, and it would contend for the live daemon's
23
- * graph.lock over an approval that has already dispatched.
24
- */
20
+ /** The one sentence that says who enacts this release and what it buys. */
25
21
  export declare function approvalEnactment(token: ApprovalDisposition, run: ApprovalRunOwner): string;
26
22
  /** The production command registered in COMMANDS; its returned bytes are what the CLI prints. */
27
23
  export declare function approve(argv: string[], cwd?: string): Promise<string>;
@@ -32,21 +32,27 @@ import { acquireApprovalSerialization, runLockOwner } from "../../run/lock.js";
32
32
  // falsehood in a new shape. Liveness itself comes from lock.ts's runLockOwner (the same inspect() the
33
33
  // acquire/unlock decision table uses), never a second `process.kill(pid, 0)`, and never the lock
34
34
  // FILE's presence: a stale lock whose recorded pid is dead is not a live run.
35
- const ownedByLiveDaemon = (owner, runId) => ({ runId, live: owner?.live === true && owner.runId === runId });
35
+ const ownedByLiveDaemon = (owner, runId) => {
36
+ const run = { runId, live: owner?.live === true && owner.runId === runId };
37
+ if (owner?.live === true && owner.runId !== undefined && owner.runId !== runId) {
38
+ // Preserve the shipped enumerable { runId, live } shape while carrying the third state.
39
+ Object.defineProperty(run, "blockingRunId", { value: owner.runId });
40
+ }
41
+ return run;
42
+ };
36
43
  /** The same read `approve` performs, for surfaces that must predict an enactment before writing. */
37
44
  export function approvalRunOwner(cwd, runId) {
38
45
  return ownedByLiveDaemon(runLockOwner(cwd), runId);
39
46
  }
40
- /**
41
- * The one sentence that says who enacts this release and what it buys. A live owner's approval is
42
- * already scheduled — it rides that daemon's next task boundary — so it must NOT be told to resume:
43
- * a second run in the same repository is forbidden, and it would contend for the live daemon's
44
- * graph.lock over an approval that has already dispatched.
45
- */
47
+ /** The one sentence that says who enacts this release and what it buys. */
46
48
  export function approvalEnactment(token, run) {
47
- return run.live
48
- ? `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}`
49
- : `run \`tickmarkr resume ${run.runId}\` to ${APPROVAL_ENACTS[token]}`;
49
+ if (run.live) {
50
+ return `the live daemon enacts this at its next task boundary — it will ${APPROVAL_ENACTS[token]}`;
51
+ }
52
+ if (run.blockingRunId) {
53
+ return `release recorded; live run \`${run.blockingRunId}\` holds the repository lock, so resume \`${run.runId}\` after it ends to ${APPROVAL_ENACTS[token]}`;
54
+ }
55
+ return `run \`tickmarkr resume ${run.runId}\` to ${APPROVAL_ENACTS[token]}`;
50
56
  }
51
57
  /** The production command registered in COMMANDS; its returned bytes are what the CLI prints. */
52
58
  export async function approve(argv, cwd = process.cwd()) {
@@ -153,12 +159,12 @@ export async function approve(argv, cwd = process.cwd()) {
153
159
  //
154
160
  // The ENACTMENT half of the message is completed here because only here is liveness known: the call
155
161
  // sites carry the decision, not the answer to who will act on it. `deferred-live` keeps its v1.89
156
- // token — machine consumers parse it — while its TEXT now names the boundary sweep, and the `resume`
157
- // field it used to carry unconditionally is withheld from the live branch it would misdirect.
162
+ // token — machine consumers parse it — while its TEXT now names the boundary sweep. A recovery
163
+ // command is emitted only with no live repository owner; a different run's live owner instead names
164
+ // the blocker and waits until it ends.
158
165
  function disposition(cwd, runId, token, message, contended) {
159
166
  const owner = runLockOwner(cwd);
160
167
  const run = ownedByLiveDaemon(owner, runId);
161
- const resume = `tickmarkr resume ${runId}`;
162
168
  const out = `approval disposition ${token}: ${message}; ${approvalEnactment(token, run)}`;
163
169
  if (!owner && !contended)
164
170
  return out;
@@ -166,9 +172,8 @@ function disposition(cwd, runId, token, message, contended) {
166
172
  const record = {
167
173
  status,
168
174
  disposition: token,
169
- // The recovery command is claimed only where it IS the enactment. A live owner's approval is
170
- // already scheduled, and a resume would contend for that daemon's graph.lock.
171
- ...(run.live ? {} : { resume }),
175
+ // Resume is safe to prescribe only when no live repository owner would contend with it.
176
+ ...(!run.live && !run.blockingRunId ? { resume: `tickmarkr resume ${runId}` } : {}),
172
177
  ...(owner?.pid === undefined ? {} : { ownerPid: owner.pid }),
173
178
  ...(owner?.runId === undefined ? {} : { ownerRunId: owner.runId }),
174
179
  };
@@ -25,7 +25,7 @@ export type DoctorOpts = {
25
25
  listTests?: (cwd: string) => Promise<VitestListResult>;
26
26
  };
27
27
  type OrcaCapability = {
28
- verdict: "pass" | "fail";
28
+ verdict: "pass" | "fail" | "warn";
29
29
  detail: string;
30
30
  };
31
31
  /**
@@ -12,7 +12,7 @@ import { graphPath, loadGraph, tickmarkrDir, stateDirName } from "../../graph/gr
12
12
  import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, hasWindowsConfig, modelLints, suggestOverlay, ttyVisual } from "../../adapters/model-lints.js";
13
13
  import { loadConfig, overlayPreferShapes } from "../../config/config.js";
14
14
  import { HerdrDriver } from "../../drivers/herdr.js";
15
- import { parseEnvelope } from "../../drivers/orca.js";
15
+ import { ORCA_FIXTURE_VERSION, parseEnvelope, resolveOrcaCliBinary } from "../../drivers/orca.js";
16
16
  import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
17
17
  import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
18
18
  import { LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
@@ -34,7 +34,9 @@ const attentionRow = (text) => ` ${statusRow("warn", text)}`;
34
34
  * stderr (Electron timestamps), so the row is byte-stable across runs.
35
35
  */
36
36
  export async function probeOrcaCapability(cwd, opts = {}) {
37
- const binary = opts.resolveOrcaBinary ? opts.resolveOrcaBinary(cwd) : resolveShellBinary("orca", cwd).resolved;
37
+ const binary = opts.resolveOrcaBinary
38
+ ? opts.resolveOrcaBinary(cwd)
39
+ : resolveOrcaCliBinary(cwd, { resolve: (bin, dir) => resolveShellBinary(bin, dir) });
38
40
  if (!binary)
39
41
  return { verdict: "fail", detail: "CLI not installed" };
40
42
  let response;
@@ -60,8 +62,16 @@ export async function probeOrcaCapability(cwd, opts = {}) {
60
62
  const reachable = typeof runtime === "object" && runtime !== null && !Array.isArray(runtime)
61
63
  ? runtime.reachable
62
64
  : undefined;
63
- if (reachable === true)
64
- return { verdict: "pass", detail: `runtime reachable (${envelope.runtimeId})` };
65
+ if (reachable === true) {
66
+ const appVersion = typeof runtime === "object" && runtime !== null && !Array.isArray(runtime)
67
+ ? runtime.appVersion
68
+ : undefined;
69
+ if (typeof appVersion !== "string" || !appVersion) {
70
+ return { verdict: "warn", detail: `runtime reachable (${envelope.runtimeId}, appVersion absent; fixture pin ${ORCA_FIXTURE_VERSION})` };
71
+ }
72
+ const detail = `runtime reachable (${envelope.runtimeId}, appVersion ${appVersion}; fixture pin ${ORCA_FIXTURE_VERSION})`;
73
+ return { verdict: appVersion === ORCA_FIXTURE_VERSION ? "pass" : "warn", detail };
74
+ }
65
75
  if (reachable === false)
66
76
  return { verdict: "fail", detail: "CLI installed but runtime unreachable" };
67
77
  return { verdict: "fail", detail: "CLI installed but runtime probe failed — status carries no reachability proof" };
@@ -17,6 +17,7 @@ import { loadRoutingProfile } from "../../run/journal.js";
17
17
  import { harnessLine, resolveHarness } from "../harness.js";
18
18
  import { shq } from "../../adapters/types.js";
19
19
  import { shGit } from "../../run/git.js";
20
+ import { driverEvidence, pickDriver } from "../../drivers/index.js";
20
21
  // T4 (v1.50): TTY-only brand pass — the title helper frames the routing table, lint/unroutable
21
22
  // markers carry the attention glyph, section labels dim to chrome (the doctor/status system).
22
23
  // Gated on ttyVisual(): the non-TTY surface returns untouched (byte-pinned, machine-consumable).
@@ -99,7 +100,7 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
99
100
  const DOCTOR_STALE_MS = 24 * 60 * 60 * 1000;
100
101
  // v1.51 T2: plan previews any mode without a config edit — same source precedence as run
101
102
  // (flag > spec front-matter > repo > global > default), same preset compiler.
102
- const { values } = parseArgs({ args: argv, options: { mode: { type: "string" } }, allowPositionals: true });
103
+ const { values } = parseArgs({ args: argv, options: { mode: { type: "string" }, driver: { type: "string" } }, allowPositionals: true });
103
104
  if (values.mode !== undefined && !ROUTING_MODES.includes(values.mode)) {
104
105
  throw new Error(`--mode must be one of ${ROUTING_MODES.join(" | ")} (got ${values.mode})`);
105
106
  }
@@ -136,6 +137,7 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
136
137
  inputRefusals.set(finding.taskId, [...(inputRefusals.get(finding.taskId) ?? []), finding.detail]);
137
138
  }
138
139
  const { cfg, mode, source } = resolveRunMode(cwd, { flag: values.mode, spec: g.mode });
140
+ const selectedDriver = pickDriver(cfg, values.driver);
139
141
  // readDoctor cache path: staleness line only fires here (probeAll fallback is fresh by construction).
140
142
  const cached = readDoctor(cwd);
141
143
  const health = cached ?? (await probeAll(adapters));
@@ -157,6 +159,7 @@ export async function plan(argv, cwd = process.cwd(), adapters = allAdapters(),
157
159
  const lines = [
158
160
  `tickmarkr plan — dry run (${channels.length} channels available)`,
159
161
  `mode: ${mode.mode} (${source}) · explore ${cfg.routing.explore?.mode ?? "on"}`,
162
+ `driver: ${driverEvidence(cfg, selectedDriver, values.driver)}`,
160
163
  harnessLine(resolveHarness(harnessFrom)),
161
164
  "",
162
165
  ];
@@ -205,6 +205,8 @@ export async function verify(argv, cwd = process.cwd()) {
205
205
  onGate: (e) => {
206
206
  if (e.phase === "start")
207
207
  console.error(`verify: → ${e.gate} (${e.index}/${e.total})`);
208
+ else if (e.phase === "note")
209
+ console.error(`verify: note ${e.gate} ${e.name} ${JSON.stringify(e.payload)}`);
208
210
  else
209
211
  console.error(`verify: ${e.result.pass ? "✓" : "✗"} ${e.result.gate} — ${e.result.details.split("\n")[0] ?? ""}`);
210
212
  },
@@ -1 +1,2 @@
1
- export declare function version(_argv?: string[]): Promise<string>;
1
+ export declare function distFingerprint(dir?: string): string;
2
+ export declare function version(argv?: string[], dir?: string): Promise<string>;
@@ -1,8 +1,29 @@
1
- import { readFileSync } from "node:fs";
2
- import { dirname, join } from "node:path";
1
+ import { createHash } from "node:crypto";
2
+ import { existsSync, readFileSync, readdirSync } from "node:fs";
3
+ import { dirname, join, relative } from "node:path";
3
4
  import { fileURLToPath } from "node:url";
4
5
  const pkgPath = join(dirname(fileURLToPath(import.meta.url)), "../../../package.json");
5
- export async function version(_argv = []) {
6
+ const distDir = join(dirname(pkgPath), "dist");
7
+ export function distFingerprint(dir = distDir) {
8
+ if (!existsSync(dir))
9
+ return "missing";
10
+ const files = [];
11
+ const visit = (path) => {
12
+ for (const entry of readdirSync(path, { withFileTypes: true })) {
13
+ const child = join(path, entry.name);
14
+ if (entry.isDirectory())
15
+ visit(child);
16
+ else if (entry.isFile())
17
+ files.push(child);
18
+ }
19
+ };
20
+ visit(dir);
21
+ const hash = createHash("sha256");
22
+ for (const file of files.sort())
23
+ hash.update(relative(dir, file)).update("\0").update(readFileSync(file)).update("\0");
24
+ return hash.digest("hex");
25
+ }
26
+ export async function version(argv = [], dir = distDir) {
6
27
  const { version: v } = JSON.parse(readFileSync(pkgPath, "utf8"));
7
- return v;
28
+ return argv.includes("--dist") ? `${v} dist:${distFingerprint(dir)}` : v;
8
29
  }
@@ -724,6 +724,7 @@ acceptance is required on every task (a nested list of observable outcomes).
724
724
  - command: <shell> (oracle: command — exit code)
725
725
  - test: <name> (oracle: test — named test)
726
726
  - judge: <rubric> (oracle: judge — LLM-judged, free text)
727
+ A judge criterion carries ONE claim; a semicolon-joined criterion warns — split its clauses.
727
728
  - <plain text> (compat: compiles as judge oracle, warns)
728
729
 
729
730
  HARD BOUNDS — these FAIL the compile, they do not warn:
@@ -839,6 +840,7 @@ acceptance is required on every task (a nested list of observable outcomes).
839
840
  ORDERING AND OWNERSHIP:
840
841
  - Every path has exactly ONE owning task. Two tasks writing one file must be ORDERED by deps, or the
841
842
  loser's work is silently dropped when the integration tip advances.
843
+ - A task changing what the daemon DOES must own every surface that TELLS the operator what the daemon does.
842
844
  - A file one task CREATES cannot be "context:" for another — only deps: carries it, and that extends to
843
845
  the task that PRODUCES a value, not just the file's existence.
844
846
  - Deleting or renaming a symbol is a cross-task contract. Sweep for consumers by symbol AND by what the
@@ -4,4 +4,5 @@ export declare const DRIVER_CHOICES: readonly ["auto", "herdr", "subprocess", "o
4
4
  export type DriverChoice = (typeof DRIVER_CHOICES)[number];
5
5
  /** Validate argv at the CLI boundary rather than casting an arbitrary string into a driver choice. */
6
6
  export declare function parseDriverOverride(override?: string): DriverChoice | undefined;
7
+ export declare function driverEvidence(cfg: TickmarkrConfig, driver: ExecutorDriver, override?: string): string;
7
8
  export declare function pickDriver(cfg: TickmarkrConfig, override?: string): ExecutorDriver;
@@ -2,6 +2,7 @@ import { HerdrDriver } from "./herdr.js";
2
2
  import { OrcaDriver } from "./orca.js";
3
3
  import { SubprocessDriver } from "./subprocess.js";
4
4
  export const DRIVER_CHOICES = ["auto", "herdr", "subprocess", "orca"];
5
+ const overrideByDriver = new WeakMap();
5
6
  /** Validate argv at the CLI boundary rather than casting an arbitrary string into a driver choice. */
6
7
  export function parseDriverOverride(override) {
7
8
  if (override === undefined)
@@ -11,18 +12,32 @@ export function parseDriverOverride(override) {
11
12
  return choice;
12
13
  throw new Error(`usage: --driver must be one of ${DRIVER_CHOICES.join(" | ")} (got ${override})`);
13
14
  }
15
+ export function driverEvidence(cfg, driver, override) {
16
+ const selectedOverride = override ?? overrideByDriver.get(driver);
17
+ const want = parseDriverOverride(selectedOverride) ?? cfg.driver;
18
+ if (selectedOverride !== undefined)
19
+ return `${driver.id} (--driver)`;
20
+ if (want !== "auto")
21
+ return `${driver.id} (config)`;
22
+ const herdrAvailable = process.env.HERDR_ENV === "1";
23
+ if (driver.id === (herdrAvailable ? "herdr" : "subprocess")) {
24
+ return `auto → ${driver.id} (${herdrAvailable ? "HERDR_ENV=1" : "HERDR_ENV unset"})`;
25
+ }
26
+ return `auto → ${driver.id} (runtime)`;
27
+ }
14
28
  export function pickDriver(cfg, override) {
15
- const want = parseDriverOverride(override) ?? cfg.driver;
29
+ const selectedOverride = parseDriverOverride(override);
30
+ const want = selectedOverride ?? cfg.driver;
16
31
  // VIS-09 item 2: plumb the per-tab cap into the HerdrDriver — the driver takes it as a constructor
17
32
  // param and never imports config (cfg is the only seam). Guaranteed present: DEFAULT_CONFIG seeds
18
33
  // workersPerTab:3 and deepMerge overlays on top, so a missing overlay key still resolves.
19
- if (want === "herdr")
20
- return new HerdrDriver("herdr", cfg.visibility.workersPerTab);
21
- if (want === "subprocess")
22
- return new SubprocessDriver();
23
- // Orca is an operator-selected execution surface. Its runtime failure stays on Orca; selection
24
- // must never substitute a hidden subprocess worker after this explicit choice.
25
- if (want === "orca")
26
- return new OrcaDriver();
27
- return HerdrDriver.available() ? new HerdrDriver("herdr", cfg.visibility.workersPerTab) : new SubprocessDriver();
34
+ const driver = want === "herdr" ? new HerdrDriver("herdr", cfg.visibility.workersPerTab)
35
+ : want === "subprocess" ? new SubprocessDriver()
36
+ // Orca is an operator-selected execution surface. Its runtime failure stays on Orca; selection
37
+ // must never substitute a hidden subprocess worker after this explicit choice.
38
+ : want === "orca" ? new OrcaDriver()
39
+ : HerdrDriver.available() ? new HerdrDriver("herdr", cfg.visibility.workersPerTab) : new SubprocessDriver();
40
+ if (selectedOverride !== undefined)
41
+ overrideByDriver.set(driver, selectedOverride);
42
+ return driver;
28
43
  }
@@ -3,7 +3,12 @@ import { type ExecutorDriver, type NotifyOpts, type Slot, type SlotOpts } from "
3
3
  /** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
4
4
  export declare const ORCA_RESPONSE_FAMILIES: readonly ["status", "create", "list", "read", "send", "wait", "show", "close"];
5
5
  export type OrcaFamily = (typeof ORCA_RESPONSE_FAMILIES)[number];
6
+ export declare const ORCA_FIXTURE_VERSION = "1.4.195";
7
+ export declare const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
6
8
  export declare const STALE_HANDLE_CODE = "terminal_handle_stale";
9
+ export declare const TERMINAL_GONE_CODE = "terminal_gone";
10
+ export declare const STALE_HANDLE_CODES: Set<string>;
11
+ export declare const WAIT_TIMEOUT_CODE = "timeout";
7
12
  export declare const NOT_WRITABLE_CODE = "terminal_not_writable";
8
13
  /** The ONLY terminal status that licenses reading a terminal's bytes or its agent state. */
9
14
  export declare const RUNNING_STATUS = "running";
@@ -38,6 +43,18 @@ export interface OrcaEnvelope {
38
43
  runtimeId: string;
39
44
  raw: string;
40
45
  }
46
+ export interface OrcaBinaryResolverOpts {
47
+ env?: NodeJS.ProcessEnv | Record<string, string | undefined>;
48
+ platform?: NodeJS.Platform;
49
+ resolve?: (bin: string, cwd: string) => {
50
+ resolved?: string;
51
+ };
52
+ }
53
+ /**
54
+ * One Orca CLI name decision for both the driver and doctor. Linux desktop systems can have
55
+ * GNOME's screen-reader `orca` on PATH, so outside an Orca terminal the app CLI is `orca-ide`.
56
+ */
57
+ export declare function resolveOrcaCliBinary(cwd?: string, opts?: OrcaBinaryResolverOpts): string | undefined;
41
58
  /**
42
59
  * The one JSON seam. Fails CLOSED on every degenerate response — empty, unparseable (a truncated
43
60
  * body lands here), non-object, no boolean `ok`, `ok:false`, `ok:true` with no result object, or a
@@ -76,6 +93,8 @@ export declare function mapAgentState(term: Record<string, unknown>, tuiIdle: bo
76
93
  export declare function joinWrapped(raw: string): string;
77
94
  export interface OrcaDriverOpts {
78
95
  bin?: string;
96
+ env?: NodeJS.ProcessEnv | Record<string, string | undefined>;
97
+ platform?: NodeJS.Platform;
79
98
  exec?: OrcaExec;
80
99
  time?: OrcaTimeSource;
81
100
  pageLines?: number;
@@ -8,7 +8,7 @@ import { formatOwnedName, panesToClose } from "./types.js";
8
8
  // worktrees, routing, gates, journal and merges; orca supplies visible terminals only. Everything
9
9
  // here is bound by the 1.4.186 conformance spike
10
10
  // (.planning/assessments/2026-08-21-orca-driver-conformance.md, CONFORMANCE-END) and the
11
- // recorded refusal transport (process rc 1 + ok:false on stdout):
11
+ // 1.4.195 drift capture (.planning/assessments/2026-09-02-orca-1.4.195-capture/):
12
12
  //
13
13
  // - reads arrive as `result.terminal.tail` line arrays with line-indexed cursors and a `status`
14
14
  // field on the same object (C2); a CLOSED terminal answers ok:true with its own dead record
@@ -32,7 +32,12 @@ import { formatOwnedName, panesToClose } from "./types.js";
32
32
  // an older run's leftover sits in a checkout this run never knew (T2).
33
33
  /** The response families the ONE shared envelope parser serves. There is no second JSON seam. */
34
34
  export const ORCA_RESPONSE_FAMILIES = ["status", "create", "list", "read", "send", "wait", "show", "close"];
35
+ export const ORCA_FIXTURE_VERSION = "1.4.195";
36
+ export const ORCA_CLI_COMMAND_ENV = "ORCA_CLI_COMMAND";
35
37
  export const STALE_HANDLE_CODE = "terminal_handle_stale";
38
+ export const TERMINAL_GONE_CODE = "terminal_gone";
39
+ export const STALE_HANDLE_CODES = new Set([STALE_HANDLE_CODE, TERMINAL_GONE_CODE]);
40
+ export const WAIT_TIMEOUT_CODE = "timeout";
36
41
  export const NOT_WRITABLE_CODE = "terminal_not_writable";
37
42
  /** The ONLY terminal status that licenses reading a terminal's bytes or its agent state. */
38
43
  export const RUNNING_STATUS = "running";
@@ -79,6 +84,19 @@ export class OrcaUnavailableError extends OrcaError {
79
84
  function str(v) {
80
85
  return typeof v === "string" && v ? v : undefined;
81
86
  }
87
+ /**
88
+ * One Orca CLI name decision for both the driver and doctor. Linux desktop systems can have
89
+ * GNOME's screen-reader `orca` on PATH, so outside an Orca terminal the app CLI is `orca-ide`.
90
+ */
91
+ export function resolveOrcaCliBinary(cwd = process.cwd(), opts = {}) {
92
+ const env = opts.env ?? process.env;
93
+ const explicit = env[ORCA_CLI_COMMAND_ENV]?.trim();
94
+ if (explicit)
95
+ return explicit;
96
+ const platform = opts.platform ?? process.platform;
97
+ const selected = platform === "linux" && env.TERM_PROGRAM !== "Orca" ? "orca-ide" : "orca";
98
+ return opts.resolve ? opts.resolve(selected, cwd).resolved : selected;
99
+ }
82
100
  /**
83
101
  * The one JSON seam. Fails CLOSED on every degenerate response — empty, unparseable (a truncated
84
102
  * body lands here), non-object, no boolean `ok`, `ok:false`, `ok:true` with no result object, or a
@@ -235,7 +253,7 @@ export class OrcaDriver {
235
253
  pollMs;
236
254
  probeStalenessMs;
237
255
  constructor(opts = {}) {
238
- this.bin = opts.bin ?? "orca";
256
+ this.bin = opts.bin ?? resolveOrcaCliBinary(process.cwd(), { env: opts.env, platform: opts.platform }) ?? "orca";
239
257
  // Config values flow into a shell here: every argv element is quoted, always.
240
258
  this.exec = opts.exec ?? ((args, cwd, timeoutMs) => {
241
259
  return sh([this.bin, ...args].map(shq).join(" "), cwd, timeoutMs);
@@ -257,14 +275,13 @@ export class OrcaDriver {
257
275
  // diagnostic is never discarded before the one shared parser reports it.
258
276
  const raw = [r.stdout, r.stderr ? `STDERR: ${r.stderr}` : ""].filter(Boolean).join("\n");
259
277
  if (r.code !== 0) {
260
- // Recorded 1.4.186 refusal transport: the process exits rc 1 with the STRUCTURED ok:false
278
+ // Recorded refusal transport: the process exits rc 1 with the STRUCTURED ok:false
261
279
  // body on stdout. Parse it so the refusal CODE survives (terminal_handle_stale,
262
280
  // terminal_not_writable) — everything downstream that recovers on a code depends on this
263
- // branch. The one documented exception is an elapsed `terminal wait`: it exits 1 but carries
264
- // an ok:true, `wait.satisfied:false` receipt. It remains the shared parser's success path;
265
- // waitCondition() validates its identity, handle, condition and running status before it
266
- // becomes the normal `false` result. Any other ok:true body on a nonzero exit stays a
267
- // transport failure (a shell/runtime crash can leave stale stdout behind).
281
+ // branch. Elapsed `terminal wait` has two recorded transports: 1.4.186's ok:true,
282
+ // `wait.satisfied:false` receipt and 1.4.195's ok:false/code:timeout refusal. Each remains
283
+ // scoped to wait only; any other ok:true body on a nonzero exit stays a transport failure (a
284
+ // shell/runtime crash can leave stale stdout behind).
268
285
  try {
269
286
  const env = parseEnvelope(family, r.stdout, raw);
270
287
  const wait = env.result.wait;
@@ -279,6 +296,21 @@ export class OrcaDriver {
279
296
  }
280
297
  }
281
298
  catch (e) {
299
+ if (family === "wait"
300
+ && r.code === 1
301
+ && !r.timedOut
302
+ && e instanceof OrcaError
303
+ && e.code === WAIT_TIMEOUT_CODE
304
+ && e.runtimeId
305
+ && e.runtimeId !== "none") {
306
+ const handle = args[args.indexOf("--terminal") + 1];
307
+ const condition = args[args.indexOf("--for") + 1];
308
+ return {
309
+ result: { wait: { handle, condition, satisfied: false, status: RUNNING_STATUS } },
310
+ runtimeId: e.runtimeId,
311
+ raw,
312
+ };
313
+ }
282
314
  // The refusal code lives in stdout alone, but the raw bytes propagated to the caller must
283
315
  // still be the COMBINED stream — stderr can carry the diagnostic that explains the refusal.
284
316
  if (e instanceof OrcaError && !(e instanceof OrcaUnavailableError) && e.code !== undefined) {
@@ -392,6 +424,10 @@ export class OrcaDriver {
392
424
  st.handle = handle;
393
425
  // The handle is bound to the runtime identity that ANSWERED its create.
394
426
  st.runtimeId = env.runtimeId;
427
+ const surface = str(term.surface);
428
+ if (surface !== undefined && surface !== "visible") {
429
+ await this.notify(`tickmarkr orca terminal created on ${surface} surface`, { tier: "attention" });
430
+ }
395
431
  }
396
432
  // ---- handle identity and restart recovery ----------------------------------------------------
397
433
  /**
@@ -450,7 +486,7 @@ export class OrcaDriver {
450
486
  await relist(e.runtimeId, e.raw);
451
487
  continue;
452
488
  }
453
- if (e.code === STALE_HANDLE_CODE) {
489
+ if (STALE_HANDLE_CODES.has(e.code)) {
454
490
  await relist(e.runtimeId, e.raw);
455
491
  continue;
456
492
  }
@@ -45,6 +45,8 @@ export interface BaselineCommand {
45
45
  * verdict: no exit code, no fingerprints, nothing forgivable.
46
46
  */
47
47
  infra?: true;
48
+ /** Runner-level exhaustion lines that invalidated a completed capture. */
49
+ invalidatingLines?: string[];
48
50
  }
49
51
  export interface BaselineFileDuration {
50
52
  file: string;
@@ -85,7 +85,7 @@ const TURBO_FAIL_RE = /^\s*(?:[\w@./-]+:\s*)*ELIFECYCLE\s+Command failed\b|^\s*F
85
85
  // at least TWO colon-joined segments (`<pkg>:<task>:`, tasks may nest — `pkg:test:unit:`), every
86
86
  // segment after the first starting with a letter — so `Error: boom` (one segment), `src/x.ts:12:`
87
87
  // (digit segment) and `12:34 error` (eslint stylish) are never stripped.
88
- const TURBO_PREFIX_RE = /^\s*[\w@./-]+(?::[A-Za-z_][\w.-]*)+:\s+/;
88
+ const TURBO_PREFIX_RE = /^\s*[\w@./-]+(?::[A-Za-z_][\w.-]*)+:(?:\s+|$)/;
89
89
  const stripTurboPrefix = (l) => {
90
90
  const m = TURBO_PREFIX_RE.exec(l);
91
91
  return m ? l.slice(m[0].length) : undefined;
@@ -161,11 +161,36 @@ export function classifyFailureOutput(output) {
161
161
  * no failure in that incomplete environment can safely become pre-existing forgiveness. Keep this
162
162
  * separate from `classifyFailureOutput` so changing capture policy cannot move gate verdicts.
163
163
  */
164
- const captureHasInvalidatingInfra = (output) => output
165
- .split("\n")
166
- .map((l) => l.replace(ANSI_RE, ""))
167
- .filter((l) => !PASS_LINE_RE.test(l))
168
- .some((l) => CAPTURE_EXHAUSTION_RE.test(l));
164
+ const VITEST_ECHO_BLOCK_RE = /^\s*std(?:out|err)\s+\|\s+\S+\.(?:test|spec)\.[cm]?[jt]sx?\s+>\s+\S/;
165
+ /** Keep runner output while omitting Vitest's echoed test-owned stdout/stderr diagnostic blocks. */
166
+ const withoutVitestEchoBlocks = (output) => {
167
+ const outside = [];
168
+ let inVitestEchoBlock = false;
169
+ for (const line of output.split("\n")) {
170
+ const clean = line.replace(ANSI_RE, "");
171
+ const runnerLine = stripTurboPrefix(clean) ?? clean;
172
+ if (inVitestEchoBlock) {
173
+ if (runnerLine.trim() === "")
174
+ inVitestEchoBlock = false;
175
+ continue;
176
+ }
177
+ if (VITEST_ECHO_BLOCK_RE.test(runnerLine)) {
178
+ inVitestEchoBlock = true;
179
+ continue;
180
+ }
181
+ outside.push(line);
182
+ }
183
+ return outside;
184
+ };
185
+ const captureInvalidatingLines = (output) => {
186
+ const invalidating = [];
187
+ for (const line of withoutVitestEchoBlocks(output)) {
188
+ const clean = line.replace(ANSI_RE, "");
189
+ if (!PASS_LINE_RE.test(clean) && CAPTURE_EXHAUSTION_RE.test(clean))
190
+ invalidating.push(line);
191
+ }
192
+ return invalidating;
193
+ };
169
194
  const normalizeLine = (l) => l.replace(/\d+/g, "#").replace(/\s+/g, " ").trim();
170
195
  // Vitest's default reporter names file durations as
171
196
  // `✓ |project| tests/example.test.ts (12 tests) 1.23s` (❯ for a red file). The runner may be
@@ -203,8 +228,7 @@ export const UNRECOGNIZED_FAILURE = "<unrecognized failure output>";
203
228
  // renders "zone +3 · run attempt 2 · 1 failed" (what T9 is chartered to draw) contributes nothing, with
204
229
  // or without box glyphs, so it cannot manufacture a fresh-fingerprint rejection on any future attempt.
205
230
  export function fingerprint(output) {
206
- const lines = output
207
- .split("\n")
231
+ const lines = withoutVitestEchoBlocks(output)
208
232
  .map((l) => l.replace(ANSI_RE, ""))
209
233
  .filter((l) => !PASS_LINE_RE.test(l));
210
234
  // GATE-FIX-4 DEFECT 4: every line is read twice — as printed, and with a turbo `<pkg>:<task>:`
@@ -403,7 +427,7 @@ export function ceilingKillResult(gate, r, ceilingMs) {
403
427
  * shell asks it to finish faster than the thing it is measuring.
404
428
  */
405
429
  export const CAPTURE_CEILING_MS = 1_800_000;
406
- const invalidCaptureEntry = (durationMs) => ({
430
+ const invalidCaptureEntry = (durationMs, invalidatingLines = []) => ({
407
431
  infra: true,
408
432
  fingerprints: [],
409
433
  durationMs,
@@ -411,6 +435,7 @@ const invalidCaptureEntry = (durationMs) => ({
411
435
  impliedParallelism: null,
412
436
  longestFile: null,
413
437
  ceilingMs: effectiveCeilingMs({ durationMs }),
438
+ ...(invalidatingLines.length ? { invalidatingLines } : {}),
414
439
  });
415
440
  export async function captureBaseline(cwd, commands) {
416
441
  const base = { commands: {} };
@@ -441,17 +466,20 @@ export async function captureBaseline(cwd, commands) {
441
466
  base.commands[name] = invalidCaptureEntry(durationMs);
442
467
  continue;
443
468
  }
444
- const raw = (r.stdout + "\n" + r.stderr).split(cwd).join("");
469
+ const combinedOutput = r.stdout + "\n" + r.stderr;
470
+ const raw = combinedOutput.split(cwd).join("");
445
471
  // Run 2137: the child exited and printed ordinary FAIL/AssertionError lines, so the gate-side
446
472
  // discriminator correctly called the mixed output a regression. But the same output also said
447
473
  // `spawn EAGAIN`: the machine had run out of processes while the pristine-tree measurement was
448
474
  // being taken. A capture cannot know which red lines predated that shortage and which it caused,
449
475
  // so none may become a fingerprint every later task gets to forgive.
450
- if (captureHasInvalidatingInfra(raw)) {
476
+ const invalidatingLines = captureInvalidatingLines(combinedOutput);
477
+ if (invalidatingLines.length) {
451
478
  console.error(`tickmarkr: baseline capture for "${name}" completed with process/resource-exhaustion evidence — `
452
479
  + `it recorded NO exit-code verdict and NO fingerprints, so nothing is forgiven for this command; `
453
- + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion.`);
454
- base.commands[name] = invalidCaptureEntry(durationMs);
480
+ + `the measurement cannot distinguish a pre-existing failure from one caused by exhaustion. `
481
+ + `First invalidating line: ${invalidatingLines[0]}`);
482
+ base.commands[name] = invalidCaptureEntry(durationMs, invalidatingLines);
455
483
  continue;
456
484
  }
457
485
  base.commands[name] = {
@@ -530,7 +558,14 @@ export async function compareToBaseline(cwd, commands, baseline, enabled) {
530
558
  if (!cmd) {
531
559
  // nothing detected for this gate in the target repo — journal an explicit skip instead of
532
560
  // vanishing silently (a lint gate with no lint script rendered as forever-open in status)
533
- results.push({ gate: name, pass: true, details: `no ${name} command detected — skipped`, meta: { skipped: true } });
561
+ const reason = `no ${name} command detected`;
562
+ const outcome = { kind: "skipped", reason };
563
+ results.push({
564
+ gate: name,
565
+ pass: true,
566
+ details: `${reason} — skipped`,
567
+ meta: { skipped: true, outcome },
568
+ });
534
569
  continue;
535
570
  }
536
571
  const entry = baseline.commands[name];
@@ -1,4 +1,4 @@
1
- import type { WorkerAdapter } from "../adapters/types.js";
1
+ import { type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type ExecutorDriver, type Slot } from "../drivers/types.js";
3
3
  export declare const GATE_PANE_SEP = " \u00B7 ";
4
4
  export declare const COMPLETION_FAKING_CHECKLIST = "## Completion-faking checklist\nHunt for these concrete completion-faking shortcuts before ruling on any criterion:\n- hardcoded-result: output or fixture hardcoded to satisfy the stated criterion instead of real logic\n- test-weakening: tests skipped, deleted, or assertions loosened until failing behavior looks green\n- vacuous-assertion: a test that cannot fail (asserts a constant, asserts its own setup, no assertion)\n- fixture-overfit: implementation narrowed to the exact test inputs rather than the described behavior\n- echo-not-implement: criterion text echoed in names, comments, or strings without the behavior itself\n- stub-left-behind: TODO, throw, or no-op stub where the real implementation should be\n- error-swallowing: catch or fallback that hides failures instead of handling them\n- self-mocking: the code under test mocked or faked so the test exercises the mock\n- check-bypass: lint, type, or CI checks disabled, relaxed, or excluded to get green\n- rename-as-work: code moved or renamed and presented as the requested change\n- scope-padding: unrelated edits padding the diff while the criterion's behavior is untouched\nWhen a criterion fails, the verdict MUST name which shortcut above it matches, or state that none does.";
package/dist/gates/llm.js CHANGED
@@ -3,6 +3,7 @@ import { randomBytes } from "node:crypto";
3
3
  import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
4
4
  import { tmpdir } from "node:os";
5
5
  import { join } from "node:path";
6
+ import { matchesTrustDialog } from "../adapters/types.js";
6
7
  import { formatOwnedName, parseOwnedName } from "../drivers/types.js";
7
8
  import { bannerShell, paneDispatchCommand } from "../brand.js";
8
9
  import { sh } from "../run/git.js";
@@ -161,6 +162,16 @@ export async function runViaDriver(adapter, model, prompt, cwd, via, timeoutMs =
161
162
  slot = await via.driver.slot(cwd, rolePaneNameFromPrompt(prompt, via.name), via.label ? { label: via.label } : undefined);
162
163
  via.onSlot?.(slot);
163
164
  await via.driver.run(slot, paneDispatchCommand(scriptPath));
165
+ if (via.driver.sendKey) {
166
+ try {
167
+ if (matchesTrustDialog(await via.driver.read(slot, 400), adapter.trustDialog)) {
168
+ await via.driver.sendKey(slot, adapter.trustDialog.key);
169
+ }
170
+ }
171
+ catch {
172
+ /* a failed pre-wait read must not replace the verdict wait */
173
+ }
174
+ }
164
175
  // nonce-suffixed exit only: a displayed bare "TICKMARKR_EXIT:" or another call's marker must not
165
176
  // false-complete — same guard the worker path uses (daemon.ts:330-331).
166
177
  const exitPattern = `TICKMARKR_EXIT_${nonce}:\\d`;
@@ -369,37 +380,41 @@ export function extractVerdictJson(raw, nonce) {
369
380
  /* fall through */
370
381
  }
371
382
  }
372
- let pos = raw.length - 1;
373
- while (pos >= 0) {
374
- const end = raw.lastIndexOf("}", pos);
375
- if (end === -1)
376
- return null;
377
- let depth = 1;
378
- let stepped = false;
379
- for (let i = end - 1; i >= 0; i--) {
380
- if (raw[i] === "}")
383
+ for (let start = raw.lastIndexOf("{"); start >= 0;) {
384
+ const nextStart = start === 0 ? -1 : raw.lastIndexOf("{", start - 1);
385
+ let depth = 0;
386
+ let quoted = false;
387
+ let escaped = false;
388
+ for (let i = start; i < raw.length; i++) {
389
+ const char = raw[i];
390
+ if (quoted) {
391
+ if (escaped)
392
+ escaped = false;
393
+ else if (char === "\\")
394
+ escaped = true;
395
+ else if (char === '"')
396
+ quoted = false;
397
+ continue;
398
+ }
399
+ if (char === '"')
400
+ quoted = true;
401
+ else if (char === "{")
381
402
  depth++;
382
- else if (raw[i] === "{") {
383
- depth--;
384
- if (depth === 0) {
385
- stepped = true;
386
- try {
387
- const v = JSON.parse(raw.slice(i, end + 1));
388
- if (v && typeof v === "object" && v.nonce === nonce) {
389
- const { nonce: _n, ...rest } = v;
390
- return rest;
391
- }
392
- }
393
- catch {
394
- /* keep scanning */
403
+ else if (char === "}" && --depth === 0) {
404
+ try {
405
+ const v = JSON.parse(raw.slice(start, i + 1));
406
+ if (v && typeof v === "object" && v.nonce === nonce) {
407
+ const { nonce: _n, ...rest } = v;
408
+ return rest;
395
409
  }
396
- pos = i - 1;
397
- break;
398
410
  }
411
+ catch {
412
+ /* keep scanning */
413
+ }
414
+ break;
399
415
  }
400
416
  }
401
- if (!stepped)
402
- return null;
417
+ start = nextStart;
403
418
  }
404
419
  return null;
405
420
  }
@@ -348,6 +348,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
348
348
  // OBS-196: name the cause and persist the raw bytes — a ruled-on "unparseable" without its
349
349
  // evidence cannot be audited, and a cutoff must never be indistinguishable from a parse defect.
350
350
  const cause = classifyVerdictCause(raw, nonce, "approve");
351
+ const bytes = Buffer.byteLength(raw, "utf8");
351
352
  let saved;
352
353
  if (artifactDir) {
353
354
  try {
@@ -372,6 +373,7 @@ The top-level comments array is optional. Use it only for actionable line-anchor
372
373
  reviewer: channelKey(reviewer),
373
374
  unparseable: true,
374
375
  cause,
376
+ ...(cause === "empty-output" ? { bytes } : {}),
375
377
  ...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
376
378
  },
377
379
  };
@@ -31,6 +31,12 @@ export type GateEvent = {
31
31
  phase: "end";
32
32
  gate: GateName;
33
33
  result: GateResult;
34
+ } | {
35
+ phase: "note";
36
+ gate: GateName;
37
+ name: string;
38
+ payload: Record<string, unknown>;
39
+ result: GateResult;
34
40
  };
35
41
  export interface GateContext {
36
42
  worktree: string;
@@ -583,6 +583,14 @@ export async function runGates(task, ctx) {
583
583
  // eligible seat keeps the ORIGINAL result so the recorded cause stays truthful (OBS-196).
584
584
  if (rv.meta?.unparseable === true && typeof rv.meta.reviewer === "string") {
585
585
  const flaked = rv.meta.reviewer;
586
+ const emptyOutput = rv.meta.cause === "empty-output";
587
+ if (emptyOutput) {
588
+ await ctx.onGate?.({
589
+ phase: "note", gate: "review", name: "reviewer-empty-output",
590
+ payload: { reviewer: flaked, bytes: typeof rv.meta.bytes === "number" ? rv.meta.bytes : 0 },
591
+ result: { ...rv, meta: { ...rv.meta, skipped: true } },
592
+ });
593
+ }
586
594
  const retryVia = ctx.via
587
595
  ? { ...ctx.via, nameFor: (role, adapter) => ctx.via.nameFor(role, adapter) + "-r1" }
588
596
  : undefined;
@@ -593,7 +601,7 @@ export async function runGates(task, ctx) {
593
601
  ...second,
594
602
  // `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
595
603
  // re-route visible in the result text a reader actually opens, including on a red retry.
596
- details: `review re-route: ${flaked} produced no parseable verdict; replaced by ${retried}\n${second.details}`,
604
+ details: `review re-route: ${flaked} produced ${emptyOutput ? "EMPTY output" : "no parseable verdict"}; replaced by ${retried}\n${second.details}`,
597
605
  meta: { ...second.meta, reviewRetry: { flaked, retried } },
598
606
  };
599
607
  }
@@ -9,6 +9,9 @@ export interface ConsultVerdict {
9
9
  reason?: string;
10
10
  guidance?: string;
11
11
  excludeAdapter?: string;
12
+ adapter?: string;
13
+ model?: string;
14
+ vendor?: string;
12
15
  }
13
16
  export declare function renderRetryGuidance(v: ConsultVerdict): string;
14
17
  export declare function augmentRetryBrief(feedback: string, opts: {
@@ -36,5 +39,9 @@ export declare function consult(d: Dossier, cfg: TickmarkrConfig, adapters: Work
36
39
  runId?: string;
37
40
  channels?: Array<{
38
41
  adapter: string;
42
+ model?: string;
43
+ vendor?: string;
44
+ channel?: "sub" | "api";
45
+ tier?: string;
39
46
  }>;
40
47
  }): Promise<ConsultVerdict>;
@@ -205,6 +205,17 @@ opts = {}) {
205
205
  // the same rule as every prefer entry — and disallowedBy carries the full deny grammar (adapter,
206
206
  // model, or adapter:model), so a model-scoped deny cannot slip past an adapter-id-only read.
207
207
  const allowedSeats = seats.filter((s) => disallowedBy(s, cfg.routing, "consult") === null);
208
+ const seatIdentity = (seat) => {
209
+ let vendor = opts.channels?.find((candidate) => candidate.adapter === seat.adapter && candidate.model === seat.model)?.vendor;
210
+ if (!vendor) {
211
+ try {
212
+ const adapter = getAdapter(seat.adapter, adapters);
213
+ vendor = adapter.channels(cfg).find((channel) => channel.model === seat.model)?.vendor ?? adapter.vendor;
214
+ }
215
+ catch { /* an unknown adapter still gets explicit unknown provenance */ }
216
+ }
217
+ return { adapter: seat.adapter, model: seat.model, vendor: vendor ?? "unknown" };
218
+ };
208
219
  if (!allowedSeats.length) {
209
220
  const d = disallowedBy(seats[seats.length - 1], cfg.routing, "consult");
210
221
  return {
@@ -216,11 +227,14 @@ opts = {}) {
216
227
  try {
217
228
  const parsed = await invokeSeat(seat.adapter, seat.model, i);
218
229
  if (parsed.verdict)
219
- return parsed.verdict;
230
+ return { ...parsed.verdict, ...seatIdentity(seat) };
220
231
  }
221
232
  catch {
222
233
  // failed seat (unknown adapter, dead driver/pane, shell error) — fall to the next entry
223
234
  }
224
235
  }
225
- return { action: "human", notes: "consult verdict unparseable — failing safe to human" };
236
+ return {
237
+ action: "human",
238
+ notes: "consult verdict unparseable — failing safe to human",
239
+ };
226
240
  }
@@ -1,5 +1,6 @@
1
1
  import { type WorkerAdapter } from "../adapters/types.js";
2
2
  import { type ModeResolution, type RoutingMode, type TickmarkrConfig } from "../config/config.js";
3
+ import { type DriverChoice } from "../drivers/index.js";
3
4
  import { type ExecutorDriver } from "../drivers/types.js";
4
5
  import { type Baseline } from "../gates/baseline.js";
5
6
  import type { GateResult } from "../gates/types.js";
@@ -13,6 +14,7 @@ export interface RunOptions {
13
14
  retryFailed?: boolean;
14
15
  concurrency?: number;
15
16
  driver?: ExecutorDriver;
17
+ driverOverride?: DriverChoice;
16
18
  adapters?: WorkerAdapter[];
17
19
  globalDir?: string;
18
20
  mode?: RoutingMode;
@@ -12,6 +12,7 @@ import { bannerShell, paneDispatchCommand } from "../brand.js";
12
12
  import { collateralHits } from "../compile/collateral.js";
13
13
  import { DEFAULT_DIFF_CAP, globalConfigDir, loadConfigWithMode, readOverlayFile, repoOverlayPath, } from "../config/config.js";
14
14
  import { DeliveryReadinessError } from "../drivers/herdr.js";
15
+ import { driverEvidence } from "../drivers/index.js";
15
16
  import { herdrSealShellPrefix, SubprocessDriver } from "../drivers/subprocess.js";
16
17
  import { formatOwnedName } from "../drivers/types.js";
17
18
  import { captureBaseline, detectGateCommands, detectVacuousOracles } from "../gates/baseline.js";
@@ -19,6 +20,7 @@ import { runGates } from "../gates/run-gates.js";
19
20
  import { filesGlob } from "../graph/files-glob.js";
20
21
  import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHash, loadGraph, pendingTasks, readyTasks, saveGraph, setStatus, taskContentDigest, tickmarkrDir } from "../graph/graph.js";
21
22
  import { GATE_NAMES } from "../graph/schema.js";
23
+ import { distFingerprint } from "../cli/commands/version.js";
22
24
  import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
23
25
  import { runEnvironment } from "./environment.js";
24
26
  import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
@@ -1255,11 +1257,25 @@ export async function runDaemon(repoRoot, opts = {}) {
1255
1257
  baseRef = await gitHead(repoRoot);
1256
1258
  baseline = await captureBaseline(repoRoot, commands);
1257
1259
  writeFileSync(join(journal.dir, "baseline.json"), JSON.stringify(baseline, null, 2));
1260
+ writeFileSync(join(journal.dir, "graph.json"), readFileSync(join(tickmarkrDir(repoRoot), "graph.json")));
1258
1261
  // v1.70 T2: environment identity beside the graph/branch identity — running tickmarkr version,
1259
1262
  // loaded-config hash, and the probed CLI version of each adapter holding a channel in the run,
1260
1263
  // gathered through the existing probe/config-load paths (no second mechanism).
1261
1264
  const environment = runEnvironment(cfg, channels, health);
1262
- journal.append("run-start", undefined, { pid: process.pid, baseRef, commands, channels: channels.map(channelKey), branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source, environment, ...(prior ? { supersedes: prior.runId } : {}) }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
1265
+ journal.append("run-start", undefined, {
1266
+ pid: process.pid, baseRef, commands, channels: channels.map(channelKey),
1267
+ channelsByRole: {
1268
+ worker: pools.worker.map(channelKey),
1269
+ judge: pools.judge.map(channelKey),
1270
+ review: pools.review.map(channelKey),
1271
+ consult: pools.consult.map(channelKey),
1272
+ },
1273
+ driver: driver.id,
1274
+ driverEvidence: driverEvidence(cfg, driver, opts.driverOverride),
1275
+ distFingerprint: distFingerprint(),
1276
+ branch, graphDefinitionHash: graphDefinitionHash(graph), mode: rm.mode.mode, modeSource: rm.source,
1277
+ environment, ...(prior ? { supersedes: prior.runId } : {}),
1278
+ }); // graphDefinitionHash: T3 engagement identity (status+resume share it); pid: v1.13 (VIS-11) liveness; mode/modeSource: v1.51 T2; supersedes: v1.53 T5
1263
1279
  runStarted = true;
1264
1280
  // v1.53 T5: mark the prior run AFTER this run's run-start exists, so the prior journal never
1265
1281
  // names a successor that has no journal. Append-only — the prior journal is never rewritten.
@@ -1786,6 +1802,7 @@ export async function runDaemon(repoRoot, opts = {}) {
1786
1802
  const applyVerdict = async (v, attempts, trigger) => {
1787
1803
  journal.append("consult-verdict", t.id, {
1788
1804
  action: v.action, notes: v.notes,
1805
+ adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
1789
1806
  ...(v.reason ? { reason: v.reason } : {}),
1790
1807
  ...(v.guidance ? { guidance: v.guidance } : {}),
1791
1808
  ...(v.excludeAdapter ? { excludeAdapter: v.excludeAdapter } : {}),
@@ -2009,6 +2026,10 @@ export async function runDaemon(repoRoot, opts = {}) {
2009
2026
  journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
2010
2027
  return;
2011
2028
  }
2029
+ if (e.phase === "note") {
2030
+ journal.append(e.name, t.id, e.payload);
2031
+ return;
2032
+ }
2012
2033
  const g = e.result;
2013
2034
  inParallelOrder(g.gate, () => {
2014
2035
  journalGateResult(g);
@@ -3452,6 +3473,10 @@ export async function runDaemon(repoRoot, opts = {}) {
3452
3473
  journal.phaseStart(t.id, phaseForGate(e.gate), { gate: e.gate, index: e.index, total: e.total, ...(e.parentAt === undefined ? {} : { parallel: true }) });
3453
3474
  return;
3454
3475
  }
3476
+ if (e.phase === "note") {
3477
+ journal.append(e.name, t.id, e.payload);
3478
+ return;
3479
+ }
3455
3480
  const g = e.result;
3456
3481
  inParallelOrder(g.gate, () => {
3457
3482
  // GATE-09 (ROADMAP SC-4): journal every judge retry as an attributable event — which gate flaked,
@@ -3654,7 +3679,11 @@ export async function runDaemon(repoRoot, opts = {}) {
3654
3679
  continue;
3655
3680
  return;
3656
3681
  }
3657
- journal.append("consult-verdict", t.id, { action: v.action, notes: v.notes, capAdvisory: true });
3682
+ journal.append("consult-verdict", t.id, {
3683
+ action: v.action, notes: v.notes,
3684
+ adapter: v.adapter ?? "unknown", model: v.model ?? "unknown", vendor: v.vendor ?? "unknown",
3685
+ capAdvisory: true,
3686
+ });
3658
3687
  }
3659
3688
  // v1.85 T3: a narrow battery over fully carried commits earns a REPAIR (decided above) — the
3660
3689
  // next dispatch carries the findings verbatim and the diff content instead of re-onboarding a
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "tickmarkr",
3
- "version": "2.2.0",
3
+ "version": "2.2.1",
4
4
  "description": "Spec in, verified work out.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -29,6 +29,7 @@ Outside a multi-agent terminal environment, run the loop directly.
29
29
  - Do not edit the compiled graph to force an outcome; fix the source spec and compile again.
30
30
  - Gates verify commits, diffs, acceptance criteria, and reviews independently. Never trust a worker's claim that work is complete.
31
31
  - Treat missing or unparseable machine results and verdicts as failures. Do not release, resume, or merge around failed gates.
32
+ - A task changing what the daemon DOES must own every surface that TELLS the operator what the daemon does.
32
33
 
33
34
  ## Act by default
34
35
 
@@ -755,6 +755,10 @@ once** — an overseer ran nine hours at 86% unable to read its own number. Arm
755
755
  Herdr's `agent_status`, which is unreliable for vendors whose screen detection is skipped (kimi) — for
756
756
  those, the spawn-time auto-approve flag is the control, not this watcher.
757
757
 
758
+ For `consult`, `surgeon`, and `reviewer` seats, the CONTEXT watcher is wake-only: arm it with
759
+ `TKR_AUTO_CLEAR=0 .claude/skills/tickmarkr-overseer/scripts/watch-context.sh <consult|surgeon|reviewer>
760
+ <agent|pane> 50 50 <handoff-file>`. Only `orchestrator` and `overseer` seats may use auto-clear.
761
+
758
762
  ```bash
759
763
  .claude/skills/tickmarkr-overseer/scripts/watch-pending-input.sh <agent|pane> [poll-s] [cap-s] [confirm-polls]
760
764
  ```
@@ -953,6 +957,13 @@ orchestrator turn boundary.
953
957
  leave clean state after a version is shipped"***. Every line below is a defect that actually happened
954
958
  on the release that produced this rule.
955
959
 
960
+ - **RUN BOTH RELEASE PROOFS BEFORE STANDING DOWN.** The first is the exported-tree suite, which
961
+ maintainers run from the private repo with its `verify:export` script — that tooling is deliberately
962
+ not part of this package, so the command does not exist in a public checkout; it must install,
963
+ build, and run the suite in the exported tree; then
964
+ `TICKMARKR_E2E=1 npx vitest run tests/e2e/orca-smoke.e2e.test.ts` must exercise the installed-Orca
965
+ smoke. A skipped smoke is not proof. Record both results before declaring the release complete.
966
+
956
967
  - **REWRITE THE MEMORY INDEX FIRST, and read it back.** Minutes after `2.1.1` hit npm, the index line
957
968
  a fresh session loads still read *"⛔ 2.1.1 CANNOT ship from run …2011"* — true when written, and by
958
969
  then the exact opposite of the truth. **The index is what everyone loads and the body is what nobody
@@ -33,7 +33,7 @@
33
33
  # usage: watch-context.sh <role-slug> <agent|pane> <warn-pct> <act-pct> [handoff-file] [poll-s] [cap-s]
34
34
  # <role-slug> is ANY seat role — orchestrator, overseer, surgeon, consult — and names the tier
35
35
  # `<role>-context`. It is deliberately NOT a closed set: see OBS-730 at the guard below.
36
- # TKR_AUTO_CLEAR=1 at act-pct WITH a fresh handoff, send /clear and re-brief instead of waking.
36
+ # TKR_AUTO_CLEAR=1 at act-pct WITH a fresh handoff, auto-clear orchestrator/overseer; other roles wake only.
37
37
  # TKR_REBRIEF=<path> the file the re-briefed seat is told to read (defaults to the handoff).
38
38
  # TKR_HANDOFF_MAX_AGE_S how fresh "fresh" is (default 900).
39
39
  # TKR_CLEAR_SETTLE_S seconds to let a cleared seat settle before the re-brief (default 6).
@@ -221,7 +221,7 @@ handoff_fresh() {
221
221
  act_on() {
222
222
  local P="$1"
223
223
  if handoff_fresh; then
224
- if [ "${TKR_AUTO_CLEAR:-0}" = "1" ]; then
224
+ if [ "${TKR_AUTO_CLEAR:-0}" = "1" ] && { [ "$ROLE" = "orchestrator" ] || [ "$ROLE" = "overseer" ]; }; then
225
225
  herdr agent prompt "$TARGET" "/clear" >/dev/null 2>&1
226
226
  sleep "$SETTLE"
227
227
  herdr agent prompt "$TARGET" "Read ${REBRIEF} and continue exactly where it says. Your context was cleared at ${P}% against that handoff; it is current as of $(date '+%H:%M'). Do not reconstruct from memory — everything you need is on disk." >/dev/null 2>&1