tickmarkr 2.4.0 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/dist/adapters/catalog-remote.d.ts +11 -0
  2. package/dist/adapters/catalog-remote.js +55 -13
  3. package/dist/adapters/codex.js +6 -7
  4. package/dist/adapters/qwen.js +3 -3
  5. package/dist/adapters/registry.js +13 -1
  6. package/dist/cli/commands/compile.d.ts +3 -0
  7. package/dist/cli/commands/compile.js +77 -46
  8. package/dist/cli/commands/doctor.js +14 -9
  9. package/dist/cli/commands/fleet.js +11 -5
  10. package/dist/cli/commands/init.js +12 -13
  11. package/dist/cli/commands/plan.js +47 -7
  12. package/dist/cli/commands/run.js +20 -1
  13. package/dist/cli/commands/status.js +20 -21
  14. package/dist/cli/commands/version.js +2 -2
  15. package/dist/compile/collateral.d.ts +14 -5
  16. package/dist/compile/collateral.js +32 -26
  17. package/dist/compile/index.js +17 -6
  18. package/dist/compile/native.js +10 -2
  19. package/dist/compile/ownership.js +15 -9
  20. package/dist/drivers/herdr.d.ts +1 -0
  21. package/dist/drivers/herdr.js +32 -3
  22. package/dist/drivers/orca.d.ts +5 -1
  23. package/dist/drivers/orca.js +51 -2
  24. package/dist/drivers/types.d.ts +1 -1
  25. package/dist/gates/baseline.d.ts +26 -2
  26. package/dist/gates/baseline.js +90 -7
  27. package/dist/gates/review.d.ts +2 -2
  28. package/dist/gates/review.js +2 -18
  29. package/dist/graph/graph.d.ts +20 -0
  30. package/dist/graph/graph.js +66 -1
  31. package/dist/route/preference.d.ts +3 -1
  32. package/dist/route/preference.js +4 -4
  33. package/dist/run/consult.js +1 -0
  34. package/dist/run/daemon.d.ts +5 -0
  35. package/dist/run/daemon.js +151 -22
  36. package/dist/run/git.d.ts +2 -0
  37. package/dist/run/git.js +18 -4
  38. package/dist/run/journal.d.ts +1 -1
  39. package/dist/run/journal.js +1 -1
  40. package/dist/run/lock.d.ts +6 -0
  41. package/dist/run/lock.js +41 -1
  42. package/package.json +1 -1
  43. package/skills/tickmarkr-overseer/SKILL.md +39 -4
  44. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
  45. package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
@@ -11,6 +11,7 @@ export interface CatalogCache {
11
11
  modelsDev: unknown;
12
12
  artificialAnalysis?: unknown;
13
13
  liveBench?: unknown;
14
+ legFetchedAt?: Partial<Record<"modelsDev" | "artificialAnalysis" | "liveBench", string>>;
14
15
  }
15
16
  export interface CatalogReadResult {
16
17
  catalog: CatalogCache;
@@ -47,11 +48,21 @@ export interface RefreshCatalogOptions {
47
48
  timeoutMs?: number;
48
49
  now?: () => Date;
49
50
  }
51
+ export type CatalogLegName = "models.dev" | "Artificial Analysis" | "LiveBench";
52
+ export type CatalogLegStatus = "updated" | "failed" | "skipped";
53
+ export interface CatalogLegResult {
54
+ leg: CatalogLegName;
55
+ status: CatalogLegStatus;
56
+ detail: string;
57
+ retry: string;
58
+ }
50
59
  export interface RefreshCatalogResult {
51
60
  updated: boolean;
52
61
  catalog: CatalogReadResult;
53
62
  warning?: string;
63
+ legs: CatalogLegResult[];
54
64
  }
65
+ export declare const formatCatalogRefreshLegs: (legs: readonly CatalogLegResult[]) => string;
55
66
  export declare const catalogCachePath: (repoRoot: string) => string;
56
67
  /**
57
68
  * Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
@@ -60,9 +60,14 @@ const validModelsDevCatalog = (value) => {
60
60
  });
61
61
  };
62
62
  const staleAt = (fetchedAt, now) => {
63
- const age = now.getTime() - Date.parse(fetchedAt);
63
+ const age = now.getTime() - Date.parse(fetchedAt ?? "");
64
64
  return !Number.isFinite(age) || age > CATALOG_CACHE_MAX_AGE_MS;
65
65
  };
66
+ const legFetchedAt = (catalog, leg) => catalog.legFetchedAt?.[leg] ?? catalog.fetchedAt;
67
+ const catalogStale = (catalog, now, source) => source === "vendored" || staleAt(legFetchedAt(catalog, "modelsDev"), now) || staleAt(legFetchedAt(catalog, "liveBench"), now)
68
+ || (catalog.artificialAnalysis !== undefined && catalog.legFetchedAt?.artificialAnalysis !== undefined
69
+ && staleAt(catalog.legFetchedAt.artificialAnalysis, now));
70
+ export const formatCatalogRefreshLegs = (legs) => `catalog refresh: ${legs.map((leg) => `${leg.leg} ${leg.status} (${leg.detail}; ${leg.retry})`).join("; ")}`;
66
71
  export const catalogCachePath = (repoRoot) => join(repoRoot, ".tickmarkr", "catalog-cache.json");
67
72
  /**
68
73
  * Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
@@ -74,14 +79,14 @@ export function readCachedCatalog(repoRoot, opts = {}) {
74
79
  const parsed = JSON.parse(readFileSync(catalogCachePath(repoRoot), "utf8"));
75
80
  if (!validCache(parsed))
76
81
  throw new Error("catalog cache schema is invalid");
77
- return { catalog: parsed, source: "cache", stale: staleAt(parsed.fetchedAt, now) };
82
+ return { catalog: parsed, source: "cache", stale: catalogStale(parsed, now, "cache") };
78
83
  }
79
84
  catch (error) {
80
85
  const missing = error.code === "ENOENT";
81
86
  return {
82
87
  catalog: VENDORED_CATALOG,
83
88
  source: "vendored",
84
- stale: staleAt(VENDORED_CATALOG.fetchedAt, now),
89
+ stale: true,
85
90
  ...(!missing ? { warning: error instanceof Error ? error.message : String(error) } : {}),
86
91
  };
87
92
  }
@@ -410,6 +415,13 @@ export async function refreshCatalogCommand(opts) {
410
415
  const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
411
416
  const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
412
417
  const warnings = [];
418
+ const legs = [];
419
+ const stamp = now().toISOString();
420
+ const legAt = {
421
+ modelsDev: legFetchedAt(current.catalog, "modelsDev"),
422
+ liveBench: legFetchedAt(current.catalog, "liveBench"),
423
+ ...(current.catalog.artificialAnalysis !== undefined ? { artificialAnalysis: legFetchedAt(current.catalog, "artificialAnalysis") } : {}),
424
+ };
413
425
  let modelsDev = current.catalog.modelsDev;
414
426
  let modelsDevUpdated = false;
415
427
  try {
@@ -417,46 +429,76 @@ export async function refreshCatalogCommand(opts) {
417
429
  if (!validModelsDevCatalog(modelsDev))
418
430
  throw new Error("models.dev catalog schema is invalid");
419
431
  modelsDevUpdated = true;
432
+ legAt.modelsDev = stamp;
433
+ legs.push({ leg: "models.dev", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
420
434
  }
421
435
  catch (error) {
422
- warnings.push(`models.dev refresh failed: ${error instanceof Error ? error.message : String(error)}`);
436
+ const detail = error instanceof Error ? error.message : String(error);
437
+ warnings.push(`models.dev refresh failed: ${detail}`);
438
+ legs.push({ leg: "models.dev", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
423
439
  }
424
440
  let artificialAnalysis = current.catalog.artificialAnalysis;
441
+ let artificialAnalysisUpdated = false;
425
442
  const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
426
443
  if (apiKey) {
427
444
  try {
428
445
  artificialAnalysis = await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs);
446
+ artificialAnalysisUpdated = true;
447
+ legAt.artificialAnalysis = stamp;
448
+ legs.push({ leg: "Artificial Analysis", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
429
449
  }
430
450
  catch (error) {
431
- warnings.push(`Artificial Analysis refresh failed: ${error instanceof Error ? error.message : String(error)}`);
451
+ const detail = error instanceof Error ? error.message : String(error);
452
+ warnings.push(`Artificial Analysis refresh failed: ${detail}`);
453
+ legs.push({ leg: "Artificial Analysis", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
432
454
  }
433
455
  }
456
+ else {
457
+ delete legAt.artificialAnalysis;
458
+ legs.push({ leg: "Artificial Analysis", status: "skipped", detail: "no API key", retry: "retry when ARTIFICIAL_ANALYSIS_API_KEY is set" });
459
+ }
434
460
  let liveBench = current.catalog.liveBench;
435
461
  let liveBenchUpdated = false;
436
462
  try {
437
463
  liveBench = await fetchLiveBench(fetcher, timeoutMs);
438
464
  liveBenchUpdated = true;
465
+ legAt.liveBench = stamp;
466
+ legs.push({ leg: "LiveBench", status: "updated", detail: `table ${LIVEBENCH_TABLE_DATE}`, retry: "next retry after this leg is stale" });
439
467
  }
440
468
  catch (error) {
441
- warnings.push(`LiveBench refresh failed: ${error instanceof Error ? error.message : String(error)}`);
469
+ const detail = error instanceof Error ? error.message : String(error);
470
+ warnings.push(`LiveBench refresh failed: ${detail}`);
471
+ legs.push({ leg: "LiveBench", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
442
472
  }
443
- // models.dev is the required cache spine. Still run every independent leg after it fails, but do
444
- // not write a cache that would make the vendored fallback look fetched.
445
- if (!modelsDevUpdated) {
446
- return { updated: false, catalog: current, warning: warnings.join("; ") };
473
+ const updated = modelsDevUpdated || artificialAnalysisUpdated || liveBenchUpdated;
474
+ // models.dev is the cache spine: do not write a cache that would make the vendored fallback look
475
+ // fetched. Independent legs may only merge onto models.dev fetched now or already held in cache.
476
+ if (!updated || (!modelsDevUpdated && current.source !== "cache")) {
477
+ const reportedLegs = !modelsDevUpdated && current.source !== "cache"
478
+ ? legs.map((leg) => leg.status === "updated"
479
+ ? {
480
+ ...leg,
481
+ status: "failed",
482
+ detail: "fetched but discarded — models.dev spine unavailable",
483
+ retry: "retry next refresh",
484
+ }
485
+ : leg)
486
+ : legs;
487
+ return { updated: false, catalog: current, legs: reportedLegs, ...(warnings.length ? { warning: warnings.join("; ") } : {}) };
447
488
  }
448
489
  const catalog = {
449
490
  schemaVersion: 1,
450
- // A failed keyless leg keeps the old age so doctor/fleet retry it rather than masking it for 7d.
451
- fetchedAt: liveBenchUpdated ? now().toISOString() : current.catalog.fetchedAt,
491
+ fetchedAt: legAt.modelsDev ?? current.catalog.fetchedAt,
452
492
  modelsDev,
453
493
  ...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
454
494
  ...(liveBench !== undefined ? { liveBench } : {}),
495
+ legFetchedAt: legAt,
455
496
  };
456
497
  writeCatalogCache(opts.repoRoot, catalog);
457
498
  return {
458
- updated: true,
499
+ updated,
459
500
  catalog: readCachedCatalog(opts.repoRoot, { now }),
501
+ legs,
460
502
  ...(warnings.length ? { warning: warnings.join("; ") } : {}),
461
503
  };
462
504
  }
@@ -184,17 +184,16 @@ export const codex = {
184
184
  probeConcurrency: 1,
185
185
  probe: async () => probeVersion("codex"),
186
186
  channels: (cfg) => channelsFromConfig("codex", cfg),
187
- // v1.65 T3: every flag the command builders below hardcode (incl. codexMcpSuppressionFlags' -c/
187
+ // v1.65 T3: every flag the command builder below hardcodes (incl. codexMcpSuppressionFlags' -c/
188
188
  // --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
189
- hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-a", "-s", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
189
+ hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
190
190
  // --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
191
191
  // MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
192
192
  // CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
193
- headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
194
- // TUI uses expanded -a never -s workspace-write (exec-only flags do not apply)
195
- // (--help 2026-07-09: valid approval policies are untrusted|on-request|never; the previously
196
- // used `on-failure` is invalid and made codex exit 2 pre-inference)
197
- interactiveCommand: (promptFile, model) => `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
193
+ headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
194
+ // OBS-889: Codex's TUI has no file/stdin prompt form. Returning null makes the daemon journal
195
+ // worker-mode-fallback before it runs the argv-safe headless command in the visible pane.
196
+ interactiveCommand: () => null,
198
197
  invoke(task, _cwd, a, ctx) {
199
198
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
200
199
  },
@@ -58,7 +58,7 @@ function decodeQwenEvents(events) {
58
58
  }
59
59
  }
60
60
  const assistantText = text.join("\n");
61
- const apiError = text.find((line) => line.startsWith("[API Error:"));
61
+ const apiError = assistantText.match(/\[API Error:[^\n]*/)?.[0];
62
62
  if (apiError)
63
63
  failed = true;
64
64
  if (!failed)
@@ -134,8 +134,8 @@ export const qwen = {
134
134
  probeCwd: "neutral",
135
135
  probe: async () => probeQwen(),
136
136
  channels: (cfg) => channelsFromConfig("qwen", cfg),
137
- hardcodedFlags: { binary: "qwen", flags: ["--approval-mode", "-m", "-o", "-p"] },
138
- headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
137
+ hardcodedFlags: { binary: "qwen", flags: ["--safe-mode", "--approval-mode", "-m", "-o", "-p"] },
138
+ headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --safe-mode --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
139
139
  // OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
140
140
  // argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
141
141
  // can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
@@ -399,6 +399,17 @@ function reasonTail(output) {
399
399
  const sp = tail.indexOf(" ");
400
400
  return sp === -1 ? tail : tail.slice(sp + 1);
401
401
  }
402
+ function parsedStartupFailureReason(adapter, output) {
403
+ try {
404
+ const parsed = adapter.parse(output, "tickmarkr-model-probe");
405
+ const cause = parsed.cause;
406
+ if (!parsed.ok && cause === "startup-failure" && parsed.summary.trim()) {
407
+ return reasonTail(parsed.summary.trim().replace(/\s+/g, " "));
408
+ }
409
+ }
410
+ catch { /* adapter parse is advisory for probe diagnostics */ }
411
+ return undefined;
412
+ }
402
413
  function probeFailure(code, stdout, stderr, timedOut, timeoutMs = MODEL_PROBE_TIMEOUT_MS) {
403
414
  // SIGKILL-timeout is not exit-1: report the budget, never the masked kill code (v1.27 T1).
404
415
  if (timedOut)
@@ -456,7 +467,8 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
456
467
  if (r.timedOut && !retry && priorModelAuth?.[model]?.reason?.includes("timed out") === true) {
457
468
  return { verdict: v(false, `probe timed out (repeat — retry skipped) (${MODEL_PROBE_TIMEOUT_MS}ms)`), timedOut: true };
458
469
  }
459
- const reason = probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
470
+ const output = `${r.stderr}\n${r.stdout}`.trim();
471
+ const reason = parsedStartupFailureReason(a, output) ?? probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
460
472
  if (!reason)
461
473
  return { verdict: v(true), timedOut: false };
462
474
  if (!retry)
@@ -1 +1,4 @@
1
+ import { type SourceScopeFinding } from "../../compile/collateral.js";
2
+ import { compileSource } from "../../compile/index.js";
3
+ export declare function nativeSourceScopeErrors(graph: ReturnType<typeof compileSource>, findings: readonly SourceScopeFinding[]): string[];
1
4
  export declare function compile(argv: string[], cwd?: string, harnessFrom?: string | undefined): Promise<string>;
@@ -1,25 +1,21 @@
1
1
  import { isAbsolute, join } from "node:path";
2
2
  import { parseArgs } from "node:util";
3
- import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
3
+ import { collateralLints, sourceScopeFindings, sourceScopeLints, } from "../../compile/collateral.js";
4
4
  import { CompileError } from "../../compile/common.js";
5
5
  import { compileSource } from "../../compile/index.js";
6
- import { saveGraph, stateDirName } from "../../graph/graph.js";
6
+ import { clearCompileRefusal, saveCompileRefusal, saveGraph, stateDirName } from "../../graph/graph.js";
7
7
  import { formatPriorFindingEvidence, readPriorRunEvidence } from "../../run/journal.js";
8
8
  import { shGit } from "../../run/git.js";
9
9
  import { acquireRunLock, releaseRunLock } from "../../run/lock.js";
10
10
  import { harnessLine, resolveHarness } from "../harness.js";
11
- function nativeSourceScopeErrors(graph, lints) {
11
+ export function nativeSourceScopeErrors(graph, findings) {
12
12
  const byId = new Map(graph.tasks.map((task) => [task.id, task]));
13
13
  const errors = [];
14
- for (const lint of lints) {
15
- const match = lint.match(/^(\S+): criteria implicate out-of-scope source not in files\[\]: (.*)$/);
16
- if (!match)
17
- continue;
18
- const task = byId.get(match[1]);
14
+ for (const finding of findings) {
15
+ const task = byId.get(finding.taskId);
19
16
  if (!task)
20
17
  continue;
21
- const paths = match[2].replace(/ \(capped\)$/, "").split(", ").filter(Boolean);
22
- for (const path of paths) {
18
+ for (const path of finding.paths) {
23
19
  if (task.goal.includes(`scope-waiver: ${path}`))
24
20
  continue;
25
21
  errors.push(`${task.id}: ${path} requires scope-waiver: ${path} in the task goal`);
@@ -60,43 +56,78 @@ export async function compile(argv, cwd = process.cwd(), harnessFrom = process.a
60
56
  const src = positionals[0];
61
57
  if (!src)
62
58
  throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run] [--strict]");
63
- // resolve against the target repo, not the process cwd (the CLI test passes a tmp repo)
64
- // Both modes reach the same pure compiler; --dry-run only removes the lock/write side effect below.
65
- const g = compileSource(isAbsolute(src) ? src : join(cwd, src), values.type, cwd, // repo root: gsd stores context[0] repo-relative so workers resolve it inside their worktree
66
- (plan) => plan, { strict: values.strict });
67
- const scopeLints = [...collateralLints(g.tasks, cwd), ...sourceScopeLints(g.tasks, cwd)];
68
- const diagnostics = scopeLints.length
69
- ? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
70
- : "";
71
- const unwaived = g.spec.source === "native" ? nativeSourceScopeErrors(g, scopeLints) : [];
72
- if (unwaived.length > 0) {
73
- throw new CompileError(`${src} has unwaived native source-scope authoring errors:\n${unwaived.map((line) => ` - ${line}`).join("\n")}${diagnostics}`);
74
- }
75
- // One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
76
- // the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
77
- // remain the source compiler's answer.
78
- const prior = readPriorRunEvidence(cwd, g.tasks);
79
- const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
80
- const stateDir = stateDirName(cwd);
81
- if (!values["dry-run"]) {
82
- // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
83
- // cannot swap graph.json under an active run between the daemon's read and act.
84
- acquireRunLock(cwd, "compile");
85
- try {
86
- saveGraph(cwd, g);
59
+ const sourcePath = isAbsolute(src) ? src : join(cwd, src);
60
+ let stateWriteStarted = false;
61
+ try {
62
+ // Resolve against the target repo, not the process cwd (the CLI test passes a tmp repo).
63
+ // Both modes reach the same pure compiler; --dry-run removes every state write below.
64
+ const g = compileSource(sourcePath, values.type, cwd, // repo root: gsd stores context[0] repo-relative so workers resolve it inside their worktree
65
+ (plan) => plan, { strict: values.strict });
66
+ const sourceFindings = sourceScopeFindings(g.tasks, cwd);
67
+ const scopeLints = [
68
+ ...collateralLints(g.tasks, cwd),
69
+ ...sourceScopeLints(g.tasks, cwd, sourceFindings),
70
+ ];
71
+ const diagnostics = scopeLints.length
72
+ ? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
73
+ : "";
74
+ const unwaived = g.spec.source === "native" ? nativeSourceScopeErrors(g, sourceFindings) : [];
75
+ if (unwaived.length > 0) {
76
+ throw new CompileError(`${src} has unwaived native source-scope authoring errors:\n${unwaived.map((line) => ` - ${line}`).join("\n")}${diagnostics}`);
87
77
  }
88
- finally {
89
- releaseRunLock(cwd);
78
+ // One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
79
+ // the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
80
+ // remain the source compiler's answer.
81
+ const prior = readPriorRunEvidence(cwd, g.tasks);
82
+ const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
83
+ const stateDir = stateDirName(cwd);
84
+ if (!values["dry-run"]) {
85
+ // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around both truth records so
86
+ // compile cannot swap graph.json under an active run between the daemon's read and act.
87
+ stateWriteStarted = true;
88
+ acquireRunLock(cwd, "compile");
89
+ try {
90
+ saveGraph(cwd, g);
91
+ clearCompileRefusal(cwd);
92
+ }
93
+ finally {
94
+ releaseRunLock(cwd);
95
+ }
96
+ }
97
+ const summary = values["dry-run"]
98
+ ? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
99
+ : `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
100
+ const priorFindings = prior.findings.length
101
+ ? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
102
+ : "";
103
+ const mergeHistory = mergedPending.length
104
+ ? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
105
+ : "";
106
+ return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
107
+ }
108
+ catch (error) {
109
+ // A dry run is a pure validation query. A real authoring refusal records the negative result
110
+ // without replacing the last good graph; run treats this sibling as newer truth than that graph.
111
+ // State-write failures are excluded: a live daemon's lock refusal is not a verdict on the spec.
112
+ if (!values["dry-run"] && !stateWriteStarted) {
113
+ try {
114
+ acquireRunLock(cwd, "compile-refusal");
115
+ try {
116
+ saveCompileRefusal(cwd, {
117
+ refusedAt: new Date().toISOString(),
118
+ source: sourcePath,
119
+ error: error instanceof Error ? error.message : String(error),
120
+ });
121
+ }
122
+ finally {
123
+ releaseRunLock(cwd);
124
+ }
125
+ }
126
+ catch {
127
+ // The refusal record is best-effort when a live run owns the state boundary. That lock
128
+ // verdict must never replace the compile error that tells the operator what to repair.
129
+ }
90
130
  }
131
+ throw error;
91
132
  }
92
- const summary = values["dry-run"]
93
- ? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
94
- : `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
95
- const priorFindings = prior.findings.length
96
- ? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
97
- : "";
98
- const mergeHistory = mergedPending.length
99
- ? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
100
- : "";
101
- return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
102
133
  }
@@ -15,13 +15,14 @@ import { HerdrDriver } from "../../drivers/herdr.js";
15
15
  import { ORCA_FIXTURE_VERSION, parseEnvelope, resolveOrcaCliBinary } from "../../drivers/orca.js";
16
16
  import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
17
17
  import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
18
- import { CATALOG_REFRESH_TIMEOUT_MS, LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
18
+ import { CATALOG_REFRESH_TIMEOUT_MS, LIVEBENCH_TABLE_DATE, formatCatalogRefreshLegs, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
19
19
  import { sh } from "../../run/git.js";
20
20
  import { auditNamedTestOracles, listVitestTests } from "../../gates/acceptance.js";
21
21
  /** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
22
22
  * concatenation and publishes no index, so the release listing is the only enumerable surface. */
23
23
  export const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
24
24
  export const LIVEBENCH_TABLE_MAX_AGE_DAYS = 90;
25
+ const initialFetch = globalThis.fetch;
25
26
  const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
26
27
  const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
27
28
  const attentionRow = (text) => ` ${statusRow("warn", text)}`;
@@ -348,19 +349,23 @@ export function liveBenchStalenessFinding(now) {
348
349
  }
349
350
  export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(), opts = {}) {
350
351
  if (_argv.length === 1 && _argv[0] === "--refresh-catalog") {
351
- const refreshed = await refreshCatalogCommand({ repoRoot: cwd, now: opts.catalogNow });
352
+ const refreshed = await refreshCatalogCommand({ repoRoot: cwd, fetcher: opts.catalogFetcher, now: opts.catalogNow });
352
353
  return refreshed.updated
353
- ? `tickmarkr doctor --refresh-catalog: model catalog refreshed${refreshed.warning ? `; ${refreshed.warning}` : ""}`
354
- : `tickmarkr doctor --refresh-catalog: catalog refresh unavailable — ${refreshed.warning ?? "unknown failure"}; retained ${refreshed.catalog.source} catalog`;
354
+ ? `tickmarkr doctor --refresh-catalog: model catalog refreshed — ${formatCatalogRefreshLegs(refreshed.legs)}${refreshed.warning ? `; ${refreshed.warning}` : ""}`
355
+ : `tickmarkr doctor --refresh-catalog: catalog refresh unavailable — ${formatCatalogRefreshLegs(refreshed.legs)}${refreshed.warning ? `; ${refreshed.warning}` : ""}`;
355
356
  }
356
357
  // Operator directive 2026-08-12 (declutter): long per-model lists render only when asked for.
357
358
  const listAllModels = _argv.includes("--models");
358
359
  const cfg = loadConfig(cwd);
359
360
  let catalog = opts.catalog ?? readCachedCatalog(cwd, { now: opts.catalogNow });
360
- let catalogRefreshWarning;
361
+ let catalogRefreshLine;
361
362
  // RULING-222-17 reverses cache-only for these two operator-facing commands only. The refresh
362
363
  // remains age-guarded; compile, plan, and run never import this path.
363
- if (!opts.catalog && catalog.source === "cache" && catalog.stale) {
364
+ const catalogRefreshAllowed = process.env.VITEST !== "true"
365
+ || opts.catalogFetcher !== undefined
366
+ || opts.catalogNow !== undefined
367
+ || globalThis.fetch !== initialFetch;
368
+ if (!opts.catalog && catalog.stale && catalogRefreshAllowed) {
364
369
  const refreshed = await refreshCatalogCommand({
365
370
  repoRoot: cwd,
366
371
  fetcher: opts.catalogFetcher,
@@ -368,7 +373,7 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
368
373
  now: opts.catalogNow,
369
374
  });
370
375
  catalog = refreshed.catalog;
371
- catalogRefreshWarning = refreshed.warning;
376
+ catalogRefreshLine = formatCatalogRefreshLegs(refreshed.legs);
372
377
  }
373
378
  // banner at START — the logo greets the operator before the ~60s probe wait, never trailing it (operator report 2026-07-17)
374
379
  if (opts.banner !== false && visual())
@@ -466,8 +471,8 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
466
471
  if (catalog.warning) {
467
472
  rows.push(attentionRow(`model catalog cache unreadable — ${catalog.warning}; using vendored fallback (advisory — routing unchanged)`));
468
473
  }
469
- if (catalogRefreshWarning) {
470
- rows.push(attentionRow(`model catalog auto-refresh failed open — ${catalogRefreshWarning}; retained ${catalog.source} catalog`));
474
+ if (catalogRefreshLine) {
475
+ rows.push(attentionRow(`model catalog auto-refresh — ${catalogRefreshLine}`));
471
476
  }
472
477
  // v1.48 T1 / v1.86 T12: advisory sweep for known agent CLIs with no drive contract. Presence is
473
478
  // resolved through the worker's login shell; advisory targets are never executed or written to health.
@@ -3,7 +3,7 @@ import { dirname } from "node:path";
3
3
  import { parseArgs } from "node:util";
4
4
  import { allAdapters, discoverChannels, doctorAgeMs, initDoctorReuse, modelAuthExclusions } from "../../adapters/registry.js";
5
5
  import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, fleetUnclassifiedModels } from "../../adapters/model-lints.js";
6
- import { CATALOG_REFRESH_TIMEOUT_MS, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
6
+ import { CATALOG_REFRESH_TIMEOUT_MS, formatCatalogRefreshLegs, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
7
7
  import { CLAUDE_ALIAS_IDENTITY_STAMPS, readClaudeAliasIdentity } from "../../adapters/claude-code.js";
8
8
  import { fleetEditableFromConfig, fleetEditableEquals, formatFleetPrint, globalConfigDir, overlayBytesLoadError, renderFleetOverlayWrite, repoOverlayPath, ROUTING_MODES, unifiedYamlDiff, } from "../../config/config.js";
9
9
  import { projectFleetWhy, renderFleetWhy } from "../../config/fleet-why.js";
@@ -14,6 +14,7 @@ import { route } from "../../route/router.js";
14
14
  import { disallowedBy } from "../../route/preference.js";
15
15
  import { resolveRunMode } from "../../run/daemon.js";
16
16
  import { loadRoutingProfile } from "../../run/journal.js";
17
+ const initialFetch = globalThis.fetch;
17
18
  const NON_TTY_MSG = "tickmarkr fleet: interactive fleet editor requires a TTY — use `tickmarkr fleet --print` for non-interactive output";
18
19
  const QUIT = "fleet: quit without writing";
19
20
  // v1.60 T3: every preview surface ranks with the SAME exploration setting as the candidate picker
@@ -98,16 +99,21 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
98
99
  const interactive = input.isTTY === true && output.isTTY === true;
99
100
  let catalog = readCachedCatalog(cwd, { now: io.catalogNow });
100
101
  let refreshReason = "";
101
- if (catalog.source === "cache" && catalog.stale) {
102
+ let catalogRefreshAttempted = false;
103
+ const catalogRefreshAllowed = process.env.VITEST !== "true"
104
+ || io.catalogFetcher !== undefined
105
+ || io.catalogNow !== undefined
106
+ || globalThis.fetch !== initialFetch;
107
+ if (catalog.stale && catalogRefreshAllowed) {
102
108
  const refreshed = await refreshCatalogCommand({
103
109
  repoRoot: cwd,
104
110
  fetcher: io.catalogFetcher,
105
111
  timeoutMs: CATALOG_REFRESH_TIMEOUT_MS,
106
112
  now: io.catalogNow,
107
113
  });
114
+ catalogRefreshAttempted = true;
108
115
  catalog = refreshed.catalog;
109
- if (refreshed.warning)
110
- refreshReason = `fleet: catalog auto-refresh failed open — ${refreshed.warning}; retained ${catalog.source} catalog`;
116
+ refreshReason = `fleet: catalog auto-refresh — ${formatCatalogRefreshLegs(refreshed.legs)}`;
111
117
  }
112
118
  if (print) {
113
119
  // v1.51 T4: the print surface names the mode and its source layer right under the header —
@@ -129,7 +135,7 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
129
135
  if (!why && interactive) {
130
136
  const { reuse } = initDoctorReuse(cwd, values.fresh ?? false);
131
137
  if (!reuse) {
132
- output.write(`${await doctor([], cwd, adapters, { banner: false, compact: true })}\n`);
138
+ output.write(`${await doctor([], cwd, adapters, { banner: false, compact: true, ...(catalogRefreshAttempted ? { catalog } : {}), catalogFetcher: io.catalogFetcher, catalogNow: io.catalogNow })}\n`);
133
139
  }
134
140
  }
135
141
  const assembled = await assembleFleetEditor(cwd, adapters, io, { globalDir, catalog });
@@ -11,6 +11,7 @@ import { BANNER, kvRow, legend, rule, statusRow, title } from "../../brand.js";
11
11
  import { tickmarkrDir } from "../../graph/graph.js";
12
12
  import { orcaHostDetected } from "../../drivers/index.js";
13
13
  import { Journal } from "../../run/journal.js";
14
+ import { runLockRunId, runStatusLine } from "../../run/lock.js";
14
15
  import { doctor } from "./doctor.js";
15
16
  import { assembleFleetEditor } from "./fleet.js";
16
17
  const SCAFFOLD_SPEC = "tickmarkr.spec.md";
@@ -24,20 +25,18 @@ const ENVIRONMENTS_FOOTER = [
24
25
  " claude code — tickmarkr init --agent installs the /tkr skills + AGENTS.md so Claude Code (or any agent CLI) drives the loop natively",
25
26
  " anywhere — no herdr or Orca terminal? same fail-closed gates, headless subprocess driver (legacy: no herdr? same fail-closed gates, headless subprocess driver when Orca markers are absent)",
26
27
  ].join("\n");
27
- /** Latest journal without a run-end event, if any. */
28
- function activeRunId(cwd) {
29
- const runId = Journal.latestRunId(cwd, { withJournal: true });
28
+ /** Current run truth, lock first; falls back to an abandoned journal. */
29
+ function activeRunLine(cwd) {
30
+ const locked = runLockRunId(cwd);
31
+ const runId = locked ?? Journal.latestRunId(cwd, { withJournal: true });
30
32
  if (!runId)
31
33
  return null;
34
+ let events = [];
32
35
  try {
33
- const events = Journal.open(cwd, runId).read();
34
- if (events.some((e) => e.event === "run-end"))
35
- return null;
36
- return runId;
37
- }
38
- catch {
39
- return null;
36
+ events = Journal.open(cwd, runId).read();
40
37
  }
38
+ catch { /* lock may exist before first journal row */ }
39
+ return runStatusLine(cwd, runId, events);
41
40
  }
42
41
  /** Relative paths of specs/*.spec.md already in the repo. */
43
42
  function existingSpecs(cwd) {
@@ -56,9 +55,9 @@ function existingSpecs(cwd) {
56
55
  }
57
56
  /** Context-aware next-steps line (operator-approved 2026-07-17). */
58
57
  function nextSteps(cwd, scaffoldedSpec) {
59
- const runId = activeRunId(cwd);
60
- if (runId)
61
- return `run ${runId} active — tickmarkr status`;
58
+ const runLine = activeRunLine(cwd);
59
+ if (runLine)
60
+ return `${runLine} — tickmarkr status`;
62
61
  const specs = existingSpecs(cwd);
63
62
  if (specs.length > 0) {
64
63
  const listed = specs.length <= 3 ? specs.join(", ") : `${specs.slice(0, 3).join(", ")}, …`;