tickmarkr 2.3.0 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/dist/adapters/catalog-remote.d.ts +12 -4
  2. package/dist/adapters/catalog-remote.js +97 -45
  3. package/dist/adapters/catalog.js +5 -3
  4. package/dist/adapters/claude-code.d.ts +1 -1
  5. package/dist/adapters/claude-code.js +8 -5
  6. package/dist/adapters/codex.js +6 -7
  7. package/dist/adapters/model-lints.d.ts +9 -5
  8. package/dist/adapters/model-lints.js +56 -15
  9. package/dist/adapters/model-windows.js +11 -0
  10. package/dist/adapters/prompt.js +1 -0
  11. package/dist/adapters/qwen.d.ts +5 -0
  12. package/dist/adapters/qwen.js +153 -0
  13. package/dist/adapters/registry.js +13 -1
  14. package/dist/adapters/types.d.ts +1 -0
  15. package/dist/adapters/types.js +1 -0
  16. package/dist/cli/commands/compile.d.ts +3 -0
  17. package/dist/cli/commands/compile.js +91 -34
  18. package/dist/cli/commands/doctor.d.ts +4 -3
  19. package/dist/cli/commands/doctor.js +28 -9
  20. package/dist/cli/commands/fleet.d.ts +4 -0
  21. package/dist/cli/commands/fleet.js +60 -15
  22. package/dist/cli/commands/init.js +12 -13
  23. package/dist/cli/commands/plan.js +75 -11
  24. package/dist/cli/commands/run.js +20 -1
  25. package/dist/cli/commands/status.d.ts +1 -0
  26. package/dist/cli/commands/status.js +59 -16
  27. package/dist/cli/commands/verify.d.ts +1 -0
  28. package/dist/cli/commands/verify.js +5 -0
  29. package/dist/cli/commands/version.js +2 -2
  30. package/dist/cli/index.d.ts +1 -1
  31. package/dist/cli/index.js +2 -2
  32. package/dist/compile/collateral.d.ts +14 -12
  33. package/dist/compile/collateral.js +32 -33
  34. package/dist/compile/index.d.ts +4 -1
  35. package/dist/compile/index.js +53 -8
  36. package/dist/compile/native.d.ts +4 -2
  37. package/dist/compile/native.js +63 -6
  38. package/dist/compile/ownership.js +41 -10
  39. package/dist/config/config.d.ts +1 -0
  40. package/dist/config/config.js +51 -5
  41. package/dist/drivers/herdr.d.ts +2 -0
  42. package/dist/drivers/herdr.js +43 -4
  43. package/dist/drivers/orca.d.ts +18 -1
  44. package/dist/drivers/orca.js +163 -15
  45. package/dist/drivers/types.d.ts +10 -0
  46. package/dist/gates/baseline.d.ts +26 -2
  47. package/dist/gates/baseline.js +115 -13
  48. package/dist/gates/review.d.ts +6 -4
  49. package/dist/gates/review.js +26 -31
  50. package/dist/gates/run-gates.d.ts +5 -2
  51. package/dist/gates/run-gates.js +34 -17
  52. package/dist/graph/graph.d.ts +20 -0
  53. package/dist/graph/graph.js +66 -1
  54. package/dist/route/preference.d.ts +6 -0
  55. package/dist/route/preference.js +40 -0
  56. package/dist/route/router.js +15 -2
  57. package/dist/run/consult.d.ts +1 -0
  58. package/dist/run/consult.js +35 -7
  59. package/dist/run/daemon.d.ts +9 -0
  60. package/dist/run/daemon.js +266 -59
  61. package/dist/run/git.d.ts +4 -0
  62. package/dist/run/git.js +51 -6
  63. package/dist/run/journal.d.ts +1 -1
  64. package/dist/run/journal.js +5 -2
  65. package/dist/run/lock.d.ts +6 -0
  66. package/dist/run/lock.js +41 -1
  67. package/dist/tui/ink/fleet-app.d.ts +4 -0
  68. package/dist/tui/ink/fleet-app.js +45 -16
  69. package/package.json +59 -1
  70. package/skills/tickmarkr-overseer/SKILL.md +39 -4
  71. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
  72. package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
@@ -0,0 +1,153 @@
1
+ import { spawnSync } from "node:child_process";
2
+ import { parseWorkerResult } from "./prompt.js";
3
+ import { channelsFromConfig, shq, } from "./types.js";
4
+ export const QWEN_VERSION_IDENTITY = /^\d+\.\d+\.\d+/;
5
+ const QWEN_SKIP_UPDATE = "QWEN_CODE_SKIP_UPDATE_CHECK_ONCE=true";
6
+ function decodeQwenEvents(events) {
7
+ const text = [];
8
+ let failed = false;
9
+ let resultText;
10
+ let errorText;
11
+ let denialText;
12
+ for (const event of events) {
13
+ if (!event || typeof event !== "object")
14
+ continue;
15
+ if ("type" in event && event.type === "assistant" && "message" in event) {
16
+ const message = event.message;
17
+ if (message && typeof message === "object" && "content" in message && Array.isArray(message.content)) {
18
+ for (const content of message.content) {
19
+ if (!content || typeof content !== "object" || !(("type" in content) && content.type === "text"))
20
+ continue;
21
+ if ("text" in content && typeof content.text === "string")
22
+ text.push(content.text);
23
+ }
24
+ }
25
+ }
26
+ if ("is_error" in event && event.is_error === true)
27
+ failed = true;
28
+ if ("type" in event && event.type === "result") {
29
+ if (!(("subtype" in event) && event.subtype === "success"))
30
+ failed = true;
31
+ if ("result" in event && typeof event.result === "string" && event.result.trim()) {
32
+ resultText ??= event.result.trim();
33
+ }
34
+ // A no-auth host answers `result/error_during_execution` with the cause under `error.message`
35
+ // (verbatim: .planning/assessments/2026-09-04-qwen-live-worker-form/no-auth-home.stdout).
36
+ const error = "error" in event ? event.error : undefined;
37
+ if (error && typeof error === "object" && "message" in error && typeof error.message === "string" && error.message.trim()) {
38
+ errorText ??= error.message.trim();
39
+ }
40
+ }
41
+ if ("permission_denials" in event && Array.isArray(event.permission_denials) && event.permission_denials.length > 0) {
42
+ failed = true;
43
+ denialText ??= event.permission_denials.map(String).join(", ");
44
+ }
45
+ if (!("stats" in event) || !event.stats || typeof event.stats !== "object" || !("models" in event.stats))
46
+ continue;
47
+ const models = event.stats.models;
48
+ if (!models || typeof models !== "object")
49
+ continue;
50
+ for (const model of Object.values(models)) {
51
+ if (!model || typeof model !== "object" || !("api" in model))
52
+ continue;
53
+ const api = model.api;
54
+ if (!api || typeof api !== "object" || !("totalErrors" in api))
55
+ continue;
56
+ if (typeof api.totalErrors === "number" && api.totalErrors > 0)
57
+ failed = true;
58
+ }
59
+ }
60
+ const assistantText = text.join("\n");
61
+ const apiError = assistantText.match(/\[API Error:[^\n]*/)?.[0];
62
+ if (apiError)
63
+ failed = true;
64
+ if (!failed)
65
+ return { assistantText };
66
+ return {
67
+ assistantText,
68
+ failure: apiError ?? errorText ?? resultText ?? (denialText ? `qwen permission denied: ${denialText}` : "qwen reported a startup failure"),
69
+ };
70
+ }
71
+ // OBS-903: the daemon hands the adapter the WHOLE captured stream, never qwen's bare stdout — the
72
+ // launch script's banner and TICKMARKR_EXIT line sit around it and the subprocess driver appends stderr
73
+ // (the headless yolo warning) into the same buffer. The event array is located, not assumed: the bare
74
+ // buffer first, then the outermost `[{` … `}]` span. A live PONG completion read as "unparseable" and
75
+ // merged only through the harvest path until this (clause-4 probe, RULING-222-42).
76
+ function eventArray(raw) {
77
+ const start = raw.indexOf("[{");
78
+ const end = raw.lastIndexOf("}]");
79
+ for (const text of [raw.trim(), start !== -1 && end > start ? raw.slice(start, end + 2) : ""]) {
80
+ if (!text.startsWith("["))
81
+ continue;
82
+ try {
83
+ const parsed = JSON.parse(text);
84
+ if (Array.isArray(parsed))
85
+ return parsed;
86
+ }
87
+ catch { /* try the next shape */ }
88
+ }
89
+ return undefined;
90
+ }
91
+ // Decode qwen's JSON envelope before scanning only decoded assistant text for the worker trailer.
92
+ export function parseQwenResult(raw, nonce) {
93
+ const malformed = {
94
+ ok: false,
95
+ summary: raw.trim() ? "unparseable qwen JSON event stream" : "qwen produced no JSON event stream",
96
+ deviations: [],
97
+ raw,
98
+ cause: raw.trim() ? "malformed-verdict" : "empty-output",
99
+ };
100
+ const parsed = eventArray(raw);
101
+ if (!parsed)
102
+ return malformed;
103
+ const decoded = decodeQwenEvents(parsed);
104
+ if (decoded.failure !== undefined) {
105
+ return {
106
+ ok: false,
107
+ summary: decoded.failure,
108
+ deviations: [],
109
+ raw,
110
+ cause: "startup-failure",
111
+ };
112
+ }
113
+ return { ...parseWorkerResult(decoded.assistantText, nonce), raw };
114
+ }
115
+ function probeQwen() {
116
+ const result = spawnSync("qwen", ["--version"], { encoding: "utf8", timeout: 10_000 });
117
+ if (result.error || result.status !== 0)
118
+ return { installed: false, authed: false, models: [] };
119
+ const version = (result.stdout || result.stderr).trim().split("\n")[0] ?? "";
120
+ if (!QWEN_VERSION_IDENTITY.test(version)) {
121
+ return { installed: false, authed: false, models: [], note: `qwen binary identity mismatch: ${version || "empty version"}` };
122
+ }
123
+ return {
124
+ installed: true,
125
+ authed: true,
126
+ version,
127
+ models: [],
128
+ note: "auth assumed; verified at dispatch (qwen reports API failures inside exit-0 success envelopes)",
129
+ };
130
+ }
131
+ export const qwen = {
132
+ id: "qwen",
133
+ vendor: "alibaba",
134
+ probeCwd: "neutral",
135
+ probe: async () => probeQwen(),
136
+ channels: (cfg) => channelsFromConfig("qwen", cfg),
137
+ hardcodedFlags: { binary: "qwen", flags: ["--safe-mode", "--approval-mode", "-m", "-o", "-p"] },
138
+ headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --safe-mode --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
139
+ // OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
140
+ // argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
141
+ // can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
142
+ // The headless form runs in the visible pane and the same parser reads it in every driver; the daemon's
143
+ // mode-fallback branch (daemon.ts, `icmd === null`) journals the choice once per run.
144
+ interactiveCommand: () => null,
145
+ invoke(task, _cwd, assignment, ctx) {
146
+ return { command: this.headlessCommand(ctx.promptFile, assignment.model) };
147
+ },
148
+ parse: parseQwenResult,
149
+ trustDialog: {
150
+ kind: "none",
151
+ reason: "qwen 0.21.15 showed no workspace-trust prompt in a fresh repository during the recorded interactive probe (.planning/assessments/2026-09-03-qwen-cli-probe/README.md, 2026-09-03); enabling security.folderTrust falsifies this declaration",
152
+ },
153
+ };
@@ -399,6 +399,17 @@ function reasonTail(output) {
399
399
  const sp = tail.indexOf(" ");
400
400
  return sp === -1 ? tail : tail.slice(sp + 1);
401
401
  }
402
+ function parsedStartupFailureReason(adapter, output) {
403
+ try {
404
+ const parsed = adapter.parse(output, "tickmarkr-model-probe");
405
+ const cause = parsed.cause;
406
+ if (!parsed.ok && cause === "startup-failure" && parsed.summary.trim()) {
407
+ return reasonTail(parsed.summary.trim().replace(/\s+/g, " "));
408
+ }
409
+ }
410
+ catch { /* adapter parse is advisory for probe diagnostics */ }
411
+ return undefined;
412
+ }
402
413
  function probeFailure(code, stdout, stderr, timedOut, timeoutMs = MODEL_PROBE_TIMEOUT_MS) {
403
414
  // SIGKILL-timeout is not exit-1: report the budget, never the masked kill code (v1.27 T1).
404
415
  if (timedOut)
@@ -456,7 +467,8 @@ export async function probeModels(cfg, repoRoot, adapters, health, onProgress) {
456
467
  if (r.timedOut && !retry && priorModelAuth?.[model]?.reason?.includes("timed out") === true) {
457
468
  return { verdict: v(false, `probe timed out (repeat — retry skipped) (${MODEL_PROBE_TIMEOUT_MS}ms)`), timedOut: true };
458
469
  }
459
- const reason = probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
470
+ const output = `${r.stderr}\n${r.stdout}`.trim();
471
+ const reason = parsedStartupFailureReason(a, output) ?? probeFailure(r.code, r.stdout, r.stderr, r.timedOut, MODEL_PROBE_TIMEOUT_MS);
460
472
  if (!reason)
461
473
  return { verdict: v(true), timedOut: false };
462
474
  if (!retry)
@@ -124,6 +124,7 @@ export declare const ADAPTER_PROMPT_GLYPHS: {
124
124
  readonly pi: ">";
125
125
  readonly grok: ">";
126
126
  readonly kimi: ">";
127
+ readonly qwen: ">";
127
128
  readonly omp: ">";
128
129
  readonly agy: ">";
129
130
  readonly "prime-agent": ">";
@@ -151,6 +151,7 @@ export const ADAPTER_PROMPT_GLYPHS = {
151
151
  "pi": ">",
152
152
  "grok": ">",
153
153
  "kimi": ">",
154
+ "qwen": ">",
154
155
  "omp": ">",
155
156
  "agy": ">",
156
157
  "prime-agent": ">",
@@ -1 +1,4 @@
1
+ import { type SourceScopeFinding } from "../../compile/collateral.js";
2
+ import { compileSource } from "../../compile/index.js";
3
+ export declare function nativeSourceScopeErrors(graph: ReturnType<typeof compileSource>, findings: readonly SourceScopeFinding[]): string[];
1
4
  export declare function compile(argv: string[], cwd?: string, harnessFrom?: string | undefined): Promise<string>;
@@ -1,12 +1,28 @@
1
1
  import { isAbsolute, join } from "node:path";
2
2
  import { parseArgs } from "node:util";
3
- import { collateralLints, sourceScopeLints } from "../../compile/collateral.js";
3
+ import { collateralLints, sourceScopeFindings, sourceScopeLints, } from "../../compile/collateral.js";
4
+ import { CompileError } from "../../compile/common.js";
4
5
  import { compileSource } from "../../compile/index.js";
5
- import { saveGraph, stateDirName } from "../../graph/graph.js";
6
+ import { clearCompileRefusal, saveCompileRefusal, saveGraph, stateDirName } from "../../graph/graph.js";
6
7
  import { formatPriorFindingEvidence, readPriorRunEvidence } from "../../run/journal.js";
7
8
  import { shGit } from "../../run/git.js";
8
9
  import { acquireRunLock, releaseRunLock } from "../../run/lock.js";
9
10
  import { harnessLine, resolveHarness } from "../harness.js";
11
+ export function nativeSourceScopeErrors(graph, findings) {
12
+ const byId = new Map(graph.tasks.map((task) => [task.id, task]));
13
+ const errors = [];
14
+ for (const finding of findings) {
15
+ const task = byId.get(finding.taskId);
16
+ if (!task)
17
+ continue;
18
+ for (const path of finding.paths) {
19
+ if (task.goal.includes(`scope-waiver: ${path}`))
20
+ continue;
21
+ errors.push(`${task.id}: ${path} requires scope-waiver: ${path} in the task goal`);
22
+ }
23
+ }
24
+ return errors;
25
+ }
10
26
  async function mergedPendingDiagnostics(cwd, pending, merges) {
11
27
  const head = await shGit("git rev-parse HEAD", cwd);
12
28
  const base = head.code === 0 ? head.stdout.trim() : "";
@@ -33,44 +49,85 @@ export async function compile(argv, cwd = process.cwd(), harnessFrom = process.a
33
49
  options: {
34
50
  type: { type: "string" },
35
51
  "dry-run": { type: "boolean" },
52
+ strict: { type: "boolean" },
36
53
  },
37
54
  allowPositionals: true,
38
55
  });
39
56
  const src = positionals[0];
40
57
  if (!src)
41
- throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run]");
42
- // resolve against the target repo, not the process cwd (the CLI test passes a tmp repo)
43
- // Both modes reach the same pure compiler; --dry-run only removes the lock/write side effect below.
44
- const g = compileSource(isAbsolute(src) ? src : join(cwd, src), values.type, cwd);
45
- // One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
46
- // the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
47
- // remain the source compiler's answer.
48
- const prior = readPriorRunEvidence(cwd, g.tasks);
49
- const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
50
- const stateDir = stateDirName(cwd);
51
- if (!values["dry-run"]) {
52
- // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around saveGraph so compile
53
- // cannot swap graph.json under an active run between the daemon's read and act.
54
- acquireRunLock(cwd, "compile");
55
- try {
56
- saveGraph(cwd, g);
58
+ throw new Error("usage: tickmarkr compile <spec-dir-or-md> [--type speckit|prd|gsd|native] [--dry-run] [--strict]");
59
+ const sourcePath = isAbsolute(src) ? src : join(cwd, src);
60
+ let stateWriteStarted = false;
61
+ try {
62
+ // Resolve against the target repo, not the process cwd (the CLI test passes a tmp repo).
63
+ // Both modes reach the same pure compiler; --dry-run removes every state write below.
64
+ const g = compileSource(sourcePath, values.type, cwd, // repo root: gsd stores context[0] repo-relative so workers resolve it inside their worktree
65
+ (plan) => plan, { strict: values.strict });
66
+ const sourceFindings = sourceScopeFindings(g.tasks, cwd);
67
+ const scopeLints = [
68
+ ...collateralLints(g.tasks, cwd),
69
+ ...sourceScopeLints(g.tasks, cwd, sourceFindings),
70
+ ];
71
+ const diagnostics = scopeLints.length
72
+ ? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
73
+ : "";
74
+ const unwaived = g.spec.source === "native" ? nativeSourceScopeErrors(g, sourceFindings) : [];
75
+ if (unwaived.length > 0) {
76
+ throw new CompileError(`${src} has unwaived native source-scope authoring errors:\n${unwaived.map((line) => ` - ${line}`).join("\n")}${diagnostics}`);
77
+ }
78
+ // One bounded read supplies both cross-run surfaces: unresolved findings below and merge facts for
79
+ // the ancestry check. Neither fact mutates the compiled graph; status and every readiness predicate
80
+ // remain the source compiler's answer.
81
+ const prior = readPriorRunEvidence(cwd, g.tasks);
82
+ const mergedPending = await mergedPendingDiagnostics(cwd, new Set(g.tasks.filter((task) => task.status === "pending").map((task) => task.id)), prior.merges);
83
+ const stateDir = stateDirName(cwd);
84
+ if (!values["dry-run"]) {
85
+ // HARD-01 / Sol #3: hold the same link(2) run lock as the daemon around both truth records so
86
+ // compile cannot swap graph.json under an active run between the daemon's read and act.
87
+ stateWriteStarted = true;
88
+ acquireRunLock(cwd, "compile");
89
+ try {
90
+ saveGraph(cwd, g);
91
+ clearCompileRefusal(cwd);
92
+ }
93
+ finally {
94
+ releaseRunLock(cwd);
95
+ }
57
96
  }
58
- finally {
59
- releaseRunLock(cwd);
97
+ const summary = values["dry-run"]
98
+ ? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
99
+ : `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
100
+ const priorFindings = prior.findings.length
101
+ ? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
102
+ : "";
103
+ const mergeHistory = mergedPending.length
104
+ ? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
105
+ : "";
106
+ return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
107
+ }
108
+ catch (error) {
109
+ // A dry run is a pure validation query. A real authoring refusal records the negative result
110
+ // without replacing the last good graph; run treats this sibling as newer truth than that graph.
111
+ // State-write failures are excluded: a live daemon's lock refusal is not a verdict on the spec.
112
+ if (!values["dry-run"] && !stateWriteStarted) {
113
+ try {
114
+ acquireRunLock(cwd, "compile-refusal");
115
+ try {
116
+ saveCompileRefusal(cwd, {
117
+ refusedAt: new Date().toISOString(),
118
+ source: sourcePath,
119
+ error: error instanceof Error ? error.message : String(error),
120
+ });
121
+ }
122
+ finally {
123
+ releaseRunLock(cwd);
124
+ }
125
+ }
126
+ catch {
127
+ // The refusal record is best-effort when a live run owns the state boundary. That lock
128
+ // verdict must never replace the compile error that tells the operator what to repair.
129
+ }
60
130
  }
131
+ throw error;
61
132
  }
62
- const summary = values["dry-run"]
63
- ? `validated ${src} (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)}) — dry run; no graph written`
64
- : `compiled ${src} → ${stateDir}/graph.json (${g.tasks.length} tasks, source ${g.spec.source}, hash ${g.spec.hash.slice(0, 12)})`;
65
- const scopeLints = [...collateralLints(g.tasks, cwd), ...sourceScopeLints(g.tasks, cwd)];
66
- const diagnostics = scopeLints.length
67
- ? `\nscope lints:\n${scopeLints.map((lint) => ` ! ${lint}`).join("\n")}`
68
- : "";
69
- const priorFindings = prior.findings.length
70
- ? `\nprior-run evidence:\n${prior.findings.map((finding) => ` ${formatPriorFindingEvidence(finding)}`).join("\n")}`
71
- : "";
72
- const mergeHistory = mergedPending.length
73
- ? `\nmerge history:\n${mergedPending.map((line) => ` ${line}`).join("\n")}`
74
- : "";
75
- return `${harnessLine(resolveHarness(harnessFrom))}\n${summary}${diagnostics}${priorFindings}${mergeHistory}`;
76
133
  }
@@ -1,7 +1,7 @@
1
1
  import { type ClaudeAlias } from "../../adapters/claude-code.js";
2
2
  import type { WorkerAdapter } from "../../adapters/types.js";
3
3
  import { type KimiDoctorTurnResult } from "../../adapters/kimi.js";
4
- import { type CatalogReadResult } from "../../adapters/catalog-remote.js";
4
+ import { type CatalogFetcher, type CatalogReadResult } from "../../adapters/catalog-remote.js";
5
5
  import { type ShResult } from "../../run/git.js";
6
6
  import { type VitestListResult } from "../../gates/acceptance.js";
7
7
  /** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
@@ -14,6 +14,7 @@ export type DoctorOpts = {
14
14
  resolveClaudeAliasIdentity?: (cwd: string, alias: ClaudeAlias) => string | undefined;
15
15
  catalog?: CatalogReadResult;
16
16
  catalogNow?: () => Date;
17
+ catalogFetcher?: CatalogFetcher;
17
18
  /** init's between-acts surface: status rows only — the model matrix and inline drift stay
18
19
  * behind `tickmarkr doctor` (files are still written; only the RETURNED string shrinks). */
19
20
  compact?: boolean;
@@ -91,8 +92,8 @@ export declare function selfShadowFinding(ownVersion: string, cwd?: string, reso
91
92
  * §4.4: `LIVEBENCH_TABLE_DATE` is a hand-bumped constant, and a constant nobody lints is a stale
92
93
  * constant — the pinned table ages silently while every tier suggestion keeps citing it. Warns past
93
94
  * 90 days, naming the pinned date and the listing where a newer table is discovered.
94
- * ponytail: a LINT, never a fetch — doctor stays cache-only (T17); the operator (or RELEASING.md's
95
- * release-time check) reads the listing and bumps the constant.
95
+ * ponytail: the listing itself is never fetched; the operator (or RELEASING.md's release-time
96
+ * check) reads it and bumps the constant.
96
97
  */
97
98
  export declare function liveBenchStalenessFinding(now: Date): string | undefined;
98
99
  export declare function doctor(_argv: string[], cwd?: string, adapters?: WorkerAdapter[], opts?: DoctorOpts): Promise<string>;
@@ -15,13 +15,14 @@ import { HerdrDriver } from "../../drivers/herdr.js";
15
15
  import { ORCA_FIXTURE_VERSION, parseEnvelope, resolveOrcaCliBinary } from "../../drivers/orca.js";
16
16
  import { kimi, probeKimiDoctorTurn } from "../../adapters/kimi.js";
17
17
  import { denyPreferCollisionLine, denyPreferCollisions, disallowedBy, excludedChannels, exclusionLine, preferRanks } from "../../route/preference.js";
18
- import { LIVEBENCH_TABLE_DATE, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
18
+ import { CATALOG_REFRESH_TIMEOUT_MS, LIVEBENCH_TABLE_DATE, formatCatalogRefreshLegs, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
19
19
  import { sh } from "../../run/git.js";
20
20
  import { auditNamedTestOracles, listVitestTests } from "../../gates/acceptance.js";
21
21
  /** Where a newer `table_<date>.csv` is discovered — the deployed site builds filenames by
22
22
  * concatenation and publishes no index, so the release listing is the only enumerable surface. */
23
23
  export const LIVEBENCH_RELEASES_URL = "https://api.github.com/repos/LiveBench/livebench.github.io/contents/public";
24
24
  export const LIVEBENCH_TABLE_MAX_AGE_DAYS = 90;
25
+ const initialFetch = globalThis.fetch;
25
26
  const visual = () => process.stdout.isTTY === true && process.env.NO_COLOR === undefined;
26
27
  const alignedStatusRow = (verdict, key, value) => ` ${statusRow(verdict, kvRow(key, value).slice(2))}`;
27
28
  const attentionRow = (text) => ` ${statusRow("warn", text)}`;
@@ -336,8 +337,8 @@ export function selfShadowFinding(ownVersion, cwd = process.cwd(), resolve = res
336
337
  * §4.4: `LIVEBENCH_TABLE_DATE` is a hand-bumped constant, and a constant nobody lints is a stale
337
338
  * constant — the pinned table ages silently while every tier suggestion keeps citing it. Warns past
338
339
  * 90 days, naming the pinned date and the listing where a newer table is discovered.
339
- * ponytail: a LINT, never a fetch — doctor stays cache-only (T17); the operator (or RELEASING.md's
340
- * release-time check) reads the listing and bumps the constant.
340
+ * ponytail: the listing itself is never fetched; the operator (or RELEASING.md's release-time
341
+ * check) reads it and bumps the constant.
341
342
  */
342
343
  export function liveBenchStalenessFinding(now) {
343
344
  const pinnedMs = Date.parse(`${LIVEBENCH_TABLE_DATE.replace(/_/g, "-")}T00:00:00Z`);
@@ -348,17 +349,32 @@ export function liveBenchStalenessFinding(now) {
348
349
  }
349
350
  export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(), opts = {}) {
350
351
  if (_argv.length === 1 && _argv[0] === "--refresh-catalog") {
351
- const refreshed = await refreshCatalogCommand({ repoRoot: cwd, now: opts.catalogNow });
352
+ const refreshed = await refreshCatalogCommand({ repoRoot: cwd, fetcher: opts.catalogFetcher, now: opts.catalogNow });
352
353
  return refreshed.updated
353
- ? `tickmarkr doctor --refresh-catalog: model catalog refreshed${refreshed.warning ? `; ${refreshed.warning}` : ""}`
354
- : `tickmarkr doctor --refresh-catalog: catalog refresh unavailable — ${refreshed.warning ?? "unknown failure"}; retained ${refreshed.catalog.source} catalog`;
354
+ ? `tickmarkr doctor --refresh-catalog: model catalog refreshed — ${formatCatalogRefreshLegs(refreshed.legs)}${refreshed.warning ? `; ${refreshed.warning}` : ""}`
355
+ : `tickmarkr doctor --refresh-catalog: catalog refresh unavailable — ${formatCatalogRefreshLegs(refreshed.legs)}${refreshed.warning ? `; ${refreshed.warning}` : ""}`;
355
356
  }
356
357
  // Operator directive 2026-08-12 (declutter): long per-model lists render only when asked for.
357
358
  const listAllModels = _argv.includes("--models");
358
359
  const cfg = loadConfig(cwd);
359
- // T17: synchronous/cache-only by operator ruling. The only network-bearing catalog function is the
360
- // separately invoked refreshCatalogCommand; ordinary doctor never calls it.
361
- const catalog = opts.catalog ?? readCachedCatalog(cwd, { now: opts.catalogNow });
360
+ let catalog = opts.catalog ?? readCachedCatalog(cwd, { now: opts.catalogNow });
361
+ let catalogRefreshLine;
362
+ // RULING-222-17 reverses cache-only for these two operator-facing commands only. The refresh
363
+ // remains age-guarded; compile, plan, and run never import this path.
364
+ const catalogRefreshAllowed = process.env.VITEST !== "true"
365
+ || opts.catalogFetcher !== undefined
366
+ || opts.catalogNow !== undefined
367
+ || globalThis.fetch !== initialFetch;
368
+ if (!opts.catalog && catalog.stale && catalogRefreshAllowed) {
369
+ const refreshed = await refreshCatalogCommand({
370
+ repoRoot: cwd,
371
+ fetcher: opts.catalogFetcher,
372
+ timeoutMs: CATALOG_REFRESH_TIMEOUT_MS,
373
+ now: opts.catalogNow,
374
+ });
375
+ catalog = refreshed.catalog;
376
+ catalogRefreshLine = formatCatalogRefreshLegs(refreshed.legs);
377
+ }
362
378
  // banner at START — the logo greets the operator before the ~60s probe wait, never trailing it (operator report 2026-07-17)
363
379
  if (opts.banner !== false && visual())
364
380
  process.stdout.write(BANNER);
@@ -455,6 +471,9 @@ export async function doctor(_argv, cwd = process.cwd(), adapters = allAdapters(
455
471
  if (catalog.warning) {
456
472
  rows.push(attentionRow(`model catalog cache unreadable — ${catalog.warning}; using vendored fallback (advisory — routing unchanged)`));
457
473
  }
474
+ if (catalogRefreshLine) {
475
+ rows.push(attentionRow(`model catalog auto-refresh — ${catalogRefreshLine}`));
476
+ }
458
477
  // v1.48 T1 / v1.86 T12: advisory sweep for known agent CLIs with no drive contract. Presence is
459
478
  // resolved through the worker's login shell; advisory targets are never executed or written to health.
460
479
  rows.push(...detectCandidateClis({ cwd }).map(({ binary, path }) => alignedStatusRow("warn", binary, `detected at ${path} (no drive contract — not routable)`)));
@@ -1,3 +1,4 @@
1
+ import { type CatalogFetcher, type CatalogReadResult } from "../../adapters/catalog-remote.js";
1
2
  import type { WorkerAdapter } from "../../adapters/types.js";
2
3
  import type { FleetEditorResult } from "../../tui/ink/fleet-app.js";
3
4
  type FleetEditorProps = Parameters<typeof import("../../tui/ink/fleet-app.js").runFleetInkEditor>[0];
@@ -16,6 +17,8 @@ export type FleetIO = {
16
17
  output?: FleetOutput;
17
18
  debug?: boolean;
18
19
  reloadGuard?: (bytes: string) => string | null;
20
+ catalogFetcher?: CatalogFetcher;
21
+ catalogNow?: () => Date;
19
22
  };
20
23
  export type FleetWriteHooks = {
21
24
  readPrior?: (path: string) => string;
@@ -36,6 +39,7 @@ export declare function fleet(argv: string[], cwd?: string, adapters?: WorkerAda
36
39
  export declare function assembleFleetEditor(cwd: string, adapters: WorkerAdapter[], io: FleetIO, opts: {
37
40
  globalDir: string;
38
41
  entry?: "presets" | "probe";
42
+ catalog?: CatalogReadResult;
39
43
  }): Promise<{
40
44
  props: FleetEditorProps;
41
45
  commit: (result: FleetEditorResult) => string;
@@ -3,7 +3,7 @@ import { dirname } from "node:path";
3
3
  import { parseArgs } from "node:util";
4
4
  import { allAdapters, discoverChannels, doctorAgeMs, initDoctorReuse, modelAuthExclusions } from "../../adapters/registry.js";
5
5
  import { catalogModelAdvisory, catalogTierRanking, declaredModelWindow, fleetUnclassifiedModels } from "../../adapters/model-lints.js";
6
- import { readCachedCatalog } from "../../adapters/catalog-remote.js";
6
+ import { CATALOG_REFRESH_TIMEOUT_MS, formatCatalogRefreshLegs, readCachedCatalog, refreshCatalogCommand } from "../../adapters/catalog-remote.js";
7
7
  import { CLAUDE_ALIAS_IDENTITY_STAMPS, readClaudeAliasIdentity } from "../../adapters/claude-code.js";
8
8
  import { fleetEditableFromConfig, fleetEditableEquals, formatFleetPrint, globalConfigDir, overlayBytesLoadError, renderFleetOverlayWrite, repoOverlayPath, ROUTING_MODES, unifiedYamlDiff, } from "../../config/config.js";
9
9
  import { projectFleetWhy, renderFleetWhy } from "../../config/fleet-why.js";
@@ -14,6 +14,7 @@ import { route } from "../../route/router.js";
14
14
  import { disallowedBy } from "../../route/preference.js";
15
15
  import { resolveRunMode } from "../../run/daemon.js";
16
16
  import { loadRoutingProfile } from "../../run/journal.js";
17
+ const initialFetch = globalThis.fetch;
17
18
  const NON_TTY_MSG = "tickmarkr fleet: interactive fleet editor requires a TTY — use `tickmarkr fleet --print` for non-interactive output";
18
19
  const QUIT = "fleet: quit without writing";
19
20
  // v1.60 T3: every preview surface ranks with the SAME exploration setting as the candidate picker
@@ -96,6 +97,24 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
96
97
  const input = io.input ?? process.stdin;
97
98
  const output = io.output ?? process.stdout;
98
99
  const interactive = input.isTTY === true && output.isTTY === true;
100
+ let catalog = readCachedCatalog(cwd, { now: io.catalogNow });
101
+ let refreshReason = "";
102
+ let catalogRefreshAttempted = false;
103
+ const catalogRefreshAllowed = process.env.VITEST !== "true"
104
+ || io.catalogFetcher !== undefined
105
+ || io.catalogNow !== undefined
106
+ || globalThis.fetch !== initialFetch;
107
+ if (catalog.stale && catalogRefreshAllowed) {
108
+ const refreshed = await refreshCatalogCommand({
109
+ repoRoot: cwd,
110
+ fetcher: io.catalogFetcher,
111
+ timeoutMs: CATALOG_REFRESH_TIMEOUT_MS,
112
+ now: io.catalogNow,
113
+ });
114
+ catalogRefreshAttempted = true;
115
+ catalog = refreshed.catalog;
116
+ refreshReason = `fleet: catalog auto-refresh — ${formatCatalogRefreshLegs(refreshed.legs)}`;
117
+ }
99
118
  if (print) {
100
119
  // v1.51 T4: the print surface names the mode and its source layer right under the header —
101
120
  // comment-prefixed so the YAML body stays machine-parseable and regex-stable.
@@ -104,10 +123,10 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
104
123
  const nl = body.indexOf("\n");
105
124
  // Steering comes from the same resolved config snapshot the editor consumes below,
106
125
  // not from another parse of either raw overlay.
107
- return `${body.slice(0, nl)}\n# mode: ${rm.mode.mode} (${rm.source})${body.slice(nl)}${formatFleetSteering(rm.cfg)}`;
126
+ return `${body.slice(0, nl)}\n# mode: ${rm.mode.mode} (${rm.source})${refreshReason ? `\n# ${refreshReason}` : ""}${body.slice(nl)}${formatFleetSteering(rm.cfg)}`;
108
127
  }
109
128
  if (!why && !interactive)
110
- return { out: NON_TTY_MSG, code: 1 };
129
+ return { out: `${refreshReason ? `${refreshReason}\n` : ""}${NON_TTY_MSG}`, code: 1 };
111
130
  // OBS-528: `--fresh` parsed since v1.92 but nothing ever RAN the probe — it forced the reuse
112
131
  // gate false and guaranteed the "probe data missing or stale" refusal. The law stands — the
113
132
  // EDITOR never probes (previews stay cache-only, `r` still exits to the operator) — but the
@@ -116,14 +135,14 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
116
135
  if (!why && interactive) {
117
136
  const { reuse } = initDoctorReuse(cwd, values.fresh ?? false);
118
137
  if (!reuse) {
119
- output.write(`${await doctor([], cwd, adapters, { banner: false, compact: true })}\n`);
138
+ output.write(`${await doctor([], cwd, adapters, { banner: false, compact: true, ...(catalogRefreshAttempted ? { catalog } : {}), catalogFetcher: io.catalogFetcher, catalogNow: io.catalogNow })}\n`);
120
139
  }
121
140
  }
122
- const assembled = await assembleFleetEditor(cwd, adapters, io, { globalDir });
141
+ const assembled = await assembleFleetEditor(cwd, adapters, io, { globalDir, catalog });
123
142
  if ("unavailable" in assembled)
124
- return { out: assembled.unavailable, code: 1 };
143
+ return { out: `${refreshReason ? `${refreshReason}\n` : ""}${assembled.unavailable}`, code: 1 };
125
144
  if (why)
126
- return assembled.renderWhy();
145
+ return `${refreshReason ? `${refreshReason}\n` : ""}${assembled.renderWhy()}`;
127
146
  // Keep Ink out of print, non-TTY, and missing-probe paths: the component runtime belongs
128
147
  // exclusively to the interactive editor. The capture window brackets only the dynamic Ink
129
148
  // import — keys typed while the module loads land in props.initialInput, never get dropped.
@@ -155,7 +174,7 @@ export async function fleet(argv, cwd = process.cwd(), adapters = allAdapters(),
155
174
  }
156
175
  assembled.props.initialInput = initialInput;
157
176
  const result = await runFleetInkEditor(assembled.props);
158
- return assembled.commit(result);
177
+ return `${refreshReason ? `${refreshReason}\n` : ""}${assembled.commit(result)}`;
159
178
  }
160
179
  /** Everything between the doctor-reuse gate and the Ink render, packaged for reuse: `tickmarkr init`
161
180
  * runs the same editor (entry="presets": Esc is HOME to the preset overlay, never a bare quit)
@@ -192,7 +211,7 @@ export async function assembleFleetEditor(cwd, adapters, io, opts) {
192
211
  // OBS-508: the same catalog evidence doctor's drift overlay prints now rides each unclassified
193
212
  // row — it prefills the classify flow and feeds the bulk `s` stage. Suggestions stay advisory:
194
213
  // only the review-diff confirm writes, so "tickmarkr never applies" holds with less typing.
195
- const catalog = readCachedCatalog(cwd);
214
+ const catalog = opts.catalog ?? readCachedCatalog(cwd);
196
215
  // The same alias→identity resolution doctor hands its advisory rows: models.dev has never heard of
197
216
  // `opus`, so without it the fleet's own frontier models drop out of the universe they are supposed
198
217
  // to anchor and fleet bands a different set than doctor for one fleet. The stored identity wins;
@@ -203,9 +222,32 @@ export async function assembleFleetEditor(cwd, adapters, io, opts) {
203
222
  : readClaudeAliasIdentity(cwd, model) ?? CLAUDE_ALIAS_IDENTITY_STAMPS[model];
204
223
  // One ranking universe for the whole screen: every unclassified row bands fleet-relatively
205
224
  // against the same set, so a suggestion never depends on which adapter group renders first.
206
- const unclassifiedRows = fleetUnclassifiedModels(cfg, health, adapters)
225
+ const detectedRows = fleetUnclassifiedModels(cfg, health, adapters)
207
226
  .map((row) => ({ ...row, resolvedModel: resolvedCatalogModel(row.adapter, row.model) }));
208
- const catalogRanking = catalogTierRanking(cfg, catalog, unclassifiedRows, resolvedCatalogModel);
227
+ const catalogRanking = catalogTierRanking(cfg, catalog, detectedRows, resolvedCatalogModel);
228
+ const foldedRows = [];
229
+ const folds = new Map();
230
+ for (const row of detectedRows) {
231
+ const advisory = catalogModelAdvisory(cfg, catalog, row.adapter, row.model, row.resolvedModel, catalogRanking);
232
+ const evidence = advisory.coverage === "covered" ? advisory.evidence : undefined;
233
+ const score = evidence?.agenticCodingScore ?? evidence?.intelligenceIndex ?? evidence?.codingScore;
234
+ const next = { ...row, advisory, ...(score !== undefined ? { score } : {}) };
235
+ const foldKey = evidence ? `${row.adapter}:${evidence.catalogId}` : undefined;
236
+ const first = foldKey ? folds.get(foldKey) : undefined;
237
+ if (first) {
238
+ first.foldedModels ??= [first.model];
239
+ first.foldedModels.push(row.model);
240
+ }
241
+ else {
242
+ foldedRows.push(next);
243
+ if (foldKey)
244
+ folds.set(foldKey, next);
245
+ }
246
+ }
247
+ const unclassifiedRows = foldedRows.sort((a, b) => Number(!!b.advisory.suggestion) - Number(!!a.advisory.suggestion)
248
+ || (b.detectedAt ?? "").localeCompare(a.detectedAt ?? "")
249
+ || (b.score ?? Number.NEGATIVE_INFINITY) - (a.score ?? Number.NEGATIVE_INFINITY)
250
+ || a.model.localeCompare(b.model));
209
251
  // OBS-508 follow-through: the browser renders the metadata the assembler always had — catalog
210
252
  // ctx/price evidence and doctor's model-probe wall clock — as columns instead of dropping them.
211
253
  const rowEvidence = (adapter, model, evidence) => {
@@ -239,14 +281,17 @@ export async function assembleFleetEditor(cwd, adapters, io, opts) {
239
281
  ...unclassified
240
282
  .filter((row) => !editable.tiers[adapter.id]?.[row.model])
241
283
  .map((row) => {
242
- const advisory = catalogModelAdvisory(cfg, catalog, adapter.id, row.model, row.resolvedModel, catalogRanking);
243
- const evidence = rowEvidence(adapter.id, row.model, advisory.coverage === "covered" ? advisory.evidence : undefined);
284
+ const evidence = rowEvidence(adapter.id, row.classifyModel ?? row.model, row.advisory.coverage === "covered" ? row.advisory.evidence : undefined);
244
285
  return {
245
286
  model: row.model,
246
287
  detectedAt: row.detectedAt,
288
+ ...(row.classifyModel ? { classifyModel: row.classifyModel } : {}),
289
+ ...(row.variants ? { variants: row.variants } : {}),
290
+ ...(row.foldedModels ? { foldedModels: row.foldedModels } : {}),
291
+ ...(row.score !== undefined ? { score: row.score } : {}),
247
292
  ...(evidence ? { evidence } : {}),
248
- ...(advisory.suggestion
249
- ? { suggestion: { tier: advisory.suggestion.tier, note: advisory.suggestion.provenanceNote } }
293
+ ...(row.advisory.suggestion
294
+ ? { suggestion: { tier: row.advisory.suggestion.tier, note: row.advisory.suggestion.provenanceNote } }
250
295
  : {}),
251
296
  };
252
297
  }),