tickmarkr 2.3.0 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/dist/adapters/catalog-remote.d.ts +12 -4
  2. package/dist/adapters/catalog-remote.js +97 -45
  3. package/dist/adapters/catalog.js +5 -3
  4. package/dist/adapters/claude-code.d.ts +1 -1
  5. package/dist/adapters/claude-code.js +8 -5
  6. package/dist/adapters/codex.js +6 -7
  7. package/dist/adapters/model-lints.d.ts +9 -5
  8. package/dist/adapters/model-lints.js +56 -15
  9. package/dist/adapters/model-windows.js +11 -0
  10. package/dist/adapters/prompt.js +1 -0
  11. package/dist/adapters/qwen.d.ts +5 -0
  12. package/dist/adapters/qwen.js +153 -0
  13. package/dist/adapters/registry.js +13 -1
  14. package/dist/adapters/types.d.ts +1 -0
  15. package/dist/adapters/types.js +1 -0
  16. package/dist/cli/commands/compile.d.ts +3 -0
  17. package/dist/cli/commands/compile.js +91 -34
  18. package/dist/cli/commands/doctor.d.ts +4 -3
  19. package/dist/cli/commands/doctor.js +28 -9
  20. package/dist/cli/commands/fleet.d.ts +4 -0
  21. package/dist/cli/commands/fleet.js +60 -15
  22. package/dist/cli/commands/init.js +12 -13
  23. package/dist/cli/commands/plan.js +75 -11
  24. package/dist/cli/commands/run.js +20 -1
  25. package/dist/cli/commands/status.d.ts +1 -0
  26. package/dist/cli/commands/status.js +59 -16
  27. package/dist/cli/commands/verify.d.ts +1 -0
  28. package/dist/cli/commands/verify.js +5 -0
  29. package/dist/cli/commands/version.js +2 -2
  30. package/dist/cli/index.d.ts +1 -1
  31. package/dist/cli/index.js +2 -2
  32. package/dist/compile/collateral.d.ts +14 -12
  33. package/dist/compile/collateral.js +32 -33
  34. package/dist/compile/index.d.ts +4 -1
  35. package/dist/compile/index.js +53 -8
  36. package/dist/compile/native.d.ts +4 -2
  37. package/dist/compile/native.js +63 -6
  38. package/dist/compile/ownership.js +41 -10
  39. package/dist/config/config.d.ts +1 -0
  40. package/dist/config/config.js +51 -5
  41. package/dist/drivers/herdr.d.ts +2 -0
  42. package/dist/drivers/herdr.js +43 -4
  43. package/dist/drivers/orca.d.ts +18 -1
  44. package/dist/drivers/orca.js +163 -15
  45. package/dist/drivers/types.d.ts +10 -0
  46. package/dist/gates/baseline.d.ts +26 -2
  47. package/dist/gates/baseline.js +115 -13
  48. package/dist/gates/review.d.ts +6 -4
  49. package/dist/gates/review.js +26 -31
  50. package/dist/gates/run-gates.d.ts +5 -2
  51. package/dist/gates/run-gates.js +34 -17
  52. package/dist/graph/graph.d.ts +20 -0
  53. package/dist/graph/graph.js +66 -1
  54. package/dist/route/preference.d.ts +6 -0
  55. package/dist/route/preference.js +40 -0
  56. package/dist/route/router.js +15 -2
  57. package/dist/run/consult.d.ts +1 -0
  58. package/dist/run/consult.js +35 -7
  59. package/dist/run/daemon.d.ts +9 -0
  60. package/dist/run/daemon.js +266 -59
  61. package/dist/run/git.d.ts +4 -0
  62. package/dist/run/git.js +51 -6
  63. package/dist/run/journal.d.ts +1 -1
  64. package/dist/run/journal.js +5 -2
  65. package/dist/run/lock.d.ts +6 -0
  66. package/dist/run/lock.js +41 -1
  67. package/dist/tui/ink/fleet-app.d.ts +4 -0
  68. package/dist/tui/ink/fleet-app.js +45 -16
  69. package/package.json +59 -1
  70. package/skills/tickmarkr-overseer/SKILL.md +39 -4
  71. package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
  72. package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
@@ -11,6 +11,7 @@ export interface CatalogCache {
11
11
  modelsDev: unknown;
12
12
  artificialAnalysis?: unknown;
13
13
  liveBench?: unknown;
14
+ legFetchedAt?: Partial<Record<"modelsDev" | "artificialAnalysis" | "liveBench", string>>;
14
15
  }
15
16
  export interface CatalogReadResult {
16
17
  catalog: CatalogCache;
@@ -47,11 +48,21 @@ export interface RefreshCatalogOptions {
47
48
  timeoutMs?: number;
48
49
  now?: () => Date;
49
50
  }
51
+ export type CatalogLegName = "models.dev" | "Artificial Analysis" | "LiveBench";
52
+ export type CatalogLegStatus = "updated" | "failed" | "skipped";
53
+ export interface CatalogLegResult {
54
+ leg: CatalogLegName;
55
+ status: CatalogLegStatus;
56
+ detail: string;
57
+ retry: string;
58
+ }
50
59
  export interface RefreshCatalogResult {
51
60
  updated: boolean;
52
61
  catalog: CatalogReadResult;
53
62
  warning?: string;
63
+ legs: CatalogLegResult[];
54
64
  }
65
+ export declare const formatCatalogRefreshLegs: (legs: readonly CatalogLegResult[]) => string;
55
66
  export declare const catalogCachePath: (repoRoot: string) => string;
56
67
  /**
57
68
  * Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
@@ -66,9 +77,6 @@ export declare function resolveCatalogModel(catalog: CatalogCache, query: {
66
77
  model: string;
67
78
  resolvedModel?: string;
68
79
  }): CatalogModelEvidence | undefined;
69
- /**
70
- * The named, explicit refresh path. No other function in this module can reach fetch.
71
- * A failed refresh preserves the previous cache byte-for-byte and returns it fail-open.
72
- */
80
+ /** Shared refresh path for the explicit command and the seven-day doctor/fleet guard. */
73
81
  export declare function refreshCatalogCommand(opts: RefreshCatalogOptions): Promise<RefreshCatalogResult>;
74
82
  export {};
@@ -9,20 +9,21 @@ export const ARTIFICIAL_ANALYSIS_CATALOG_URL = "https://artificialanalysis.ai/ap
9
9
  export const LIVEBENCH_TABLE_DATE = "2026_06_25";
10
10
  export const LIVEBENCH_TABLE_URL = `https://livebench.ai/table_${LIVEBENCH_TABLE_DATE}.csv`;
11
11
  export const LIVEBENCH_CATEGORIES_URL = `https://livebench.ai/categories_${LIVEBENCH_TABLE_DATE}.json`;
12
- export const CATALOG_CACHE_MAX_AGE_MS = 30 * 86_400_000;
12
+ export const CATALOG_CACHE_MAX_AGE_MS = 7 * 86_400_000;
13
13
  export const CATALOG_REFRESH_TIMEOUT_MS = 10_000;
14
14
  const ARTIFICIAL_ANALYSIS_PAGE_SIZE = 100;
15
15
  const ARTIFICIAL_ANALYSIS_MAX_PAGES = 100;
16
16
  const VENDORED_CATALOG = {
17
17
  schemaVersion: 1,
18
18
  // Package fallback copied from https://models.dev/api.json on 2026-08-05. It is deliberately
19
- // small: an explicit refresh owns broad/current coverage, while doctor remains cache-only.
19
+ // small: the explicit or seven-day operator-surface refresh owns broad/current coverage.
20
20
  fetchedAt: "2026-08-05T20:00:05.000Z",
21
21
  modelsDev: {
22
22
  anthropic: {
23
23
  id: "anthropic",
24
24
  models: {
25
- "claude-fable-5": { id: "claude-fable-5", cost: { input: 10, output: 50 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
25
+ // SHIP (OBS-871, 2026-09-03): users resolving the current Fable alias need the 5.1 identity even when refresh is unavailable.
26
+ "claude-fable-5-1": { id: "claude-fable-5-1", cost: { input: 10, output: 50 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
26
27
  "claude-opus-4-8": { id: "claude-opus-4-8", cost: { input: 5, output: 25 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
27
28
  "claude-sonnet-5": { id: "claude-sonnet-5", cost: { input: 2, output: 10 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
28
29
  "claude-haiku-4-5-20251001": { id: "claude-haiku-4-5-20251001", cost: { input: 1, output: 5 }, limit: { context: 200_000, output: 64_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
@@ -59,9 +60,14 @@ const validModelsDevCatalog = (value) => {
59
60
  });
60
61
  };
61
62
  const staleAt = (fetchedAt, now) => {
62
- const age = now.getTime() - Date.parse(fetchedAt);
63
+ const age = now.getTime() - Date.parse(fetchedAt ?? "");
63
64
  return !Number.isFinite(age) || age > CATALOG_CACHE_MAX_AGE_MS;
64
65
  };
66
+ const legFetchedAt = (catalog, leg) => catalog.legFetchedAt?.[leg] ?? catalog.fetchedAt;
67
+ const catalogStale = (catalog, now, source) => source === "vendored" || staleAt(legFetchedAt(catalog, "modelsDev"), now) || staleAt(legFetchedAt(catalog, "liveBench"), now)
68
+ || (catalog.artificialAnalysis !== undefined && catalog.legFetchedAt?.artificialAnalysis !== undefined
69
+ && staleAt(catalog.legFetchedAt.artificialAnalysis, now));
70
+ export const formatCatalogRefreshLegs = (legs) => `catalog refresh: ${legs.map((leg) => `${leg.leg} ${leg.status} (${leg.detail}; ${leg.retry})`).join("; ")}`;
65
71
  export const catalogCachePath = (repoRoot) => join(repoRoot, ".tickmarkr", "catalog-cache.json");
66
72
  /**
67
73
  * Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
@@ -73,14 +79,14 @@ export function readCachedCatalog(repoRoot, opts = {}) {
73
79
  const parsed = JSON.parse(readFileSync(catalogCachePath(repoRoot), "utf8"));
74
80
  if (!validCache(parsed))
75
81
  throw new Error("catalog cache schema is invalid");
76
- return { catalog: parsed, source: "cache", stale: staleAt(parsed.fetchedAt, now) };
82
+ return { catalog: parsed, source: "cache", stale: catalogStale(parsed, now, "cache") };
77
83
  }
78
84
  catch (error) {
79
85
  const missing = error.code === "ENOENT";
80
86
  return {
81
87
  catalog: VENDORED_CATALOG,
82
88
  source: "vendored",
83
- stale: staleAt(VENDORED_CATALOG.fetchedAt, now),
89
+ stale: true,
84
90
  ...(!missing ? { warning: error instanceof Error ? error.message : String(error) } : {}),
85
91
  };
86
92
  }
@@ -226,6 +232,7 @@ function assertUsableLiveBench(rows, categories) {
226
232
  // `-thinking-auto-medium-effort`); the highest effort is the model at its best. Rank by
227
233
  // hyphen-delimited token so `xhigh` never reads as `high`.
228
234
  const LIVEBENCH_EFFORT_RANK = { max: 5, xhigh: 4, high: 3, medium: 2, low: 1 };
235
+ const LIVEBENCH_VARIANT_RE = /^-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast)(?:-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast))*$/;
229
236
  const liveBenchEffortRank = (residue) => residue.split("-").reduce((rank, token) => Math.max(rank, LIVEBENCH_EFFORT_RANK[token] ?? 0), 0);
230
237
  const liveBenchCategoryMean = (row, tasks) => {
231
238
  const scores = (Array.isArray(tasks) ? tasks : [])
@@ -233,8 +240,8 @@ const liveBenchCategoryMean = (row, tasks) => {
233
240
  .filter((score) => score !== undefined);
234
241
  return scores.length > 0 ? scores.reduce((sum, score) => sum + score, 0) / scores.length : undefined;
235
242
  };
236
- /** LiveBench hyphenates where models.dev spaces: `Kimi K3` is `kimi-k3` in the table. */
237
- const liveBenchIdentity = (value) => value.trim().toLowerCase().replace(/\s+/g, "-");
243
+ /** LiveBench treats spaces, dots, and punctuation as the same word boundary. */
244
+ const liveBenchIdentity = (value) => value.trim().toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
238
245
  function liveBenchIndex(value, identities) {
239
246
  const root = record(value);
240
247
  if (!root || !Array.isArray(root.rows))
@@ -249,7 +256,7 @@ function liveBenchIndex(value, identities) {
249
256
  // Bare startsWith is a false-positive machine: `glm-5` would claim `glm-5.2`. The residue after
250
257
  // the fleet id must be empty or an effort suffix.
251
258
  const residue = wanted.map((identity) => model.startsWith(identity) ? model.slice(identity.length) : undefined)
252
- .find((rest) => rest === "" || rest?.startsWith("-"));
259
+ .find((rest) => rest === "" || LIVEBENCH_VARIANT_RE.test(rest ?? ""));
253
260
  if (residue === undefined)
254
261
  continue;
255
262
  const rank = liveBenchEffortRank(residue);
@@ -285,6 +292,7 @@ export function resolveCatalogModel(catalog, query) {
285
292
  const identities = [
286
293
  modelId,
287
294
  query.model,
295
+ ...[modelId, query.model].flatMap((identity) => identity.includes("/") ? [identity.slice(identity.lastIndexOf("/") + 1)] : []),
288
296
  ...(typeof model.name === "string" ? [model.name] : []),
289
297
  ];
290
298
  const intelligence = artificialAnalysisIndex(catalog.artificialAnalysis, [
@@ -400,53 +408,97 @@ async function fetchLiveBench(fetcher, timeoutMs) {
400
408
  assertUsableLiveBench(rows, categories);
401
409
  return { tableDate: LIVEBENCH_TABLE_DATE, categories, rows };
402
410
  }
403
- /**
404
- * The named, explicit refresh path. No other function in this module can reach fetch.
405
- * A failed refresh preserves the previous cache byte-for-byte and returns it fail-open.
406
- */
411
+ /** Shared refresh path for the explicit command and the seven-day doctor/fleet guard. */
407
412
  export async function refreshCatalogCommand(opts) {
408
413
  const now = opts.now ?? (() => new Date());
409
414
  const current = readCachedCatalog(opts.repoRoot, { now });
415
+ const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
416
+ const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
417
+ const warnings = [];
418
+ const legs = [];
419
+ const stamp = now().toISOString();
420
+ const legAt = {
421
+ modelsDev: legFetchedAt(current.catalog, "modelsDev"),
422
+ liveBench: legFetchedAt(current.catalog, "liveBench"),
423
+ ...(current.catalog.artificialAnalysis !== undefined ? { artificialAnalysis: legFetchedAt(current.catalog, "artificialAnalysis") } : {}),
424
+ };
425
+ let modelsDev = current.catalog.modelsDev;
426
+ let modelsDevUpdated = false;
410
427
  try {
411
- const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
412
- const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
413
- const modelsDev = await fetchCatalog(fetcher, MODELS_DEV_CATALOG_URL, {}, timeoutMs, (r) => r.json());
428
+ modelsDev = await fetchCatalog(fetcher, MODELS_DEV_CATALOG_URL, {}, timeoutMs, (r) => r.json());
414
429
  if (!validModelsDevCatalog(modelsDev))
415
430
  throw new Error("models.dev catalog schema is invalid");
416
- const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
417
- const artificialAnalysis = apiKey
418
- ? await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs)
419
- : undefined;
420
- // The LiveBench leg is keyless and never costs the models.dev refresh: a failure keeps the
421
- // previous section verbatim and names the leg in the warning.
422
- let liveBench = current.catalog.liveBench;
423
- let warning;
431
+ modelsDevUpdated = true;
432
+ legAt.modelsDev = stamp;
433
+ legs.push({ leg: "models.dev", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
434
+ }
435
+ catch (error) {
436
+ const detail = error instanceof Error ? error.message : String(error);
437
+ warnings.push(`models.dev refresh failed: ${detail}`);
438
+ legs.push({ leg: "models.dev", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
439
+ }
440
+ let artificialAnalysis = current.catalog.artificialAnalysis;
441
+ let artificialAnalysisUpdated = false;
442
+ const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
443
+ if (apiKey) {
424
444
  try {
425
- liveBench = await fetchLiveBench(fetcher, timeoutMs);
445
+ artificialAnalysis = await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs);
446
+ artificialAnalysisUpdated = true;
447
+ legAt.artificialAnalysis = stamp;
448
+ legs.push({ leg: "Artificial Analysis", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
426
449
  }
427
450
  catch (error) {
428
- const message = error instanceof Error ? error.message : String(error);
429
- warning = message.startsWith("LiveBench") ? message : `LiveBench refresh failed: ${message}`;
451
+ const detail = error instanceof Error ? error.message : String(error);
452
+ warnings.push(`Artificial Analysis refresh failed: ${detail}`);
453
+ legs.push({ leg: "Artificial Analysis", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
430
454
  }
431
- const catalog = {
432
- schemaVersion: 1,
433
- fetchedAt: now().toISOString(),
434
- modelsDev,
435
- ...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
436
- ...(liveBench !== undefined ? { liveBench } : {}),
437
- };
438
- writeCatalogCache(opts.repoRoot, catalog);
439
- return {
440
- updated: true,
441
- catalog: readCachedCatalog(opts.repoRoot, { now }),
442
- ...(warning !== undefined ? { warning } : {}),
443
- };
455
+ }
456
+ else {
457
+ delete legAt.artificialAnalysis;
458
+ legs.push({ leg: "Artificial Analysis", status: "skipped", detail: "no API key", retry: "retry when ARTIFICIAL_ANALYSIS_API_KEY is set" });
459
+ }
460
+ let liveBench = current.catalog.liveBench;
461
+ let liveBenchUpdated = false;
462
+ try {
463
+ liveBench = await fetchLiveBench(fetcher, timeoutMs);
464
+ liveBenchUpdated = true;
465
+ legAt.liveBench = stamp;
466
+ legs.push({ leg: "LiveBench", status: "updated", detail: `table ${LIVEBENCH_TABLE_DATE}`, retry: "next retry after this leg is stale" });
444
467
  }
445
468
  catch (error) {
446
- return {
447
- updated: false,
448
- catalog: current,
449
- warning: error instanceof Error ? error.message : String(error),
450
- };
469
+ const detail = error instanceof Error ? error.message : String(error);
470
+ warnings.push(`LiveBench refresh failed: ${detail}`);
471
+ legs.push({ leg: "LiveBench", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
451
472
  }
473
+ const updated = modelsDevUpdated || artificialAnalysisUpdated || liveBenchUpdated;
474
+ // models.dev is the cache spine: do not write a cache that would make the vendored fallback look
475
+ // fetched. Independent legs may only merge onto models.dev fetched now or already held in cache.
476
+ if (!updated || (!modelsDevUpdated && current.source !== "cache")) {
477
+ const reportedLegs = !modelsDevUpdated && current.source !== "cache"
478
+ ? legs.map((leg) => leg.status === "updated"
479
+ ? {
480
+ ...leg,
481
+ status: "failed",
482
+ detail: "fetched but discarded — models.dev spine unavailable",
483
+ retry: "retry next refresh",
484
+ }
485
+ : leg)
486
+ : legs;
487
+ return { updated: false, catalog: current, legs: reportedLegs, ...(warnings.length ? { warning: warnings.join("; ") } : {}) };
488
+ }
489
+ const catalog = {
490
+ schemaVersion: 1,
491
+ fetchedAt: legAt.modelsDev ?? current.catalog.fetchedAt,
492
+ modelsDev,
493
+ ...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
494
+ ...(liveBench !== undefined ? { liveBench } : {}),
495
+ legFetchedAt: legAt,
496
+ };
497
+ writeCatalogCache(opts.repoRoot, catalog);
498
+ return {
499
+ updated,
500
+ catalog: readCachedCatalog(opts.repoRoot, { now }),
501
+ legs,
502
+ ...(warnings.length ? { warning: warnings.join("; ") } : {}),
503
+ };
452
504
  }
@@ -9,6 +9,7 @@ import { cursorAgent } from "./cursor-agent.js";
9
9
  import { grok } from "./grok.js";
10
10
  import { kimi } from "./kimi.js";
11
11
  import { opencode } from "./opencode.js";
12
+ import { qwen, QWEN_VERSION_IDENTITY } from "./qwen.js";
12
13
  import { pi } from "./pi.js";
13
14
  import { TRUST_DIALOG_VARIANTS, TrustDialogSchema } from "./types.js";
14
15
  export const CLI_NAME_RE = /^[a-z0-9-]+$/;
@@ -105,10 +106,10 @@ function validateCliEntry(entry) {
105
106
  }
106
107
  // Package-owned, deterministic order. This is the sole shipped definition array: candidate-name
107
108
  // compatibility and advisory/routable projections below are all derived from it.
108
- const native = (adapter, binary) => ({
109
+ const native = (adapter, binary, identity = ".+") => ({
109
110
  id: adapter.id,
110
111
  binary,
111
- identity: ".+",
112
+ identity,
112
113
  vendor: adapter.vendor,
113
114
  drive: { adapter },
114
115
  });
@@ -122,7 +123,8 @@ export const CLI_CATALOG = [
122
123
  native(pi, "pi"),
123
124
  native(grok, "grok"),
124
125
  native(kimi, "kimi"),
125
- "gemini", "qwen", "aider", "goose", "amp", "droid", "auggie", "crush",
126
+ native(qwen, "qwen", QWEN_VERSION_IDENTITY.source),
127
+ "gemini", "aider", "goose", "amp", "droid", "auggie", "crush",
126
128
  {
127
129
  id: "omp",
128
130
  binary: "omp",
@@ -1,6 +1,6 @@
1
1
  import { type AuthHealth, type TrustDialog, type WorkerAdapter } from "./types.js";
2
2
  export declare const CLAUDE_ALIAS_IDENTITY_STAMPS: {
3
- readonly fable: "claude-fable-5";
3
+ readonly fable: "claude-fable-5-1";
4
4
  readonly opus: "claude-opus-4-8";
5
5
  readonly sonnet: "claude-sonnet-5";
6
6
  readonly haiku: "claude-haiku-4-5-20251001";
@@ -29,7 +29,8 @@ const MAX_SESSION_BYTES = 8_000_000; // per-file cap; a runaway JSONL cannot mak
29
29
  // claude-opus-5 channel to the repo overlay — it did not re-date this alias's stamps, so the
30
30
  // alias channel still carries 4-8-dated tier/pricing while serving 5. That warning is true.
31
31
  export const CLAUDE_ALIAS_IDENTITY_STAMPS = {
32
- fable: "claude-fable-5",
32
+ // OBS-871, 2026-09-03: Fable's floating alias now resolves to the 5.1 benchmark identity.
33
+ fable: "claude-fable-5-1",
33
34
  opus: "claude-opus-4-8",
34
35
  sonnet: "claude-sonnet-5",
35
36
  haiku: "claude-haiku-4-5-20251001",
@@ -209,7 +210,7 @@ export const claudeCode = {
209
210
  channels: (cfg) => channelsFromConfig("claude-code", cfg),
210
211
  // v1.65 T3: every flag the command builders below hardcode — doctor checks `claude --help` still
211
212
  // lists each (all present on claude 2.x, verified 2026-07-22). Advisory only, never routing.
212
- hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions"] },
213
+ hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions", "--settings"] },
213
214
  // --strict-mcp-config --mcp-config '{"mcpServers":{}}': pin the MCP surface to empty so fresh-worktree
214
215
  // workers/gates don't load project .mcp.json servers (herdr scrapes dialogs as idle — v1.4 incident,
215
216
  // memory tickmarkr-worker-mcp-dialog-stall). Live-verified 2026-07-10 on claude 2.1.205 (operator check):
@@ -219,7 +220,9 @@ export const claudeCode = {
219
220
  // Gotchas (both bit the 2026-07-10 live check): bare '{}' is REJECTED ("mcpServers: expected record"),
220
221
  // and --mcp-config is VARIADIC — a positional after it is eaten as a config-file path, so another
221
222
  // flag must always follow the value, never the prompt.
222
- headlessCommand: (promptFile, model) => `claude -p "$(cat ${shq(promptFile)})" --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text`,
223
+ // The empty -p argument selects print mode while stdin carries the prompt, keeping its nonce out
224
+ // of process argv. The redirect path is shell-quoted independently from the model.
225
+ headlessCommand: (promptFile, model) => `claude -p '' --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text < ${shq(promptFile)}`,
223
226
  // HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
224
227
  // Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
225
228
  // the daemon safely answers only the exact adapter-declared dialog once per slot.
@@ -236,12 +239,12 @@ export const claudeCode = {
236
239
  // live check ate the prompt), and --prompt-suggestions takes an OPTIONAL value — appended directly
237
240
  // before the prompt it would swallow it the same way. So the setting's value is always followed by
238
241
  // another flag, never by the prompt positional.
239
- interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
242
+ interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
240
243
  trustDialog: CLAUDE_TRUST_DIALOG,
241
244
  inputBox: CLAUDE_INPUT_BOX,
242
245
  // A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
243
246
  // and the same value-then-flag placement.
244
- resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
247
+ resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
245
248
  invoke(task, _cwd, a, ctx) {
246
249
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
247
250
  },
@@ -184,17 +184,16 @@ export const codex = {
184
184
  probeConcurrency: 1,
185
185
  probe: async () => probeVersion("codex"),
186
186
  channels: (cfg) => channelsFromConfig("codex", cfg),
187
- // v1.65 T3: every flag the command builders below hardcode (incl. codexMcpSuppressionFlags' -c/
187
+ // v1.65 T3: every flag the command builder below hardcodes (incl. codexMcpSuppressionFlags' -c/
188
188
  // --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
189
- hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-a", "-s", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
189
+ hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
190
190
  // --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
191
191
  // MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
192
192
  // CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
193
- headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
194
- // TUI uses expanded -a never -s workspace-write (exec-only flags do not apply)
195
- // (--help 2026-07-09: valid approval policies are untrusted|on-request|never; the previously
196
- // used `on-failure` is invalid and made codex exit 2 pre-inference)
197
- interactiveCommand: (promptFile, model) => `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
193
+ headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
194
+ // OBS-889: Codex's TUI has no file/stdin prompt form. Returning null makes the daemon journal
195
+ // worker-mode-fallback before it runs the argv-safe headless command in the visible pane.
196
+ interactiveCommand: () => null,
198
197
  invoke(task, _cwd, a, ctx) {
199
198
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
200
199
  },
@@ -4,6 +4,14 @@ import { type AuthHealth, type WorkerAdapter } from "./types.js";
4
4
  import { type CatalogModelEvidence, type CatalogReadResult } from "./catalog-remote.js";
5
5
  export declare const SEED_STAMPED = "2026-07-09";
6
6
  export declare const MODEL_STALE_DAYS = 30;
7
+ export type FleetUnclassifiedModel = {
8
+ adapter: string;
9
+ model: string;
10
+ detectedAt?: string;
11
+ variants?: string[];
12
+ /** The real CLI id written by classify when the display row is a collapsed base. */
13
+ classifyModel?: string;
14
+ };
7
15
  /**
8
16
  * Operator directive 2026-08-13 ("we should exclude all retired models"): the fleet models
9
17
  * screen hides these classes BY DEFAULT — omp alone reports 218 ids, most of them dated
@@ -84,9 +92,5 @@ export declare function suggestOverlay(cfg: TickmarkrConfig, health: Record<stri
84
92
  resolvedModel?: (adapter: string, model: string) => string | undefined;
85
93
  }): string;
86
94
  /** Unclassified models surfaced for fleet screen 2 (doctor matrix math, no tier fabrication). */
87
- export declare function fleetUnclassifiedModels(cfg: TickmarkrConfig, health: Record<string, AuthHealth>, adapters: WorkerAdapter[]): {
88
- adapter: string;
89
- model: string;
90
- detectedAt?: string;
91
- }[];
95
+ export declare function fleetUnclassifiedModels(cfg: TickmarkrConfig, health: Record<string, AuthHealth>, adapters: WorkerAdapter[]): FleetUnclassifiedModel[];
92
96
  export {};
@@ -11,12 +11,52 @@ export const SEED_STAMPED = "2026-07-09";
11
11
  // knowledge past this age gets a "rerun tickmarkr doctor" nudge (BLOCKED_POLL_MS-style named constant).
12
12
  export const MODEL_STALE_DAYS = 30;
13
13
  const DAY_MS = 86400000;
14
- // cursor-agent 2026.07.08 reports 193 mostly-parameterized ids (e.g. gpt-5.3-codex-high-fast); filter the `auto`
15
- // pseudo-model + effort/speed variant suffixes from the unconfigured-lint aggregation ONLY — doctor.json keeps the
16
- // raw list (verified 2026-07-10). Data stays raw; lints stay signal. -max/-none/-thinking joined the suffix set
17
- // 2026-08-12 (D-OBS-11: cursor's residual lint list was still mostly effort variants of configured bases).
18
- const LINT_VARIANT_RE = /^auto$|-(fast|minimal|none|low|medium|high|xhigh|max|thinking)$/;
14
+ // cursor-agent reports mostly effort/speed variants. Keep doctor.json raw, but collapse those ids
15
+ // into one base row; a base-less classify writes the highest-effort real id.
16
+ const VARIANT_SUFFIX_RE = /-(fast|minimal|none|low|medium|high|xhigh|max|thinking)$/;
17
+ const VARIANT_RANK = { max: 6, xhigh: 5, high: 4, medium: 3, low: 2, minimal: 1, none: 0, thinking: 0, fast: 0 };
19
18
  const LINT_CAP = 5;
19
+ const variantBase = (model) => {
20
+ let base = model;
21
+ while (VARIANT_SUFFIX_RE.test(base))
22
+ base = base.replace(VARIANT_SUFFIX_RE, "");
23
+ return base;
24
+ };
25
+ const variantRank = (model) => {
26
+ let rank = 0;
27
+ for (const token of model.split("-"))
28
+ rank = Math.max(rank, VARIANT_RANK[token] ?? 0);
29
+ return rank;
30
+ };
31
+ function collapseUnclassified(detected, configured) {
32
+ const groups = new Map();
33
+ for (const model of detected) {
34
+ if (model === "auto" || configured.has(model))
35
+ continue;
36
+ const base = variantBase(model);
37
+ if (configured.has(base))
38
+ continue;
39
+ const group = groups.get(base) ?? { bare: false, variants: [] };
40
+ if (model === base)
41
+ group.bare = true;
42
+ else
43
+ group.variants.push(model);
44
+ groups.set(base, group);
45
+ }
46
+ return [...groups].map(([model, group]) => {
47
+ if (!group.variants.length)
48
+ return { adapter: "", model };
49
+ const classifyModel = group.bare
50
+ ? model
51
+ : group.variants.reduce((best, candidate) => variantRank(candidate) > variantRank(best) ? candidate : best);
52
+ return {
53
+ adapter: "",
54
+ model,
55
+ variants: group.variants,
56
+ ...(classifyModel !== model ? { classifyModel } : {}),
57
+ };
58
+ });
59
+ }
20
60
  /**
21
61
  * Operator directive 2026-08-13 ("we should exclude all retired models"): the fleet models
22
62
  * screen hides these classes BY DEFAULT — omp alone reports 218 ids, most of them dated
@@ -477,9 +517,9 @@ export function modelLints(cfg, health, adapters, opts) {
477
517
  lints.push(`${id}: tiers lists ${model} — CLI no longer reports it; tombstone it (${model}: null overlay) or verify the id`);
478
518
  }
479
519
  }
480
- const extra = detected.filter((m) => !configured.includes(m) && !LINT_VARIANT_RE.test(m));
520
+ const extra = collapseUnclassified(detected, new Set(configured));
481
521
  if (extra.length) {
482
- const shown = extra.slice(0, cap).join(", ");
522
+ const shown = extra.slice(0, cap).map((row) => row.model).join(", ");
483
523
  const tail = extra.length > cap ? `, +${extra.length - cap} more${doctorRef}` : "";
484
524
  lints.push(`${id}: reports ${extra.length} model(s) not in tiers (${shown}${tail}) — classify before routing (benchmark policy)`);
485
525
  }
@@ -533,7 +573,7 @@ export function suggestOverlay(cfg, health, adapters, stateDir = DEFAULT_STATE_D
533
573
  // Tombstones: configured ids the CLI no longer reports. Ids are operator-authored (from cfg) → MODEL_ID_RE only.
534
574
  const tombstones = configured.filter((model) => !detected.includes(model) && MODEL_ID_RE.test(model));
535
575
  // Additions: detected ids not in cfg. WHOLE line commented, no tier (MODEL-06). Ids come from an external
536
- // CLI → MODEL_ID_RE (defense-in-depth, T-21-01) + the variant filter (cursor's ~193 parameterized ids).
576
+ // CLI → MODEL_ID_RE (defense-in-depth, T-21-01); effort variants collapse before this loop.
537
577
  // RELATIONAL gate (no capability judgment — "looks like an embedding model" is auto-tiering's cousin, the
538
578
  // NaN-routing class the v1.5 decision forbids): a detected id is suggested iff it shares a provider prefix
539
579
  // (clause a) OR a canonical segment (clause b, the RENAME case: opencode/glm-5.2 ⇒ zai-coding-plan/glm-5.2)
@@ -547,8 +587,9 @@ export function suggestOverlay(cfg, health, adapters, stateDir = DEFAULT_STATE_D
547
587
  const cfgCanon = new Set(configured.map(canonical));
548
588
  const additions = [];
549
589
  let omitted = 0;
550
- for (const model of detected) {
551
- if (configured.includes(model) || !MODEL_ID_RE.test(model) || LINT_VARIANT_RE.test(model))
590
+ for (const row of collapseUnclassified(detected, new Set(configured))) {
591
+ const model = row.classifyModel ?? row.model;
592
+ if (!MODEL_ID_RE.test(model))
552
593
  continue;
553
594
  if (configured.length > 0 && !cfgPrefixes.has(providerPrefix(model)) && !cfgCanon.has(canonical(model))) {
554
595
  omitted++;
@@ -619,11 +660,11 @@ export function fleetUnclassifiedModels(cfg, health, adapters) {
619
660
  continue;
620
661
  const configured = new Set(Object.keys(cfg.tiers[id]?.models ?? {}));
621
662
  const date = h?.modelsDetectedAt?.split("T")[0];
622
- for (const model of detected) {
623
- if (configured.has(model) || LINT_VARIANT_RE.test(model))
624
- continue;
625
- out.push({ adapter: id, model, detectedAt: date });
626
- }
663
+ out.push(...collapseUnclassified(detected, configured).map((row) => ({
664
+ ...row,
665
+ adapter: id,
666
+ ...(date ? { detectedAt: date } : {}),
667
+ })));
627
668
  }
628
669
  return out;
629
670
  }
@@ -28,6 +28,9 @@ const CURSOR_SOURCE = "https://cursor.com/help/ai-features/max-mode";
28
28
  const ZAI_SOURCE = "https://z.ai/blog/glm-5.2";
29
29
  const XAI_SOURCE = "https://docs.x.ai/developers/models/grok-4.5";
30
30
  const KIMI_SOURCE = "https://www.kimi.com/code/docs/en/kimi-code-cli/configuration/config-files.html";
31
+ const GOOGLE_SOURCE = "https://ai.google.dev/gemini-api/docs/models";
32
+ const QWEN_SOURCE = "https://qwenlm.github.io/";
33
+ const OBS_871_READ_DATE = "2026-09-03";
31
34
  const READ_DATE = "2026-08-05";
32
35
  const VENDORED_MODEL_WINDOW_CLAIMS = [
33
36
  { modelId: "fable", window: 1_000_000, source: ANTHROPIC_SOURCE, readDate: READ_DATE },
@@ -40,8 +43,16 @@ const VENDORED_MODEL_WINDOW_CLAIMS = [
40
43
  { modelId: "gpt-5.6-luna", window: 1_050_000, source: OPENAI_SOURCE, readDate: READ_DATE },
41
44
  { modelId: "composer-2.5", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
42
45
  { modelId: "composer-2.5-fast", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
46
+ { modelId: "claude-fable-5-1", window: 1_000_000, source: ANTHROPIC_SOURCE, readDate: OBS_871_READ_DATE },
47
+ { modelId: "gemini-3.8-flash", window: 1_000_000, source: GOOGLE_SOURCE, readDate: OBS_871_READ_DATE },
48
+ { modelId: "google/gemini-3.8-flash", window: 1_000_000, source: GOOGLE_SOURCE, readDate: OBS_871_READ_DATE },
43
49
  { modelId: "zai-coding-plan/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: READ_DATE },
44
50
  { modelId: "zai/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: READ_DATE },
51
+ { modelId: "zai/glm-5.3", window: 1_000_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
52
+ { modelId: "zai/glm-5.3-flash", window: 200_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
53
+ { modelId: "alibaba/qwen3.8-max", window: 1_000_000, source: QWEN_SOURCE, readDate: OBS_871_READ_DATE },
54
+ { modelId: "qwen3.8-max", window: 1_000_000, source: QWEN_SOURCE, readDate: OBS_871_READ_DATE },
55
+ { modelId: "prime-inference/z-ai/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
45
56
  { modelId: "grok-4.5", window: 500_000, source: XAI_SOURCE, readDate: READ_DATE },
46
57
  { modelId: "grok-composer-2.5-fast", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
47
58
  { modelId: "kimi-code/k3", window: 1_048_576, source: KIMI_SOURCE, readDate: READ_DATE },
@@ -23,6 +23,7 @@ ${task.files.length ? `\n## File scope — touch ONLY paths matching:\n${list(ta
23
23
  - Work only inside the current directory (your isolated worktree). Never push. Never switch branches.
24
24
  - Make small atomic git commits as you go (git add + git commit, conventional messages).
25
25
  - Touch ONLY paths matching the file scope. Out-of-scope edits FAIL the scope gate. The operator's allowlist is fixed when the run starts and nothing you do can change it while your work is judged; declaring a deviation never passes the gate either. If you cannot complete the task without an out-of-scope edit, stop and report ok:false explaining why in "summary". List any out-of-scope paths you did touch, each with a reason, in "deviations" (journaled for the operator's audit).
26
+ - No background process may outlive the worker, and no suite may run beside another.
26
27
  - Do not ask questions; you are unattended. Make the smallest correct change.
27
28
  ${feedback ? `\n## Previous attempt failed gates — fix these specifically\n${feedback}\n` : ""}
28
29
  When finished, end your final message with exactly one line (no code fence):
@@ -0,0 +1,5 @@
1
+ import { type ClassifiedWorkerResult } from "./prompt.js";
2
+ import { type WorkerAdapter } from "./types.js";
3
+ export declare const QWEN_VERSION_IDENTITY: RegExp;
4
+ export declare function parseQwenResult(raw: string, nonce: string): ClassifiedWorkerResult;
5
+ export declare const qwen: WorkerAdapter;