tickmarkr 2.2.1 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/README.md +10 -9
  2. package/dist/adapters/catalog-remote.d.ts +1 -4
  3. package/dist/adapters/catalog-remote.js +52 -42
  4. package/dist/adapters/catalog.js +5 -3
  5. package/dist/adapters/claude-code.d.ts +1 -1
  6. package/dist/adapters/claude-code.js +8 -5
  7. package/dist/adapters/model-lints.d.ts +9 -5
  8. package/dist/adapters/model-lints.js +56 -15
  9. package/dist/adapters/model-windows.js +11 -0
  10. package/dist/adapters/prompt.js +1 -0
  11. package/dist/adapters/qwen.d.ts +5 -0
  12. package/dist/adapters/qwen.js +153 -0
  13. package/dist/adapters/types.d.ts +21 -1
  14. package/dist/adapters/types.js +43 -2
  15. package/dist/cli/commands/approve.js +5 -4
  16. package/dist/cli/commands/beat.js +7 -4
  17. package/dist/cli/commands/compile.js +32 -6
  18. package/dist/cli/commands/doctor.d.ts +9 -4
  19. package/dist/cli/commands/doctor.js +87 -13
  20. package/dist/cli/commands/fleet.d.ts +4 -0
  21. package/dist/cli/commands/fleet.js +53 -14
  22. package/dist/cli/commands/init.js +36 -21
  23. package/dist/cli/commands/plan.js +45 -7
  24. package/dist/cli/commands/report.js +37 -1
  25. package/dist/cli/commands/status.d.ts +1 -0
  26. package/dist/cli/commands/status.js +45 -1
  27. package/dist/cli/commands/verify.d.ts +6 -0
  28. package/dist/cli/commands/verify.js +145 -25
  29. package/dist/cli/index.d.ts +1 -1
  30. package/dist/cli/index.js +2 -2
  31. package/dist/compile/collateral.d.ts +2 -9
  32. package/dist/compile/collateral.js +17 -18
  33. package/dist/compile/index.d.ts +4 -1
  34. package/dist/compile/index.js +41 -7
  35. package/dist/compile/native.d.ts +4 -2
  36. package/dist/compile/native.js +58 -9
  37. package/dist/compile/ownership.js +34 -9
  38. package/dist/config/config.d.ts +1 -0
  39. package/dist/config/config.js +52 -6
  40. package/dist/drivers/herdr.d.ts +1 -0
  41. package/dist/drivers/herdr.js +11 -1
  42. package/dist/drivers/index.d.ts +6 -0
  43. package/dist/drivers/index.js +19 -4
  44. package/dist/drivers/orca.d.ts +35 -1
  45. package/dist/drivers/orca.js +260 -20
  46. package/dist/drivers/subprocess.d.ts +3 -3
  47. package/dist/drivers/subprocess.js +16 -9
  48. package/dist/drivers/types.d.ts +12 -0
  49. package/dist/gates/baseline.d.ts +2 -0
  50. package/dist/gates/baseline.js +47 -11
  51. package/dist/gates/llm.d.ts +6 -0
  52. package/dist/gates/llm.js +25 -9
  53. package/dist/gates/review.d.ts +7 -3
  54. package/dist/gates/review.js +61 -22
  55. package/dist/gates/run-gates.d.ts +5 -2
  56. package/dist/gates/run-gates.js +50 -26
  57. package/dist/gates/verdict-cause.d.ts +6 -2
  58. package/dist/gates/verdict-cause.js +8 -4
  59. package/dist/route/preference.d.ts +4 -0
  60. package/dist/route/preference.js +40 -0
  61. package/dist/route/router.js +15 -2
  62. package/dist/run/consult.d.ts +1 -0
  63. package/dist/run/consult.js +39 -8
  64. package/dist/run/daemon.d.ts +16 -0
  65. package/dist/run/daemon.js +345 -74
  66. package/dist/run/git.d.ts +3 -0
  67. package/dist/run/git.js +40 -5
  68. package/dist/run/journal.d.ts +15 -2
  69. package/dist/run/journal.js +73 -12
  70. package/dist/run/supervision.d.ts +6 -0
  71. package/dist/run/supervision.js +29 -1
  72. package/dist/tui/ink/fleet-app.d.ts +4 -0
  73. package/dist/tui/ink/fleet-app.js +45 -16
  74. package/dist/tui/ink/init-app.js +4 -4
  75. package/package.json +59 -1
  76. package/skills/tickmarkr-overseer/SKILL.md +77 -18
  77. package/skills/tickmarkr-overseer/scripts/seat-send.sh +88 -18
  78. package/skills/tickmarkr-overseer/scripts/watch-artifacts.sh +36 -2
  79. package/skills/tickmarkr-overseer/scripts/watch-contamination.sh +36 -15
  80. package/skills/tickmarkr-overseer/scripts/watch-context.sh +33 -7
  81. package/skills/tickmarkr-overseer/scripts/watch-pending-input.sh +32 -8
package/README.md CHANGED
@@ -13,8 +13,9 @@ tickmarkr is a spec-driven orchestration harness for AI coding agent CLIs. You w
13
13
  acceptance criteria; the engine routes tasks to the best installed agent CLI (claude-code, codex,
14
14
  cursor-agent, opencode, grok, pi, kimi) by cost and capability, dispatches work in git worktrees for
15
15
  change isolation — as interactive TUIs when running under [herdr](https://herdr.dev), headless
16
- subprocesses otherwise, or in [Orca](https://onorca.dev) terminals when you name that driver
17
- yourself — and independently verifies each committed result by checking for no new
16
+ subprocesses otherwise, or in [Orca](https://onorca.dev) terminals when auto detects both Orca
17
+ markers (name that driver explicitly outside one) — and independently verifies each committed
18
+ result by checking for no new
18
19
  baseline failures per task, then strictly verifying the integration tip. Green tasks consolidate onto a
19
20
  `tickmarkr/<runId>` branch; merging to your mainline is always your call, never automated. Engage
20
21
  with full visibility into routing decisions, worker progress, and gate verdicts — or run headless
@@ -251,14 +252,14 @@ and first-attempt success rate. Cost reporting follows strict honesty rules and
251
252
  When running under [herdr](https://herdr.dev), tickmarkr creates a labeled pane-and-tab workspace
252
253
  for real-time visibility (optional — omit `--driver herdr` or run headless if preferred).
253
254
 
254
- ### Orca: an explicit-selection execution surface
255
+ ### Orca: a detected-or-named execution surface
255
256
 
256
- [Orca](https://onorca.dev) is the third execution surface, and the only one you must ask for by
257
- name: `--driver orca` or `driver: orca` in config. `--driver auto` never selects it — auto picks
258
- herdr when a herdr session is live and subprocess otherwise — so Orca is never inherited from an
259
- ambient environment variable, and an Orca that is installed but unreachable is not silently
260
- downgraded to a hidden subprocess worker either. Naming it is the whole gate; its runtime failures
261
- stay Orca's, reported as failures.
257
+ [Orca](https://onorca.dev) is the third execution surface. `auto` resolves herdr first when
258
+ `HERDR_ENV=1`, then Orca only when both Orca-authored markers `TERM_PROGRAM=Orca` and
259
+ `ORCA_TERMINAL_HANDLE` are present, then subprocess. This environment-only choice executes no
260
+ binary or runtime probe. Outside an Orca terminal, name it explicitly with `--driver orca` or
261
+ `driver: orca` in config. Once selected either way, an unreachable Orca stays a loud Orca driver
262
+ failure and is never silently replaced by a hidden subprocess worker.
262
263
 
263
264
  What Orca supplies is terminals. What tickmarkr keeps is everything that decides whether work
264
265
  ships: **it creates and owns the git worktree** for every task (Orca is told which checkout to bind
@@ -66,9 +66,6 @@ export declare function resolveCatalogModel(catalog: CatalogCache, query: {
66
66
  model: string;
67
67
  resolvedModel?: string;
68
68
  }): CatalogModelEvidence | undefined;
69
- /**
70
- * The named, explicit refresh path. No other function in this module can reach fetch.
71
- * A failed refresh preserves the previous cache byte-for-byte and returns it fail-open.
72
- */
69
+ /** Shared refresh path for the explicit command and the seven-day doctor/fleet guard. */
73
70
  export declare function refreshCatalogCommand(opts: RefreshCatalogOptions): Promise<RefreshCatalogResult>;
74
71
  export {};
@@ -9,20 +9,21 @@ export const ARTIFICIAL_ANALYSIS_CATALOG_URL = "https://artificialanalysis.ai/ap
9
9
  export const LIVEBENCH_TABLE_DATE = "2026_06_25";
10
10
  export const LIVEBENCH_TABLE_URL = `https://livebench.ai/table_${LIVEBENCH_TABLE_DATE}.csv`;
11
11
  export const LIVEBENCH_CATEGORIES_URL = `https://livebench.ai/categories_${LIVEBENCH_TABLE_DATE}.json`;
12
- export const CATALOG_CACHE_MAX_AGE_MS = 30 * 86_400_000;
12
+ export const CATALOG_CACHE_MAX_AGE_MS = 7 * 86_400_000;
13
13
  export const CATALOG_REFRESH_TIMEOUT_MS = 10_000;
14
14
  const ARTIFICIAL_ANALYSIS_PAGE_SIZE = 100;
15
15
  const ARTIFICIAL_ANALYSIS_MAX_PAGES = 100;
16
16
  const VENDORED_CATALOG = {
17
17
  schemaVersion: 1,
18
18
  // Package fallback copied from https://models.dev/api.json on 2026-08-05. It is deliberately
19
- // small: an explicit refresh owns broad/current coverage, while doctor remains cache-only.
19
+ // small: the explicit or seven-day operator-surface refresh owns broad/current coverage.
20
20
  fetchedAt: "2026-08-05T20:00:05.000Z",
21
21
  modelsDev: {
22
22
  anthropic: {
23
23
  id: "anthropic",
24
24
  models: {
25
- "claude-fable-5": { id: "claude-fable-5", cost: { input: 10, output: 50 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
25
+ // SHIP (OBS-871, 2026-09-03): users resolving the current Fable alias need the 5.1 identity even when refresh is unavailable.
26
+ "claude-fable-5-1": { id: "claude-fable-5-1", cost: { input: 10, output: 50 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
26
27
  "claude-opus-4-8": { id: "claude-opus-4-8", cost: { input: 5, output: 25 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
27
28
  "claude-sonnet-5": { id: "claude-sonnet-5", cost: { input: 2, output: 10 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
28
29
  "claude-haiku-4-5-20251001": { id: "claude-haiku-4-5-20251001", cost: { input: 1, output: 5 }, limit: { context: 200_000, output: 64_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
@@ -226,6 +227,7 @@ function assertUsableLiveBench(rows, categories) {
226
227
  // `-thinking-auto-medium-effort`); the highest effort is the model at its best. Rank by
227
228
  // hyphen-delimited token so `xhigh` never reads as `high`.
228
229
  const LIVEBENCH_EFFORT_RANK = { max: 5, xhigh: 4, high: 3, medium: 2, low: 1 };
230
+ const LIVEBENCH_VARIANT_RE = /^-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast)(?:-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast))*$/;
229
231
  const liveBenchEffortRank = (residue) => residue.split("-").reduce((rank, token) => Math.max(rank, LIVEBENCH_EFFORT_RANK[token] ?? 0), 0);
230
232
  const liveBenchCategoryMean = (row, tasks) => {
231
233
  const scores = (Array.isArray(tasks) ? tasks : [])
@@ -233,8 +235,8 @@ const liveBenchCategoryMean = (row, tasks) => {
233
235
  .filter((score) => score !== undefined);
234
236
  return scores.length > 0 ? scores.reduce((sum, score) => sum + score, 0) / scores.length : undefined;
235
237
  };
236
- /** LiveBench hyphenates where models.dev spaces: `Kimi K3` is `kimi-k3` in the table. */
237
- const liveBenchIdentity = (value) => value.trim().toLowerCase().replace(/\s+/g, "-");
238
+ /** LiveBench treats spaces, dots, and punctuation as the same word boundary. */
239
+ const liveBenchIdentity = (value) => value.trim().toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
238
240
  function liveBenchIndex(value, identities) {
239
241
  const root = record(value);
240
242
  if (!root || !Array.isArray(root.rows))
@@ -249,7 +251,7 @@ function liveBenchIndex(value, identities) {
249
251
  // Bare startsWith is a false-positive machine: `glm-5` would claim `glm-5.2`. The residue after
250
252
  // the fleet id must be empty or an effort suffix.
251
253
  const residue = wanted.map((identity) => model.startsWith(identity) ? model.slice(identity.length) : undefined)
252
- .find((rest) => rest === "" || rest?.startsWith("-"));
254
+ .find((rest) => rest === "" || LIVEBENCH_VARIANT_RE.test(rest ?? ""));
253
255
  if (residue === undefined)
254
256
  continue;
255
257
  const rank = liveBenchEffortRank(residue);
@@ -285,6 +287,7 @@ export function resolveCatalogModel(catalog, query) {
285
287
  const identities = [
286
288
  modelId,
287
289
  query.model,
290
+ ...[modelId, query.model].flatMap((identity) => identity.includes("/") ? [identity.slice(identity.lastIndexOf("/") + 1)] : []),
288
291
  ...(typeof model.name === "string" ? [model.name] : []),
289
292
  ];
290
293
  const intelligence = artificialAnalysisIndex(catalog.artificialAnalysis, [
@@ -400,53 +403,60 @@ async function fetchLiveBench(fetcher, timeoutMs) {
400
403
  assertUsableLiveBench(rows, categories);
401
404
  return { tableDate: LIVEBENCH_TABLE_DATE, categories, rows };
402
405
  }
403
- /**
404
- * The named, explicit refresh path. No other function in this module can reach fetch.
405
- * A failed refresh preserves the previous cache byte-for-byte and returns it fail-open.
406
- */
406
+ /** Shared refresh path for the explicit command and the seven-day doctor/fleet guard. */
407
407
  export async function refreshCatalogCommand(opts) {
408
408
  const now = opts.now ?? (() => new Date());
409
409
  const current = readCachedCatalog(opts.repoRoot, { now });
410
+ const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
411
+ const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
412
+ const warnings = [];
413
+ let modelsDev = current.catalog.modelsDev;
414
+ let modelsDevUpdated = false;
410
415
  try {
411
- const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
412
- const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
413
- const modelsDev = await fetchCatalog(fetcher, MODELS_DEV_CATALOG_URL, {}, timeoutMs, (r) => r.json());
416
+ modelsDev = await fetchCatalog(fetcher, MODELS_DEV_CATALOG_URL, {}, timeoutMs, (r) => r.json());
414
417
  if (!validModelsDevCatalog(modelsDev))
415
418
  throw new Error("models.dev catalog schema is invalid");
416
- const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
417
- const artificialAnalysis = apiKey
418
- ? await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs)
419
- : undefined;
420
- // The LiveBench leg is keyless and never costs the models.dev refresh: a failure keeps the
421
- // previous section verbatim and names the leg in the warning.
422
- let liveBench = current.catalog.liveBench;
423
- let warning;
419
+ modelsDevUpdated = true;
420
+ }
421
+ catch (error) {
422
+ warnings.push(`models.dev refresh failed: ${error instanceof Error ? error.message : String(error)}`);
423
+ }
424
+ let artificialAnalysis = current.catalog.artificialAnalysis;
425
+ const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
426
+ if (apiKey) {
424
427
  try {
425
- liveBench = await fetchLiveBench(fetcher, timeoutMs);
428
+ artificialAnalysis = await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs);
426
429
  }
427
430
  catch (error) {
428
- const message = error instanceof Error ? error.message : String(error);
429
- warning = message.startsWith("LiveBench") ? message : `LiveBench refresh failed: ${message}`;
431
+ warnings.push(`Artificial Analysis refresh failed: ${error instanceof Error ? error.message : String(error)}`);
430
432
  }
431
- const catalog = {
432
- schemaVersion: 1,
433
- fetchedAt: now().toISOString(),
434
- modelsDev,
435
- ...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
436
- ...(liveBench !== undefined ? { liveBench } : {}),
437
- };
438
- writeCatalogCache(opts.repoRoot, catalog);
439
- return {
440
- updated: true,
441
- catalog: readCachedCatalog(opts.repoRoot, { now }),
442
- ...(warning !== undefined ? { warning } : {}),
443
- };
433
+ }
434
+ let liveBench = current.catalog.liveBench;
435
+ let liveBenchUpdated = false;
436
+ try {
437
+ liveBench = await fetchLiveBench(fetcher, timeoutMs);
438
+ liveBenchUpdated = true;
444
439
  }
445
440
  catch (error) {
446
- return {
447
- updated: false,
448
- catalog: current,
449
- warning: error instanceof Error ? error.message : String(error),
450
- };
441
+ warnings.push(`LiveBench refresh failed: ${error instanceof Error ? error.message : String(error)}`);
451
442
  }
443
+ // models.dev is the required cache spine. Still run every independent leg after it fails, but do
444
+ // not write a cache that would make the vendored fallback look fetched.
445
+ if (!modelsDevUpdated) {
446
+ return { updated: false, catalog: current, warning: warnings.join("; ") };
447
+ }
448
+ const catalog = {
449
+ schemaVersion: 1,
450
+ // A failed keyless leg keeps the old age so doctor/fleet retry it rather than masking it for 7d.
451
+ fetchedAt: liveBenchUpdated ? now().toISOString() : current.catalog.fetchedAt,
452
+ modelsDev,
453
+ ...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
454
+ ...(liveBench !== undefined ? { liveBench } : {}),
455
+ };
456
+ writeCatalogCache(opts.repoRoot, catalog);
457
+ return {
458
+ updated: true,
459
+ catalog: readCachedCatalog(opts.repoRoot, { now }),
460
+ ...(warnings.length ? { warning: warnings.join("; ") } : {}),
461
+ };
452
462
  }
@@ -9,6 +9,7 @@ import { cursorAgent } from "./cursor-agent.js";
9
9
  import { grok } from "./grok.js";
10
10
  import { kimi } from "./kimi.js";
11
11
  import { opencode } from "./opencode.js";
12
+ import { qwen, QWEN_VERSION_IDENTITY } from "./qwen.js";
12
13
  import { pi } from "./pi.js";
13
14
  import { TRUST_DIALOG_VARIANTS, TrustDialogSchema } from "./types.js";
14
15
  export const CLI_NAME_RE = /^[a-z0-9-]+$/;
@@ -105,10 +106,10 @@ function validateCliEntry(entry) {
105
106
  }
106
107
  // Package-owned, deterministic order. This is the sole shipped definition array: candidate-name
107
108
  // compatibility and advisory/routable projections below are all derived from it.
108
- const native = (adapter, binary) => ({
109
+ const native = (adapter, binary, identity = ".+") => ({
109
110
  id: adapter.id,
110
111
  binary,
111
- identity: ".+",
112
+ identity,
112
113
  vendor: adapter.vendor,
113
114
  drive: { adapter },
114
115
  });
@@ -122,7 +123,8 @@ export const CLI_CATALOG = [
122
123
  native(pi, "pi"),
123
124
  native(grok, "grok"),
124
125
  native(kimi, "kimi"),
125
- "gemini", "qwen", "aider", "goose", "amp", "droid", "auggie", "crush",
126
+ native(qwen, "qwen", QWEN_VERSION_IDENTITY.source),
127
+ "gemini", "aider", "goose", "amp", "droid", "auggie", "crush",
126
128
  {
127
129
  id: "omp",
128
130
  binary: "omp",
@@ -1,6 +1,6 @@
1
1
  import { type AuthHealth, type TrustDialog, type WorkerAdapter } from "./types.js";
2
2
  export declare const CLAUDE_ALIAS_IDENTITY_STAMPS: {
3
- readonly fable: "claude-fable-5";
3
+ readonly fable: "claude-fable-5-1";
4
4
  readonly opus: "claude-opus-4-8";
5
5
  readonly sonnet: "claude-sonnet-5";
6
6
  readonly haiku: "claude-haiku-4-5-20251001";
@@ -29,7 +29,8 @@ const MAX_SESSION_BYTES = 8_000_000; // per-file cap; a runaway JSONL cannot mak
29
29
  // claude-opus-5 channel to the repo overlay — it did not re-date this alias's stamps, so the
30
30
  // alias channel still carries 4-8-dated tier/pricing while serving 5. That warning is true.
31
31
  export const CLAUDE_ALIAS_IDENTITY_STAMPS = {
32
- fable: "claude-fable-5",
32
+ // OBS-871, 2026-09-03: Fable's floating alias now resolves to the 5.1 benchmark identity.
33
+ fable: "claude-fable-5-1",
33
34
  opus: "claude-opus-4-8",
34
35
  sonnet: "claude-sonnet-5",
35
36
  haiku: "claude-haiku-4-5-20251001",
@@ -209,7 +210,7 @@ export const claudeCode = {
209
210
  channels: (cfg) => channelsFromConfig("claude-code", cfg),
210
211
  // v1.65 T3: every flag the command builders below hardcode — doctor checks `claude --help` still
211
212
  // lists each (all present on claude 2.x, verified 2026-07-22). Advisory only, never routing.
212
- hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions"] },
213
+ hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions", "--settings"] },
213
214
  // --strict-mcp-config --mcp-config '{"mcpServers":{}}': pin the MCP surface to empty so fresh-worktree
214
215
  // workers/gates don't load project .mcp.json servers (herdr scrapes dialogs as idle — v1.4 incident,
215
216
  // memory tickmarkr-worker-mcp-dialog-stall). Live-verified 2026-07-10 on claude 2.1.205 (operator check):
@@ -219,7 +220,9 @@ export const claudeCode = {
219
220
  // Gotchas (both bit the 2026-07-10 live check): bare '{}' is REJECTED ("mcpServers: expected record"),
220
221
  // and --mcp-config is VARIADIC — a positional after it is eaten as a config-file path, so another
221
222
  // flag must always follow the value, never the prompt.
222
- headlessCommand: (promptFile, model) => `claude -p "$(cat ${shq(promptFile)})" --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text`,
223
+ // The empty -p argument selects print mode while stdin carries the prompt, keeping its nonce out
224
+ // of process argv. The redirect path is shell-quoted independently from the model.
225
+ headlessCommand: (promptFile, model) => `claude -p '' --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text < ${shq(promptFile)}`,
223
226
  // HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
224
227
  // Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
225
228
  // the daemon safely answers only the exact adapter-declared dialog once per slot.
@@ -236,12 +239,12 @@ export const claudeCode = {
236
239
  // live check ate the prompt), and --prompt-suggestions takes an OPTIONAL value — appended directly
237
240
  // before the prompt it would swallow it the same way. So the setting's value is always followed by
238
241
  // another flag, never by the prompt positional.
239
- interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
242
+ interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
240
243
  trustDialog: CLAUDE_TRUST_DIALOG,
241
244
  inputBox: CLAUDE_INPUT_BOX,
242
245
  // A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
243
246
  // and the same value-then-flag placement.
244
- resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
247
+ resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
245
248
  invoke(task, _cwd, a, ctx) {
246
249
  return { command: this.headlessCommand(ctx.promptFile, a.model) };
247
250
  },
@@ -4,6 +4,14 @@ import { type AuthHealth, type WorkerAdapter } from "./types.js";
4
4
  import { type CatalogModelEvidence, type CatalogReadResult } from "./catalog-remote.js";
5
5
  export declare const SEED_STAMPED = "2026-07-09";
6
6
  export declare const MODEL_STALE_DAYS = 30;
7
+ export type FleetUnclassifiedModel = {
8
+ adapter: string;
9
+ model: string;
10
+ detectedAt?: string;
11
+ variants?: string[];
12
+ /** The real CLI id written by classify when the display row is a collapsed base. */
13
+ classifyModel?: string;
14
+ };
7
15
  /**
8
16
  * Operator directive 2026-08-13 ("we should exclude all retired models"): the fleet models
9
17
  * screen hides these classes BY DEFAULT — omp alone reports 218 ids, most of them dated
@@ -84,9 +92,5 @@ export declare function suggestOverlay(cfg: TickmarkrConfig, health: Record<stri
84
92
  resolvedModel?: (adapter: string, model: string) => string | undefined;
85
93
  }): string;
86
94
  /** Unclassified models surfaced for fleet screen 2 (doctor matrix math, no tier fabrication). */
87
- export declare function fleetUnclassifiedModels(cfg: TickmarkrConfig, health: Record<string, AuthHealth>, adapters: WorkerAdapter[]): {
88
- adapter: string;
89
- model: string;
90
- detectedAt?: string;
91
- }[];
95
+ export declare function fleetUnclassifiedModels(cfg: TickmarkrConfig, health: Record<string, AuthHealth>, adapters: WorkerAdapter[]): FleetUnclassifiedModel[];
92
96
  export {};
@@ -11,12 +11,52 @@ export const SEED_STAMPED = "2026-07-09";
11
11
  // knowledge past this age gets a "rerun tickmarkr doctor" nudge (BLOCKED_POLL_MS-style named constant).
12
12
  export const MODEL_STALE_DAYS = 30;
13
13
  const DAY_MS = 86400000;
14
- // cursor-agent 2026.07.08 reports 193 mostly-parameterized ids (e.g. gpt-5.3-codex-high-fast); filter the `auto`
15
- // pseudo-model + effort/speed variant suffixes from the unconfigured-lint aggregation ONLY — doctor.json keeps the
16
- // raw list (verified 2026-07-10). Data stays raw; lints stay signal. -max/-none/-thinking joined the suffix set
17
- // 2026-08-12 (D-OBS-11: cursor's residual lint list was still mostly effort variants of configured bases).
18
- const LINT_VARIANT_RE = /^auto$|-(fast|minimal|none|low|medium|high|xhigh|max|thinking)$/;
14
+ // cursor-agent reports mostly effort/speed variants. Keep doctor.json raw, but collapse those ids
15
+ // into one base row; a base-less classify writes the highest-effort real id.
16
+ const VARIANT_SUFFIX_RE = /-(fast|minimal|none|low|medium|high|xhigh|max|thinking)$/;
17
+ const VARIANT_RANK = { max: 6, xhigh: 5, high: 4, medium: 3, low: 2, minimal: 1, none: 0, thinking: 0, fast: 0 };
19
18
  const LINT_CAP = 5;
19
+ const variantBase = (model) => {
20
+ let base = model;
21
+ while (VARIANT_SUFFIX_RE.test(base))
22
+ base = base.replace(VARIANT_SUFFIX_RE, "");
23
+ return base;
24
+ };
25
+ const variantRank = (model) => {
26
+ let rank = 0;
27
+ for (const token of model.split("-"))
28
+ rank = Math.max(rank, VARIANT_RANK[token] ?? 0);
29
+ return rank;
30
+ };
31
+ function collapseUnclassified(detected, configured) {
32
+ const groups = new Map();
33
+ for (const model of detected) {
34
+ if (model === "auto" || configured.has(model))
35
+ continue;
36
+ const base = variantBase(model);
37
+ if (configured.has(base))
38
+ continue;
39
+ const group = groups.get(base) ?? { bare: false, variants: [] };
40
+ if (model === base)
41
+ group.bare = true;
42
+ else
43
+ group.variants.push(model);
44
+ groups.set(base, group);
45
+ }
46
+ return [...groups].map(([model, group]) => {
47
+ if (!group.variants.length)
48
+ return { adapter: "", model };
49
+ const classifyModel = group.bare
50
+ ? model
51
+ : group.variants.reduce((best, candidate) => variantRank(candidate) > variantRank(best) ? candidate : best);
52
+ return {
53
+ adapter: "",
54
+ model,
55
+ variants: group.variants,
56
+ ...(classifyModel !== model ? { classifyModel } : {}),
57
+ };
58
+ });
59
+ }
20
60
  /**
21
61
  * Operator directive 2026-08-13 ("we should exclude all retired models"): the fleet models
22
62
  * screen hides these classes BY DEFAULT — omp alone reports 218 ids, most of them dated
@@ -477,9 +517,9 @@ export function modelLints(cfg, health, adapters, opts) {
477
517
  lints.push(`${id}: tiers lists ${model} — CLI no longer reports it; tombstone it (${model}: null overlay) or verify the id`);
478
518
  }
479
519
  }
480
- const extra = detected.filter((m) => !configured.includes(m) && !LINT_VARIANT_RE.test(m));
520
+ const extra = collapseUnclassified(detected, new Set(configured));
481
521
  if (extra.length) {
482
- const shown = extra.slice(0, cap).join(", ");
522
+ const shown = extra.slice(0, cap).map((row) => row.model).join(", ");
483
523
  const tail = extra.length > cap ? `, +${extra.length - cap} more${doctorRef}` : "";
484
524
  lints.push(`${id}: reports ${extra.length} model(s) not in tiers (${shown}${tail}) — classify before routing (benchmark policy)`);
485
525
  }
@@ -533,7 +573,7 @@ export function suggestOverlay(cfg, health, adapters, stateDir = DEFAULT_STATE_D
533
573
  // Tombstones: configured ids the CLI no longer reports. Ids are operator-authored (from cfg) → MODEL_ID_RE only.
534
574
  const tombstones = configured.filter((model) => !detected.includes(model) && MODEL_ID_RE.test(model));
535
575
  // Additions: detected ids not in cfg. WHOLE line commented, no tier (MODEL-06). Ids come from an external
536
- // CLI → MODEL_ID_RE (defense-in-depth, T-21-01) + the variant filter (cursor's ~193 parameterized ids).
576
+ // CLI → MODEL_ID_RE (defense-in-depth, T-21-01); effort variants collapse before this loop.
537
577
  // RELATIONAL gate (no capability judgment — "looks like an embedding model" is auto-tiering's cousin, the
538
578
  // NaN-routing class the v1.5 decision forbids): a detected id is suggested iff it shares a provider prefix
539
579
  // (clause a) OR a canonical segment (clause b, the RENAME case: opencode/glm-5.2 ⇒ zai-coding-plan/glm-5.2)
@@ -547,8 +587,9 @@ export function suggestOverlay(cfg, health, adapters, stateDir = DEFAULT_STATE_D
547
587
  const cfgCanon = new Set(configured.map(canonical));
548
588
  const additions = [];
549
589
  let omitted = 0;
550
- for (const model of detected) {
551
- if (configured.includes(model) || !MODEL_ID_RE.test(model) || LINT_VARIANT_RE.test(model))
590
+ for (const row of collapseUnclassified(detected, new Set(configured))) {
591
+ const model = row.classifyModel ?? row.model;
592
+ if (!MODEL_ID_RE.test(model))
552
593
  continue;
553
594
  if (configured.length > 0 && !cfgPrefixes.has(providerPrefix(model)) && !cfgCanon.has(canonical(model))) {
554
595
  omitted++;
@@ -619,11 +660,11 @@ export function fleetUnclassifiedModels(cfg, health, adapters) {
619
660
  continue;
620
661
  const configured = new Set(Object.keys(cfg.tiers[id]?.models ?? {}));
621
662
  const date = h?.modelsDetectedAt?.split("T")[0];
622
- for (const model of detected) {
623
- if (configured.has(model) || LINT_VARIANT_RE.test(model))
624
- continue;
625
- out.push({ adapter: id, model, detectedAt: date });
626
- }
663
+ out.push(...collapseUnclassified(detected, configured).map((row) => ({
664
+ ...row,
665
+ adapter: id,
666
+ ...(date ? { detectedAt: date } : {}),
667
+ })));
627
668
  }
628
669
  return out;
629
670
  }
@@ -28,6 +28,9 @@ const CURSOR_SOURCE = "https://cursor.com/help/ai-features/max-mode";
28
28
  const ZAI_SOURCE = "https://z.ai/blog/glm-5.2";
29
29
  const XAI_SOURCE = "https://docs.x.ai/developers/models/grok-4.5";
30
30
  const KIMI_SOURCE = "https://www.kimi.com/code/docs/en/kimi-code-cli/configuration/config-files.html";
31
+ const GOOGLE_SOURCE = "https://ai.google.dev/gemini-api/docs/models";
32
+ const QWEN_SOURCE = "https://qwenlm.github.io/";
33
+ const OBS_871_READ_DATE = "2026-09-03";
31
34
  const READ_DATE = "2026-08-05";
32
35
  const VENDORED_MODEL_WINDOW_CLAIMS = [
33
36
  { modelId: "fable", window: 1_000_000, source: ANTHROPIC_SOURCE, readDate: READ_DATE },
@@ -40,8 +43,16 @@ const VENDORED_MODEL_WINDOW_CLAIMS = [
40
43
  { modelId: "gpt-5.6-luna", window: 1_050_000, source: OPENAI_SOURCE, readDate: READ_DATE },
41
44
  { modelId: "composer-2.5", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
42
45
  { modelId: "composer-2.5-fast", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
46
+ { modelId: "claude-fable-5-1", window: 1_000_000, source: ANTHROPIC_SOURCE, readDate: OBS_871_READ_DATE },
47
+ { modelId: "gemini-3.8-flash", window: 1_000_000, source: GOOGLE_SOURCE, readDate: OBS_871_READ_DATE },
48
+ { modelId: "google/gemini-3.8-flash", window: 1_000_000, source: GOOGLE_SOURCE, readDate: OBS_871_READ_DATE },
43
49
  { modelId: "zai-coding-plan/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: READ_DATE },
44
50
  { modelId: "zai/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: READ_DATE },
51
+ { modelId: "zai/glm-5.3", window: 1_000_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
52
+ { modelId: "zai/glm-5.3-flash", window: 200_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
53
+ { modelId: "alibaba/qwen3.8-max", window: 1_000_000, source: QWEN_SOURCE, readDate: OBS_871_READ_DATE },
54
+ { modelId: "qwen3.8-max", window: 1_000_000, source: QWEN_SOURCE, readDate: OBS_871_READ_DATE },
55
+ { modelId: "prime-inference/z-ai/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
45
56
  { modelId: "grok-4.5", window: 500_000, source: XAI_SOURCE, readDate: READ_DATE },
46
57
  { modelId: "grok-composer-2.5-fast", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
47
58
  { modelId: "kimi-code/k3", window: 1_048_576, source: KIMI_SOURCE, readDate: READ_DATE },
@@ -23,6 +23,7 @@ ${task.files.length ? `\n## File scope — touch ONLY paths matching:\n${list(ta
23
23
  - Work only inside the current directory (your isolated worktree). Never push. Never switch branches.
24
24
  - Make small atomic git commits as you go (git add + git commit, conventional messages).
25
25
  - Touch ONLY paths matching the file scope. Out-of-scope edits FAIL the scope gate. The operator's allowlist is fixed when the run starts and nothing you do can change it while your work is judged; declaring a deviation never passes the gate either. If you cannot complete the task without an out-of-scope edit, stop and report ok:false explaining why in "summary". List any out-of-scope paths you did touch, each with a reason, in "deviations" (journaled for the operator's audit).
26
+ - No background process may outlive the worker, and no suite may run beside another.
26
27
  - Do not ask questions; you are unattended. Make the smallest correct change.
27
28
  ${feedback ? `\n## Previous attempt failed gates — fix these specifically\n${feedback}\n` : ""}
28
29
  When finished, end your final message with exactly one line (no code fence):
@@ -0,0 +1,5 @@
1
+ import { type ClassifiedWorkerResult } from "./prompt.js";
2
+ import { type WorkerAdapter } from "./types.js";
3
+ export declare const QWEN_VERSION_IDENTITY: RegExp;
4
+ export declare function parseQwenResult(raw: string, nonce: string): ClassifiedWorkerResult;
5
+ export declare const qwen: WorkerAdapter;
@@ -0,0 +1,153 @@
1
+ import { spawnSync } from "node:child_process";
2
+ import { parseWorkerResult } from "./prompt.js";
3
+ import { channelsFromConfig, shq, } from "./types.js";
4
+ export const QWEN_VERSION_IDENTITY = /^\d+\.\d+\.\d+/;
5
+ const QWEN_SKIP_UPDATE = "QWEN_CODE_SKIP_UPDATE_CHECK_ONCE=true";
6
+ function decodeQwenEvents(events) {
7
+ const text = [];
8
+ let failed = false;
9
+ let resultText;
10
+ let errorText;
11
+ let denialText;
12
+ for (const event of events) {
13
+ if (!event || typeof event !== "object")
14
+ continue;
15
+ if ("type" in event && event.type === "assistant" && "message" in event) {
16
+ const message = event.message;
17
+ if (message && typeof message === "object" && "content" in message && Array.isArray(message.content)) {
18
+ for (const content of message.content) {
19
+ if (!content || typeof content !== "object" || !(("type" in content) && content.type === "text"))
20
+ continue;
21
+ if ("text" in content && typeof content.text === "string")
22
+ text.push(content.text);
23
+ }
24
+ }
25
+ }
26
+ if ("is_error" in event && event.is_error === true)
27
+ failed = true;
28
+ if ("type" in event && event.type === "result") {
29
+ if (!(("subtype" in event) && event.subtype === "success"))
30
+ failed = true;
31
+ if ("result" in event && typeof event.result === "string" && event.result.trim()) {
32
+ resultText ??= event.result.trim();
33
+ }
34
+ // A no-auth host answers `result/error_during_execution` with the cause under `error.message`
35
+ // (verbatim: .planning/assessments/2026-09-04-qwen-live-worker-form/no-auth-home.stdout).
36
+ const error = "error" in event ? event.error : undefined;
37
+ if (error && typeof error === "object" && "message" in error && typeof error.message === "string" && error.message.trim()) {
38
+ errorText ??= error.message.trim();
39
+ }
40
+ }
41
+ if ("permission_denials" in event && Array.isArray(event.permission_denials) && event.permission_denials.length > 0) {
42
+ failed = true;
43
+ denialText ??= event.permission_denials.map(String).join(", ");
44
+ }
45
+ if (!("stats" in event) || !event.stats || typeof event.stats !== "object" || !("models" in event.stats))
46
+ continue;
47
+ const models = event.stats.models;
48
+ if (!models || typeof models !== "object")
49
+ continue;
50
+ for (const model of Object.values(models)) {
51
+ if (!model || typeof model !== "object" || !("api" in model))
52
+ continue;
53
+ const api = model.api;
54
+ if (!api || typeof api !== "object" || !("totalErrors" in api))
55
+ continue;
56
+ if (typeof api.totalErrors === "number" && api.totalErrors > 0)
57
+ failed = true;
58
+ }
59
+ }
60
+ const assistantText = text.join("\n");
61
+ const apiError = text.find((line) => line.startsWith("[API Error:"));
62
+ if (apiError)
63
+ failed = true;
64
+ if (!failed)
65
+ return { assistantText };
66
+ return {
67
+ assistantText,
68
+ failure: apiError ?? errorText ?? resultText ?? (denialText ? `qwen permission denied: ${denialText}` : "qwen reported a startup failure"),
69
+ };
70
+ }
71
+ // OBS-903: the daemon hands the adapter the WHOLE captured stream, never qwen's bare stdout — the
72
+ // launch script's banner and TICKMARKR_EXIT line sit around it and the subprocess driver appends stderr
73
+ // (the headless yolo warning) into the same buffer. The event array is located, not assumed: the bare
74
+ // buffer first, then the outermost `[{` … `}]` span. A live PONG completion read as "unparseable" and
75
+ // merged only through the harvest path until this (clause-4 probe, RULING-222-42).
76
+ function eventArray(raw) {
77
+ const start = raw.indexOf("[{");
78
+ const end = raw.lastIndexOf("}]");
79
+ for (const text of [raw.trim(), start !== -1 && end > start ? raw.slice(start, end + 2) : ""]) {
80
+ if (!text.startsWith("["))
81
+ continue;
82
+ try {
83
+ const parsed = JSON.parse(text);
84
+ if (Array.isArray(parsed))
85
+ return parsed;
86
+ }
87
+ catch { /* try the next shape */ }
88
+ }
89
+ return undefined;
90
+ }
91
+ // Decode qwen's JSON envelope before scanning only decoded assistant text for the worker trailer.
92
+ export function parseQwenResult(raw, nonce) {
93
+ const malformed = {
94
+ ok: false,
95
+ summary: raw.trim() ? "unparseable qwen JSON event stream" : "qwen produced no JSON event stream",
96
+ deviations: [],
97
+ raw,
98
+ cause: raw.trim() ? "malformed-verdict" : "empty-output",
99
+ };
100
+ const parsed = eventArray(raw);
101
+ if (!parsed)
102
+ return malformed;
103
+ const decoded = decodeQwenEvents(parsed);
104
+ if (decoded.failure !== undefined) {
105
+ return {
106
+ ok: false,
107
+ summary: decoded.failure,
108
+ deviations: [],
109
+ raw,
110
+ cause: "startup-failure",
111
+ };
112
+ }
113
+ return { ...parseWorkerResult(decoded.assistantText, nonce), raw };
114
+ }
115
+ function probeQwen() {
116
+ const result = spawnSync("qwen", ["--version"], { encoding: "utf8", timeout: 10_000 });
117
+ if (result.error || result.status !== 0)
118
+ return { installed: false, authed: false, models: [] };
119
+ const version = (result.stdout || result.stderr).trim().split("\n")[0] ?? "";
120
+ if (!QWEN_VERSION_IDENTITY.test(version)) {
121
+ return { installed: false, authed: false, models: [], note: `qwen binary identity mismatch: ${version || "empty version"}` };
122
+ }
123
+ return {
124
+ installed: true,
125
+ authed: true,
126
+ version,
127
+ models: [],
128
+ note: "auth assumed; verified at dispatch (qwen reports API failures inside exit-0 success envelopes)",
129
+ };
130
+ }
131
+ export const qwen = {
132
+ id: "qwen",
133
+ vendor: "alibaba",
134
+ probeCwd: "neutral",
135
+ probe: async () => probeQwen(),
136
+ channels: (cfg) => channelsFromConfig("qwen", cfg),
137
+ hardcodedFlags: { binary: "qwen", flags: ["--approval-mode", "-m", "-o", "-p"] },
138
+ headlessCommand: (promptFile, model) => `${QWEN_SKIP_UPDATE} qwen --approval-mode yolo -m ${shq(model)} -o json -p '' < ${shq(promptFile)}`,
139
+ // OBS-905: qwen has NO interactive form. The `-i "$(cat prompt)"` TUI launch put the whole prompt in
140
+ // argv (the OBS-889 leak-and-census shape) and produced a rendered transcript the JSON decoder above
141
+ // can never read — under the herdr driver every qwen task read "unparseable" and merged only by harvest.
142
+ // The headless form runs in the visible pane and the same parser reads it in every driver; the daemon's
143
+ // mode-fallback branch (daemon.ts, `icmd === null`) journals the choice once per run.
144
+ interactiveCommand: () => null,
145
+ invoke(task, _cwd, assignment, ctx) {
146
+ return { command: this.headlessCommand(ctx.promptFile, assignment.model) };
147
+ },
148
+ parse: parseQwenResult,
149
+ trustDialog: {
150
+ kind: "none",
151
+ reason: "qwen 0.21.15 showed no workspace-trust prompt in a fresh repository during the recorded interactive probe (.planning/assessments/2026-09-03-qwen-cli-probe/README.md, 2026-09-03); enabling security.folderTrust falsifies this declaration",
152
+ },
153
+ };