tickmarkr 2.3.0 → 2.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +12 -4
- package/dist/adapters/catalog-remote.js +97 -45
- package/dist/adapters/catalog.js +5 -3
- package/dist/adapters/claude-code.d.ts +1 -1
- package/dist/adapters/claude-code.js +8 -5
- package/dist/adapters/codex.js +6 -7
- package/dist/adapters/model-lints.d.ts +9 -5
- package/dist/adapters/model-lints.js +56 -15
- package/dist/adapters/model-windows.js +11 -0
- package/dist/adapters/prompt.js +1 -0
- package/dist/adapters/qwen.d.ts +5 -0
- package/dist/adapters/qwen.js +153 -0
- package/dist/adapters/registry.js +13 -1
- package/dist/adapters/types.d.ts +1 -0
- package/dist/adapters/types.js +1 -0
- package/dist/cli/commands/compile.d.ts +3 -0
- package/dist/cli/commands/compile.js +91 -34
- package/dist/cli/commands/doctor.d.ts +4 -3
- package/dist/cli/commands/doctor.js +28 -9
- package/dist/cli/commands/fleet.d.ts +4 -0
- package/dist/cli/commands/fleet.js +60 -15
- package/dist/cli/commands/init.js +12 -13
- package/dist/cli/commands/plan.js +75 -11
- package/dist/cli/commands/run.js +20 -1
- package/dist/cli/commands/status.d.ts +1 -0
- package/dist/cli/commands/status.js +59 -16
- package/dist/cli/commands/verify.d.ts +1 -0
- package/dist/cli/commands/verify.js +5 -0
- package/dist/cli/commands/version.js +2 -2
- package/dist/cli/index.d.ts +1 -1
- package/dist/cli/index.js +2 -2
- package/dist/compile/collateral.d.ts +14 -12
- package/dist/compile/collateral.js +32 -33
- package/dist/compile/index.d.ts +4 -1
- package/dist/compile/index.js +53 -8
- package/dist/compile/native.d.ts +4 -2
- package/dist/compile/native.js +63 -6
- package/dist/compile/ownership.js +41 -10
- package/dist/config/config.d.ts +1 -0
- package/dist/config/config.js +51 -5
- package/dist/drivers/herdr.d.ts +2 -0
- package/dist/drivers/herdr.js +43 -4
- package/dist/drivers/orca.d.ts +18 -1
- package/dist/drivers/orca.js +163 -15
- package/dist/drivers/types.d.ts +10 -0
- package/dist/gates/baseline.d.ts +26 -2
- package/dist/gates/baseline.js +115 -13
- package/dist/gates/review.d.ts +6 -4
- package/dist/gates/review.js +26 -31
- package/dist/gates/run-gates.d.ts +5 -2
- package/dist/gates/run-gates.js +34 -17
- package/dist/graph/graph.d.ts +20 -0
- package/dist/graph/graph.js +66 -1
- package/dist/route/preference.d.ts +6 -0
- package/dist/route/preference.js +40 -0
- package/dist/route/router.js +15 -2
- package/dist/run/consult.d.ts +1 -0
- package/dist/run/consult.js +35 -7
- package/dist/run/daemon.d.ts +9 -0
- package/dist/run/daemon.js +266 -59
- package/dist/run/git.d.ts +4 -0
- package/dist/run/git.js +51 -6
- package/dist/run/journal.d.ts +1 -1
- package/dist/run/journal.js +5 -2
- package/dist/run/lock.d.ts +6 -0
- package/dist/run/lock.js +41 -1
- package/dist/tui/ink/fleet-app.d.ts +4 -0
- package/dist/tui/ink/fleet-app.js +45 -16
- package/package.json +59 -1
- package/skills/tickmarkr-overseer/SKILL.md +39 -4
- package/skills/tickmarkr-overseer/scripts/grade-ci.sh +79 -0
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +90 -4
|
@@ -11,6 +11,7 @@ export interface CatalogCache {
|
|
|
11
11
|
modelsDev: unknown;
|
|
12
12
|
artificialAnalysis?: unknown;
|
|
13
13
|
liveBench?: unknown;
|
|
14
|
+
legFetchedAt?: Partial<Record<"modelsDev" | "artificialAnalysis" | "liveBench", string>>;
|
|
14
15
|
}
|
|
15
16
|
export interface CatalogReadResult {
|
|
16
17
|
catalog: CatalogCache;
|
|
@@ -47,11 +48,21 @@ export interface RefreshCatalogOptions {
|
|
|
47
48
|
timeoutMs?: number;
|
|
48
49
|
now?: () => Date;
|
|
49
50
|
}
|
|
51
|
+
export type CatalogLegName = "models.dev" | "Artificial Analysis" | "LiveBench";
|
|
52
|
+
export type CatalogLegStatus = "updated" | "failed" | "skipped";
|
|
53
|
+
export interface CatalogLegResult {
|
|
54
|
+
leg: CatalogLegName;
|
|
55
|
+
status: CatalogLegStatus;
|
|
56
|
+
detail: string;
|
|
57
|
+
retry: string;
|
|
58
|
+
}
|
|
50
59
|
export interface RefreshCatalogResult {
|
|
51
60
|
updated: boolean;
|
|
52
61
|
catalog: CatalogReadResult;
|
|
53
62
|
warning?: string;
|
|
63
|
+
legs: CatalogLegResult[];
|
|
54
64
|
}
|
|
65
|
+
export declare const formatCatalogRefreshLegs: (legs: readonly CatalogLegResult[]) => string;
|
|
55
66
|
export declare const catalogCachePath: (repoRoot: string) => string;
|
|
56
67
|
/**
|
|
57
68
|
* Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
|
|
@@ -66,9 +77,6 @@ export declare function resolveCatalogModel(catalog: CatalogCache, query: {
|
|
|
66
77
|
model: string;
|
|
67
78
|
resolvedModel?: string;
|
|
68
79
|
}): CatalogModelEvidence | undefined;
|
|
69
|
-
/**
|
|
70
|
-
* The named, explicit refresh path. No other function in this module can reach fetch.
|
|
71
|
-
* A failed refresh preserves the previous cache byte-for-byte and returns it fail-open.
|
|
72
|
-
*/
|
|
80
|
+
/** Shared refresh path for the explicit command and the seven-day doctor/fleet guard. */
|
|
73
81
|
export declare function refreshCatalogCommand(opts: RefreshCatalogOptions): Promise<RefreshCatalogResult>;
|
|
74
82
|
export {};
|
|
@@ -9,20 +9,21 @@ export const ARTIFICIAL_ANALYSIS_CATALOG_URL = "https://artificialanalysis.ai/ap
|
|
|
9
9
|
export const LIVEBENCH_TABLE_DATE = "2026_06_25";
|
|
10
10
|
export const LIVEBENCH_TABLE_URL = `https://livebench.ai/table_${LIVEBENCH_TABLE_DATE}.csv`;
|
|
11
11
|
export const LIVEBENCH_CATEGORIES_URL = `https://livebench.ai/categories_${LIVEBENCH_TABLE_DATE}.json`;
|
|
12
|
-
export const CATALOG_CACHE_MAX_AGE_MS =
|
|
12
|
+
export const CATALOG_CACHE_MAX_AGE_MS = 7 * 86_400_000;
|
|
13
13
|
export const CATALOG_REFRESH_TIMEOUT_MS = 10_000;
|
|
14
14
|
const ARTIFICIAL_ANALYSIS_PAGE_SIZE = 100;
|
|
15
15
|
const ARTIFICIAL_ANALYSIS_MAX_PAGES = 100;
|
|
16
16
|
const VENDORED_CATALOG = {
|
|
17
17
|
schemaVersion: 1,
|
|
18
18
|
// Package fallback copied from https://models.dev/api.json on 2026-08-05. It is deliberately
|
|
19
|
-
// small:
|
|
19
|
+
// small: the explicit or seven-day operator-surface refresh owns broad/current coverage.
|
|
20
20
|
fetchedAt: "2026-08-05T20:00:05.000Z",
|
|
21
21
|
modelsDev: {
|
|
22
22
|
anthropic: {
|
|
23
23
|
id: "anthropic",
|
|
24
24
|
models: {
|
|
25
|
-
|
|
25
|
+
// SHIP (OBS-871, 2026-09-03): users resolving the current Fable alias need the 5.1 identity even when refresh is unavailable.
|
|
26
|
+
"claude-fable-5-1": { id: "claude-fable-5-1", cost: { input: 10, output: 50 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
|
|
26
27
|
"claude-opus-4-8": { id: "claude-opus-4-8", cost: { input: 5, output: 25 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
|
|
27
28
|
"claude-sonnet-5": { id: "claude-sonnet-5", cost: { input: 2, output: 10 }, limit: { context: 1_000_000, output: 128_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
|
|
28
29
|
"claude-haiku-4-5-20251001": { id: "claude-haiku-4-5-20251001", cost: { input: 1, output: 5 }, limit: { context: 200_000, output: 64_000 }, reasoning: true, tool_call: true, structured_output: true, attachment: true },
|
|
@@ -59,9 +60,14 @@ const validModelsDevCatalog = (value) => {
|
|
|
59
60
|
});
|
|
60
61
|
};
|
|
61
62
|
const staleAt = (fetchedAt, now) => {
|
|
62
|
-
const age = now.getTime() - Date.parse(fetchedAt);
|
|
63
|
+
const age = now.getTime() - Date.parse(fetchedAt ?? "");
|
|
63
64
|
return !Number.isFinite(age) || age > CATALOG_CACHE_MAX_AGE_MS;
|
|
64
65
|
};
|
|
66
|
+
const legFetchedAt = (catalog, leg) => catalog.legFetchedAt?.[leg] ?? catalog.fetchedAt;
|
|
67
|
+
const catalogStale = (catalog, now, source) => source === "vendored" || staleAt(legFetchedAt(catalog, "modelsDev"), now) || staleAt(legFetchedAt(catalog, "liveBench"), now)
|
|
68
|
+
|| (catalog.artificialAnalysis !== undefined && catalog.legFetchedAt?.artificialAnalysis !== undefined
|
|
69
|
+
&& staleAt(catalog.legFetchedAt.artificialAnalysis, now));
|
|
70
|
+
export const formatCatalogRefreshLegs = (legs) => `catalog refresh: ${legs.map((leg) => `${leg.leg} ${leg.status} (${leg.detail}; ${leg.retry})`).join("; ")}`;
|
|
65
71
|
export const catalogCachePath = (repoRoot) => join(repoRoot, ".tickmarkr", "catalog-cache.json");
|
|
66
72
|
/**
|
|
67
73
|
* Doctor's catalog seam: one synchronous local read with a vendored fail-open fallback.
|
|
@@ -73,14 +79,14 @@ export function readCachedCatalog(repoRoot, opts = {}) {
|
|
|
73
79
|
const parsed = JSON.parse(readFileSync(catalogCachePath(repoRoot), "utf8"));
|
|
74
80
|
if (!validCache(parsed))
|
|
75
81
|
throw new Error("catalog cache schema is invalid");
|
|
76
|
-
return { catalog: parsed, source: "cache", stale:
|
|
82
|
+
return { catalog: parsed, source: "cache", stale: catalogStale(parsed, now, "cache") };
|
|
77
83
|
}
|
|
78
84
|
catch (error) {
|
|
79
85
|
const missing = error.code === "ENOENT";
|
|
80
86
|
return {
|
|
81
87
|
catalog: VENDORED_CATALOG,
|
|
82
88
|
source: "vendored",
|
|
83
|
-
stale:
|
|
89
|
+
stale: true,
|
|
84
90
|
...(!missing ? { warning: error instanceof Error ? error.message : String(error) } : {}),
|
|
85
91
|
};
|
|
86
92
|
}
|
|
@@ -226,6 +232,7 @@ function assertUsableLiveBench(rows, categories) {
|
|
|
226
232
|
// `-thinking-auto-medium-effort`); the highest effort is the model at its best. Rank by
|
|
227
233
|
// hyphen-delimited token so `xhigh` never reads as `high`.
|
|
228
234
|
const LIVEBENCH_EFFORT_RANK = { max: 5, xhigh: 4, high: 3, medium: 2, low: 1 };
|
|
235
|
+
const LIVEBENCH_VARIANT_RE = /^-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast)(?:-(?:max|xhigh|high|medium|low|thinking|auto|effort|fast))*$/;
|
|
229
236
|
const liveBenchEffortRank = (residue) => residue.split("-").reduce((rank, token) => Math.max(rank, LIVEBENCH_EFFORT_RANK[token] ?? 0), 0);
|
|
230
237
|
const liveBenchCategoryMean = (row, tasks) => {
|
|
231
238
|
const scores = (Array.isArray(tasks) ? tasks : [])
|
|
@@ -233,8 +240,8 @@ const liveBenchCategoryMean = (row, tasks) => {
|
|
|
233
240
|
.filter((score) => score !== undefined);
|
|
234
241
|
return scores.length > 0 ? scores.reduce((sum, score) => sum + score, 0) / scores.length : undefined;
|
|
235
242
|
};
|
|
236
|
-
/** LiveBench
|
|
237
|
-
const liveBenchIdentity = (value) => value.trim().toLowerCase().replace(
|
|
243
|
+
/** LiveBench treats spaces, dots, and punctuation as the same word boundary. */
|
|
244
|
+
const liveBenchIdentity = (value) => value.trim().toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-|-$/g, "");
|
|
238
245
|
function liveBenchIndex(value, identities) {
|
|
239
246
|
const root = record(value);
|
|
240
247
|
if (!root || !Array.isArray(root.rows))
|
|
@@ -249,7 +256,7 @@ function liveBenchIndex(value, identities) {
|
|
|
249
256
|
// Bare startsWith is a false-positive machine: `glm-5` would claim `glm-5.2`. The residue after
|
|
250
257
|
// the fleet id must be empty or an effort suffix.
|
|
251
258
|
const residue = wanted.map((identity) => model.startsWith(identity) ? model.slice(identity.length) : undefined)
|
|
252
|
-
.find((rest) => rest === "" || rest
|
|
259
|
+
.find((rest) => rest === "" || LIVEBENCH_VARIANT_RE.test(rest ?? ""));
|
|
253
260
|
if (residue === undefined)
|
|
254
261
|
continue;
|
|
255
262
|
const rank = liveBenchEffortRank(residue);
|
|
@@ -285,6 +292,7 @@ export function resolveCatalogModel(catalog, query) {
|
|
|
285
292
|
const identities = [
|
|
286
293
|
modelId,
|
|
287
294
|
query.model,
|
|
295
|
+
...[modelId, query.model].flatMap((identity) => identity.includes("/") ? [identity.slice(identity.lastIndexOf("/") + 1)] : []),
|
|
288
296
|
...(typeof model.name === "string" ? [model.name] : []),
|
|
289
297
|
];
|
|
290
298
|
const intelligence = artificialAnalysisIndex(catalog.artificialAnalysis, [
|
|
@@ -400,53 +408,97 @@ async function fetchLiveBench(fetcher, timeoutMs) {
|
|
|
400
408
|
assertUsableLiveBench(rows, categories);
|
|
401
409
|
return { tableDate: LIVEBENCH_TABLE_DATE, categories, rows };
|
|
402
410
|
}
|
|
403
|
-
/**
|
|
404
|
-
* The named, explicit refresh path. No other function in this module can reach fetch.
|
|
405
|
-
* A failed refresh preserves the previous cache byte-for-byte and returns it fail-open.
|
|
406
|
-
*/
|
|
411
|
+
/** Shared refresh path for the explicit command and the seven-day doctor/fleet guard. */
|
|
407
412
|
export async function refreshCatalogCommand(opts) {
|
|
408
413
|
const now = opts.now ?? (() => new Date());
|
|
409
414
|
const current = readCachedCatalog(opts.repoRoot, { now });
|
|
415
|
+
const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
|
|
416
|
+
const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
|
|
417
|
+
const warnings = [];
|
|
418
|
+
const legs = [];
|
|
419
|
+
const stamp = now().toISOString();
|
|
420
|
+
const legAt = {
|
|
421
|
+
modelsDev: legFetchedAt(current.catalog, "modelsDev"),
|
|
422
|
+
liveBench: legFetchedAt(current.catalog, "liveBench"),
|
|
423
|
+
...(current.catalog.artificialAnalysis !== undefined ? { artificialAnalysis: legFetchedAt(current.catalog, "artificialAnalysis") } : {}),
|
|
424
|
+
};
|
|
425
|
+
let modelsDev = current.catalog.modelsDev;
|
|
426
|
+
let modelsDevUpdated = false;
|
|
410
427
|
try {
|
|
411
|
-
|
|
412
|
-
const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
|
|
413
|
-
const modelsDev = await fetchCatalog(fetcher, MODELS_DEV_CATALOG_URL, {}, timeoutMs, (r) => r.json());
|
|
428
|
+
modelsDev = await fetchCatalog(fetcher, MODELS_DEV_CATALOG_URL, {}, timeoutMs, (r) => r.json());
|
|
414
429
|
if (!validModelsDevCatalog(modelsDev))
|
|
415
430
|
throw new Error("models.dev catalog schema is invalid");
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
431
|
+
modelsDevUpdated = true;
|
|
432
|
+
legAt.modelsDev = stamp;
|
|
433
|
+
legs.push({ leg: "models.dev", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
|
|
434
|
+
}
|
|
435
|
+
catch (error) {
|
|
436
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
437
|
+
warnings.push(`models.dev refresh failed: ${detail}`);
|
|
438
|
+
legs.push({ leg: "models.dev", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
439
|
+
}
|
|
440
|
+
let artificialAnalysis = current.catalog.artificialAnalysis;
|
|
441
|
+
let artificialAnalysisUpdated = false;
|
|
442
|
+
const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
|
|
443
|
+
if (apiKey) {
|
|
424
444
|
try {
|
|
425
|
-
|
|
445
|
+
artificialAnalysis = await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs);
|
|
446
|
+
artificialAnalysisUpdated = true;
|
|
447
|
+
legAt.artificialAnalysis = stamp;
|
|
448
|
+
legs.push({ leg: "Artificial Analysis", status: "updated", detail: "fetched", retry: "next retry after this leg is stale" });
|
|
426
449
|
}
|
|
427
450
|
catch (error) {
|
|
428
|
-
const
|
|
429
|
-
|
|
451
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
452
|
+
warnings.push(`Artificial Analysis refresh failed: ${detail}`);
|
|
453
|
+
legs.push({ leg: "Artificial Analysis", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
430
454
|
}
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
};
|
|
455
|
+
}
|
|
456
|
+
else {
|
|
457
|
+
delete legAt.artificialAnalysis;
|
|
458
|
+
legs.push({ leg: "Artificial Analysis", status: "skipped", detail: "no API key", retry: "retry when ARTIFICIAL_ANALYSIS_API_KEY is set" });
|
|
459
|
+
}
|
|
460
|
+
let liveBench = current.catalog.liveBench;
|
|
461
|
+
let liveBenchUpdated = false;
|
|
462
|
+
try {
|
|
463
|
+
liveBench = await fetchLiveBench(fetcher, timeoutMs);
|
|
464
|
+
liveBenchUpdated = true;
|
|
465
|
+
legAt.liveBench = stamp;
|
|
466
|
+
legs.push({ leg: "LiveBench", status: "updated", detail: `table ${LIVEBENCH_TABLE_DATE}`, retry: "next retry after this leg is stale" });
|
|
444
467
|
}
|
|
445
468
|
catch (error) {
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
warning: error instanceof Error ? error.message : String(error),
|
|
450
|
-
};
|
|
469
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
470
|
+
warnings.push(`LiveBench refresh failed: ${detail}`);
|
|
471
|
+
legs.push({ leg: "LiveBench", status: "failed", detail, retry: "retry next refresh because this leg kept its prior age" });
|
|
451
472
|
}
|
|
473
|
+
const updated = modelsDevUpdated || artificialAnalysisUpdated || liveBenchUpdated;
|
|
474
|
+
// models.dev is the cache spine: do not write a cache that would make the vendored fallback look
|
|
475
|
+
// fetched. Independent legs may only merge onto models.dev fetched now or already held in cache.
|
|
476
|
+
if (!updated || (!modelsDevUpdated && current.source !== "cache")) {
|
|
477
|
+
const reportedLegs = !modelsDevUpdated && current.source !== "cache"
|
|
478
|
+
? legs.map((leg) => leg.status === "updated"
|
|
479
|
+
? {
|
|
480
|
+
...leg,
|
|
481
|
+
status: "failed",
|
|
482
|
+
detail: "fetched but discarded — models.dev spine unavailable",
|
|
483
|
+
retry: "retry next refresh",
|
|
484
|
+
}
|
|
485
|
+
: leg)
|
|
486
|
+
: legs;
|
|
487
|
+
return { updated: false, catalog: current, legs: reportedLegs, ...(warnings.length ? { warning: warnings.join("; ") } : {}) };
|
|
488
|
+
}
|
|
489
|
+
const catalog = {
|
|
490
|
+
schemaVersion: 1,
|
|
491
|
+
fetchedAt: legAt.modelsDev ?? current.catalog.fetchedAt,
|
|
492
|
+
modelsDev,
|
|
493
|
+
...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
|
|
494
|
+
...(liveBench !== undefined ? { liveBench } : {}),
|
|
495
|
+
legFetchedAt: legAt,
|
|
496
|
+
};
|
|
497
|
+
writeCatalogCache(opts.repoRoot, catalog);
|
|
498
|
+
return {
|
|
499
|
+
updated,
|
|
500
|
+
catalog: readCachedCatalog(opts.repoRoot, { now }),
|
|
501
|
+
legs,
|
|
502
|
+
...(warnings.length ? { warning: warnings.join("; ") } : {}),
|
|
503
|
+
};
|
|
452
504
|
}
|
package/dist/adapters/catalog.js
CHANGED
|
@@ -9,6 +9,7 @@ import { cursorAgent } from "./cursor-agent.js";
|
|
|
9
9
|
import { grok } from "./grok.js";
|
|
10
10
|
import { kimi } from "./kimi.js";
|
|
11
11
|
import { opencode } from "./opencode.js";
|
|
12
|
+
import { qwen, QWEN_VERSION_IDENTITY } from "./qwen.js";
|
|
12
13
|
import { pi } from "./pi.js";
|
|
13
14
|
import { TRUST_DIALOG_VARIANTS, TrustDialogSchema } from "./types.js";
|
|
14
15
|
export const CLI_NAME_RE = /^[a-z0-9-]+$/;
|
|
@@ -105,10 +106,10 @@ function validateCliEntry(entry) {
|
|
|
105
106
|
}
|
|
106
107
|
// Package-owned, deterministic order. This is the sole shipped definition array: candidate-name
|
|
107
108
|
// compatibility and advisory/routable projections below are all derived from it.
|
|
108
|
-
const native = (adapter, binary) => ({
|
|
109
|
+
const native = (adapter, binary, identity = ".+") => ({
|
|
109
110
|
id: adapter.id,
|
|
110
111
|
binary,
|
|
111
|
-
identity
|
|
112
|
+
identity,
|
|
112
113
|
vendor: adapter.vendor,
|
|
113
114
|
drive: { adapter },
|
|
114
115
|
});
|
|
@@ -122,7 +123,8 @@ export const CLI_CATALOG = [
|
|
|
122
123
|
native(pi, "pi"),
|
|
123
124
|
native(grok, "grok"),
|
|
124
125
|
native(kimi, "kimi"),
|
|
125
|
-
|
|
126
|
+
native(qwen, "qwen", QWEN_VERSION_IDENTITY.source),
|
|
127
|
+
"gemini", "aider", "goose", "amp", "droid", "auggie", "crush",
|
|
126
128
|
{
|
|
127
129
|
id: "omp",
|
|
128
130
|
binary: "omp",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { type AuthHealth, type TrustDialog, type WorkerAdapter } from "./types.js";
|
|
2
2
|
export declare const CLAUDE_ALIAS_IDENTITY_STAMPS: {
|
|
3
|
-
readonly fable: "claude-fable-5";
|
|
3
|
+
readonly fable: "claude-fable-5-1";
|
|
4
4
|
readonly opus: "claude-opus-4-8";
|
|
5
5
|
readonly sonnet: "claude-sonnet-5";
|
|
6
6
|
readonly haiku: "claude-haiku-4-5-20251001";
|
|
@@ -29,7 +29,8 @@ const MAX_SESSION_BYTES = 8_000_000; // per-file cap; a runaway JSONL cannot mak
|
|
|
29
29
|
// claude-opus-5 channel to the repo overlay — it did not re-date this alias's stamps, so the
|
|
30
30
|
// alias channel still carries 4-8-dated tier/pricing while serving 5. That warning is true.
|
|
31
31
|
export const CLAUDE_ALIAS_IDENTITY_STAMPS = {
|
|
32
|
-
|
|
32
|
+
// OBS-871, 2026-09-03: Fable's floating alias now resolves to the 5.1 benchmark identity.
|
|
33
|
+
fable: "claude-fable-5-1",
|
|
33
34
|
opus: "claude-opus-4-8",
|
|
34
35
|
sonnet: "claude-sonnet-5",
|
|
35
36
|
haiku: "claude-haiku-4-5-20251001",
|
|
@@ -209,7 +210,7 @@ export const claudeCode = {
|
|
|
209
210
|
channels: (cfg) => channelsFromConfig("claude-code", cfg),
|
|
210
211
|
// v1.65 T3: every flag the command builders below hardcode — doctor checks `claude --help` still
|
|
211
212
|
// lists each (all present on claude 2.x, verified 2026-07-22). Advisory only, never routing.
|
|
212
|
-
hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions"] },
|
|
213
|
+
hardcodedFlags: { binary: "claude", flags: ["-p", "--model", "--permission-mode", "--strict-mcp-config", "--mcp-config", "--output-format", "-r", "--prompt-suggestions", "--settings"] },
|
|
213
214
|
// --strict-mcp-config --mcp-config '{"mcpServers":{}}': pin the MCP surface to empty so fresh-worktree
|
|
214
215
|
// workers/gates don't load project .mcp.json servers (herdr scrapes dialogs as idle — v1.4 incident,
|
|
215
216
|
// memory tickmarkr-worker-mcp-dialog-stall). Live-verified 2026-07-10 on claude 2.1.205 (operator check):
|
|
@@ -219,7 +220,9 @@ export const claudeCode = {
|
|
|
219
220
|
// Gotchas (both bit the 2026-07-10 live check): bare '{}' is REJECTED ("mcpServers: expected record"),
|
|
220
221
|
// and --mcp-config is VARIADIC — a positional after it is eaten as a config-file path, so another
|
|
221
222
|
// flag must always follow the value, never the prompt.
|
|
222
|
-
|
|
223
|
+
// The empty -p argument selects print mode while stdin carries the prompt, keeping its nonce out
|
|
224
|
+
// of process argv. The redirect path is shell-quoted independently from the model.
|
|
225
|
+
headlessCommand: (promptFile, model) => `claude -p '' --model ${shq(model)} --permission-mode bypassPermissions --strict-mcp-config --mcp-config '{"mcpServers":{}}' --output-format text < ${shq(promptFile)}`,
|
|
223
226
|
// HYG-03 / OBS-137: the residual first-entry dialog is workspace trust, not MCP config loading.
|
|
224
227
|
// Claude's only store is global last-writer-wins ~/.claude.json, so tickmarkr still does not seed it;
|
|
225
228
|
// the daemon safely answers only the exact adapter-declared dialog once per slot.
|
|
@@ -236,12 +239,12 @@ export const claudeCode = {
|
|
|
236
239
|
// live check ate the prompt), and --prompt-suggestions takes an OPTIONAL value — appended directly
|
|
237
240
|
// before the prompt it would swallow it the same way. So the setting's value is always followed by
|
|
238
241
|
// another flag, never by the prompt positional.
|
|
239
|
-
interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
242
|
+
interactiveCommand: (promptFile, model) => `claude --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
240
243
|
trustDialog: CLAUDE_TRUST_DIALOG,
|
|
241
244
|
inputBox: CLAUDE_INPUT_BOX,
|
|
242
245
|
// A resumed attempt lands in the same painted editor, so it carries the same ghost-text suppression
|
|
243
246
|
// and the same value-then-flag placement.
|
|
244
|
-
resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
247
|
+
resumeCommand: (sessionId, promptFile, model) => `claude -r ${shq(sessionId)} --model ${shq(model)} --strict-mcp-config --mcp-config '{"mcpServers":{}}' --settings '{"promptSuggestionEnabled":false}' --prompt-suggestions false --permission-mode bypassPermissions "$(cat ${shq(promptFile)})"`,
|
|
245
248
|
invoke(task, _cwd, a, ctx) {
|
|
246
249
|
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
|
247
250
|
},
|
package/dist/adapters/codex.js
CHANGED
|
@@ -184,17 +184,16 @@ export const codex = {
|
|
|
184
184
|
probeConcurrency: 1,
|
|
185
185
|
probe: async () => probeVersion("codex"),
|
|
186
186
|
channels: (cfg) => channelsFromConfig("codex", cfg),
|
|
187
|
-
// v1.65 T3: every flag the command
|
|
187
|
+
// v1.65 T3: every flag the command builder below hardcodes (incl. codexMcpSuppressionFlags' -c/
|
|
188
188
|
// --disable and GITDIR_WRITABLE's -c) — all listed by top-level `codex --help`, verified 2026-07-22.
|
|
189
|
-
hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-
|
|
189
|
+
hardcodedFlags: { binary: "codex", flags: ["--sandbox", "--model", "-c", "--disable", "--dangerously-bypass-hook-trust"] },
|
|
190
190
|
// --sandbox workspace-write is the autonomous sandbox mode (codex v0.144.1+)
|
|
191
191
|
// MCP suppression built per dispatch (config can change between runs) — see codexMcpSuppressionFlags.
|
|
192
192
|
// CODEX_HOOK_TRUST (OBS-125) clears the per-worktree "Hooks need review" gate while keeping the sandbox.
|
|
193
|
-
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)}
|
|
194
|
-
// TUI
|
|
195
|
-
//
|
|
196
|
-
|
|
197
|
-
interactiveCommand: (promptFile, model) => `codex -a never -s workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} "$(cat ${shq(promptFile)})"`,
|
|
193
|
+
headlessCommand: (promptFile, model) => `codex exec --sandbox workspace-write ${CODEX_HOOK_TRUST} ${codexMcpSuppressionFlags()} ${GITDIR_WRITABLE} --model ${shq(model)} - < ${shq(promptFile)}`,
|
|
194
|
+
// OBS-889: Codex's TUI has no file/stdin prompt form. Returning null makes the daemon journal
|
|
195
|
+
// worker-mode-fallback before it runs the argv-safe headless command in the visible pane.
|
|
196
|
+
interactiveCommand: () => null,
|
|
198
197
|
invoke(task, _cwd, a, ctx) {
|
|
199
198
|
return { command: this.headlessCommand(ctx.promptFile, a.model) };
|
|
200
199
|
},
|
|
@@ -4,6 +4,14 @@ import { type AuthHealth, type WorkerAdapter } from "./types.js";
|
|
|
4
4
|
import { type CatalogModelEvidence, type CatalogReadResult } from "./catalog-remote.js";
|
|
5
5
|
export declare const SEED_STAMPED = "2026-07-09";
|
|
6
6
|
export declare const MODEL_STALE_DAYS = 30;
|
|
7
|
+
export type FleetUnclassifiedModel = {
|
|
8
|
+
adapter: string;
|
|
9
|
+
model: string;
|
|
10
|
+
detectedAt?: string;
|
|
11
|
+
variants?: string[];
|
|
12
|
+
/** The real CLI id written by classify when the display row is a collapsed base. */
|
|
13
|
+
classifyModel?: string;
|
|
14
|
+
};
|
|
7
15
|
/**
|
|
8
16
|
* Operator directive 2026-08-13 ("we should exclude all retired models"): the fleet models
|
|
9
17
|
* screen hides these classes BY DEFAULT — omp alone reports 218 ids, most of them dated
|
|
@@ -84,9 +92,5 @@ export declare function suggestOverlay(cfg: TickmarkrConfig, health: Record<stri
|
|
|
84
92
|
resolvedModel?: (adapter: string, model: string) => string | undefined;
|
|
85
93
|
}): string;
|
|
86
94
|
/** Unclassified models surfaced for fleet screen 2 (doctor matrix math, no tier fabrication). */
|
|
87
|
-
export declare function fleetUnclassifiedModels(cfg: TickmarkrConfig, health: Record<string, AuthHealth>, adapters: WorkerAdapter[]):
|
|
88
|
-
adapter: string;
|
|
89
|
-
model: string;
|
|
90
|
-
detectedAt?: string;
|
|
91
|
-
}[];
|
|
95
|
+
export declare function fleetUnclassifiedModels(cfg: TickmarkrConfig, health: Record<string, AuthHealth>, adapters: WorkerAdapter[]): FleetUnclassifiedModel[];
|
|
92
96
|
export {};
|
|
@@ -11,12 +11,52 @@ export const SEED_STAMPED = "2026-07-09";
|
|
|
11
11
|
// knowledge past this age gets a "rerun tickmarkr doctor" nudge (BLOCKED_POLL_MS-style named constant).
|
|
12
12
|
export const MODEL_STALE_DAYS = 30;
|
|
13
13
|
const DAY_MS = 86400000;
|
|
14
|
-
// cursor-agent
|
|
15
|
-
//
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
const LINT_VARIANT_RE = /^auto$|-(fast|minimal|none|low|medium|high|xhigh|max|thinking)$/;
|
|
14
|
+
// cursor-agent reports mostly effort/speed variants. Keep doctor.json raw, but collapse those ids
|
|
15
|
+
// into one base row; a base-less classify writes the highest-effort real id.
|
|
16
|
+
const VARIANT_SUFFIX_RE = /-(fast|minimal|none|low|medium|high|xhigh|max|thinking)$/;
|
|
17
|
+
const VARIANT_RANK = { max: 6, xhigh: 5, high: 4, medium: 3, low: 2, minimal: 1, none: 0, thinking: 0, fast: 0 };
|
|
19
18
|
const LINT_CAP = 5;
|
|
19
|
+
const variantBase = (model) => {
|
|
20
|
+
let base = model;
|
|
21
|
+
while (VARIANT_SUFFIX_RE.test(base))
|
|
22
|
+
base = base.replace(VARIANT_SUFFIX_RE, "");
|
|
23
|
+
return base;
|
|
24
|
+
};
|
|
25
|
+
const variantRank = (model) => {
|
|
26
|
+
let rank = 0;
|
|
27
|
+
for (const token of model.split("-"))
|
|
28
|
+
rank = Math.max(rank, VARIANT_RANK[token] ?? 0);
|
|
29
|
+
return rank;
|
|
30
|
+
};
|
|
31
|
+
function collapseUnclassified(detected, configured) {
|
|
32
|
+
const groups = new Map();
|
|
33
|
+
for (const model of detected) {
|
|
34
|
+
if (model === "auto" || configured.has(model))
|
|
35
|
+
continue;
|
|
36
|
+
const base = variantBase(model);
|
|
37
|
+
if (configured.has(base))
|
|
38
|
+
continue;
|
|
39
|
+
const group = groups.get(base) ?? { bare: false, variants: [] };
|
|
40
|
+
if (model === base)
|
|
41
|
+
group.bare = true;
|
|
42
|
+
else
|
|
43
|
+
group.variants.push(model);
|
|
44
|
+
groups.set(base, group);
|
|
45
|
+
}
|
|
46
|
+
return [...groups].map(([model, group]) => {
|
|
47
|
+
if (!group.variants.length)
|
|
48
|
+
return { adapter: "", model };
|
|
49
|
+
const classifyModel = group.bare
|
|
50
|
+
? model
|
|
51
|
+
: group.variants.reduce((best, candidate) => variantRank(candidate) > variantRank(best) ? candidate : best);
|
|
52
|
+
return {
|
|
53
|
+
adapter: "",
|
|
54
|
+
model,
|
|
55
|
+
variants: group.variants,
|
|
56
|
+
...(classifyModel !== model ? { classifyModel } : {}),
|
|
57
|
+
};
|
|
58
|
+
});
|
|
59
|
+
}
|
|
20
60
|
/**
|
|
21
61
|
* Operator directive 2026-08-13 ("we should exclude all retired models"): the fleet models
|
|
22
62
|
* screen hides these classes BY DEFAULT — omp alone reports 218 ids, most of them dated
|
|
@@ -477,9 +517,9 @@ export function modelLints(cfg, health, adapters, opts) {
|
|
|
477
517
|
lints.push(`${id}: tiers lists ${model} — CLI no longer reports it; tombstone it (${model}: null overlay) or verify the id`);
|
|
478
518
|
}
|
|
479
519
|
}
|
|
480
|
-
const extra = detected
|
|
520
|
+
const extra = collapseUnclassified(detected, new Set(configured));
|
|
481
521
|
if (extra.length) {
|
|
482
|
-
const shown = extra.slice(0, cap).join(", ");
|
|
522
|
+
const shown = extra.slice(0, cap).map((row) => row.model).join(", ");
|
|
483
523
|
const tail = extra.length > cap ? `, +${extra.length - cap} more${doctorRef}` : "";
|
|
484
524
|
lints.push(`${id}: reports ${extra.length} model(s) not in tiers (${shown}${tail}) — classify before routing (benchmark policy)`);
|
|
485
525
|
}
|
|
@@ -533,7 +573,7 @@ export function suggestOverlay(cfg, health, adapters, stateDir = DEFAULT_STATE_D
|
|
|
533
573
|
// Tombstones: configured ids the CLI no longer reports. Ids are operator-authored (from cfg) → MODEL_ID_RE only.
|
|
534
574
|
const tombstones = configured.filter((model) => !detected.includes(model) && MODEL_ID_RE.test(model));
|
|
535
575
|
// Additions: detected ids not in cfg. WHOLE line commented, no tier (MODEL-06). Ids come from an external
|
|
536
|
-
// CLI → MODEL_ID_RE (defense-in-depth, T-21-01)
|
|
576
|
+
// CLI → MODEL_ID_RE (defense-in-depth, T-21-01); effort variants collapse before this loop.
|
|
537
577
|
// RELATIONAL gate (no capability judgment — "looks like an embedding model" is auto-tiering's cousin, the
|
|
538
578
|
// NaN-routing class the v1.5 decision forbids): a detected id is suggested iff it shares a provider prefix
|
|
539
579
|
// (clause a) OR a canonical segment (clause b, the RENAME case: opencode/glm-5.2 ⇒ zai-coding-plan/glm-5.2)
|
|
@@ -547,8 +587,9 @@ export function suggestOverlay(cfg, health, adapters, stateDir = DEFAULT_STATE_D
|
|
|
547
587
|
const cfgCanon = new Set(configured.map(canonical));
|
|
548
588
|
const additions = [];
|
|
549
589
|
let omitted = 0;
|
|
550
|
-
for (const
|
|
551
|
-
|
|
590
|
+
for (const row of collapseUnclassified(detected, new Set(configured))) {
|
|
591
|
+
const model = row.classifyModel ?? row.model;
|
|
592
|
+
if (!MODEL_ID_RE.test(model))
|
|
552
593
|
continue;
|
|
553
594
|
if (configured.length > 0 && !cfgPrefixes.has(providerPrefix(model)) && !cfgCanon.has(canonical(model))) {
|
|
554
595
|
omitted++;
|
|
@@ -619,11 +660,11 @@ export function fleetUnclassifiedModels(cfg, health, adapters) {
|
|
|
619
660
|
continue;
|
|
620
661
|
const configured = new Set(Object.keys(cfg.tiers[id]?.models ?? {}));
|
|
621
662
|
const date = h?.modelsDetectedAt?.split("T")[0];
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
}
|
|
663
|
+
out.push(...collapseUnclassified(detected, configured).map((row) => ({
|
|
664
|
+
...row,
|
|
665
|
+
adapter: id,
|
|
666
|
+
...(date ? { detectedAt: date } : {}),
|
|
667
|
+
})));
|
|
627
668
|
}
|
|
628
669
|
return out;
|
|
629
670
|
}
|
|
@@ -28,6 +28,9 @@ const CURSOR_SOURCE = "https://cursor.com/help/ai-features/max-mode";
|
|
|
28
28
|
const ZAI_SOURCE = "https://z.ai/blog/glm-5.2";
|
|
29
29
|
const XAI_SOURCE = "https://docs.x.ai/developers/models/grok-4.5";
|
|
30
30
|
const KIMI_SOURCE = "https://www.kimi.com/code/docs/en/kimi-code-cli/configuration/config-files.html";
|
|
31
|
+
const GOOGLE_SOURCE = "https://ai.google.dev/gemini-api/docs/models";
|
|
32
|
+
const QWEN_SOURCE = "https://qwenlm.github.io/";
|
|
33
|
+
const OBS_871_READ_DATE = "2026-09-03";
|
|
31
34
|
const READ_DATE = "2026-08-05";
|
|
32
35
|
const VENDORED_MODEL_WINDOW_CLAIMS = [
|
|
33
36
|
{ modelId: "fable", window: 1_000_000, source: ANTHROPIC_SOURCE, readDate: READ_DATE },
|
|
@@ -40,8 +43,16 @@ const VENDORED_MODEL_WINDOW_CLAIMS = [
|
|
|
40
43
|
{ modelId: "gpt-5.6-luna", window: 1_050_000, source: OPENAI_SOURCE, readDate: READ_DATE },
|
|
41
44
|
{ modelId: "composer-2.5", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
|
|
42
45
|
{ modelId: "composer-2.5-fast", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
|
|
46
|
+
{ modelId: "claude-fable-5-1", window: 1_000_000, source: ANTHROPIC_SOURCE, readDate: OBS_871_READ_DATE },
|
|
47
|
+
{ modelId: "gemini-3.8-flash", window: 1_000_000, source: GOOGLE_SOURCE, readDate: OBS_871_READ_DATE },
|
|
48
|
+
{ modelId: "google/gemini-3.8-flash", window: 1_000_000, source: GOOGLE_SOURCE, readDate: OBS_871_READ_DATE },
|
|
43
49
|
{ modelId: "zai-coding-plan/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: READ_DATE },
|
|
44
50
|
{ modelId: "zai/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: READ_DATE },
|
|
51
|
+
{ modelId: "zai/glm-5.3", window: 1_000_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
|
|
52
|
+
{ modelId: "zai/glm-5.3-flash", window: 200_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
|
|
53
|
+
{ modelId: "alibaba/qwen3.8-max", window: 1_000_000, source: QWEN_SOURCE, readDate: OBS_871_READ_DATE },
|
|
54
|
+
{ modelId: "qwen3.8-max", window: 1_000_000, source: QWEN_SOURCE, readDate: OBS_871_READ_DATE },
|
|
55
|
+
{ modelId: "prime-inference/z-ai/glm-5.2", window: 1_000_000, source: ZAI_SOURCE, readDate: OBS_871_READ_DATE },
|
|
45
56
|
{ modelId: "grok-4.5", window: 500_000, source: XAI_SOURCE, readDate: READ_DATE },
|
|
46
57
|
{ modelId: "grok-composer-2.5-fast", window: 200_000, source: CURSOR_SOURCE, readDate: READ_DATE },
|
|
47
58
|
{ modelId: "kimi-code/k3", window: 1_048_576, source: KIMI_SOURCE, readDate: READ_DATE },
|
package/dist/adapters/prompt.js
CHANGED
|
@@ -23,6 +23,7 @@ ${task.files.length ? `\n## File scope — touch ONLY paths matching:\n${list(ta
|
|
|
23
23
|
- Work only inside the current directory (your isolated worktree). Never push. Never switch branches.
|
|
24
24
|
- Make small atomic git commits as you go (git add + git commit, conventional messages).
|
|
25
25
|
- Touch ONLY paths matching the file scope. Out-of-scope edits FAIL the scope gate. The operator's allowlist is fixed when the run starts and nothing you do can change it while your work is judged; declaring a deviation never passes the gate either. If you cannot complete the task without an out-of-scope edit, stop and report ok:false explaining why in "summary". List any out-of-scope paths you did touch, each with a reason, in "deviations" (journaled for the operator's audit).
|
|
26
|
+
- No background process may outlive the worker, and no suite may run beside another.
|
|
26
27
|
- Do not ask questions; you are unattended. Make the smallest correct change.
|
|
27
28
|
${feedback ? `\n## Previous attempt failed gates — fix these specifically\n${feedback}\n` : ""}
|
|
28
29
|
When finished, end your final message with exactly one line (no code fence):
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import { type ClassifiedWorkerResult } from "./prompt.js";
|
|
2
|
+
import { type WorkerAdapter } from "./types.js";
|
|
3
|
+
export declare const QWEN_VERSION_IDENTITY: RegExp;
|
|
4
|
+
export declare function parseQwenResult(raw: string, nonce: string): ClassifiedWorkerResult;
|
|
5
|
+
export declare const qwen: WorkerAdapter;
|