tickmarkr 1.90.8 → 1.91.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/catalog-remote.d.ts +11 -1
- package/dist/adapters/catalog-remote.js +169 -19
- package/dist/adapters/catalog.js +59 -0
- package/dist/adapters/model-lints.d.ts +36 -2
- package/dist/adapters/model-lints.js +151 -35
- package/dist/adapters/registry.js +32 -6
- package/dist/cli/commands/doctor.d.ts +12 -0
- package/dist/cli/commands/doctor.js +57 -15
- package/dist/cli/commands/fleet.js +31 -3
- package/dist/compile/native.js +5 -4
- package/dist/config/config.js +5 -0
- package/dist/config/fleet-overlay.js +20 -4
- package/dist/gates/acceptance.js +13 -4
- package/dist/run/daemon.js +1 -1
- package/dist/tui/ink/components.d.ts +16 -1
- package/dist/tui/ink/components.js +16 -2
- package/dist/tui/ink/fleet-app.d.ts +8 -1
- package/dist/tui/ink/fleet-app.js +184 -54
- package/package.json +1 -1
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
export declare const MODELS_DEV_CATALOG_URL = "https://models.dev/api.json";
|
|
2
|
-
export declare const ARTIFICIAL_ANALYSIS_CATALOG_URL = "https://artificialanalysis.ai/api/v2/
|
|
2
|
+
export declare const ARTIFICIAL_ANALYSIS_CATALOG_URL = "https://artificialanalysis.ai/api/v2/data/llms/models";
|
|
3
|
+
export declare const LIVEBENCH_TABLE_DATE = "2026_06_25";
|
|
4
|
+
export declare const LIVEBENCH_TABLE_URL = "https://livebench.ai/table_2026_06_25.csv";
|
|
5
|
+
export declare const LIVEBENCH_CATEGORIES_URL = "https://livebench.ai/categories_2026_06_25.json";
|
|
3
6
|
export declare const CATALOG_CACHE_MAX_AGE_MS: number;
|
|
4
7
|
export declare const CATALOG_REFRESH_TIMEOUT_MS = 10000;
|
|
5
8
|
export interface CatalogCache {
|
|
@@ -7,6 +10,7 @@ export interface CatalogCache {
|
|
|
7
10
|
fetchedAt: string;
|
|
8
11
|
modelsDev: unknown;
|
|
9
12
|
artificialAnalysis?: unknown;
|
|
13
|
+
liveBench?: unknown;
|
|
10
14
|
}
|
|
11
15
|
export interface CatalogReadResult {
|
|
12
16
|
catalog: CatalogCache;
|
|
@@ -16,6 +20,7 @@ export interface CatalogReadResult {
|
|
|
16
20
|
}
|
|
17
21
|
export interface CatalogModelEvidence {
|
|
18
22
|
modelId: string;
|
|
23
|
+
catalogId: string;
|
|
19
24
|
inputCostPerMtok?: number;
|
|
20
25
|
outputCostPerMtok?: number;
|
|
21
26
|
contextWindow?: number;
|
|
@@ -23,11 +28,16 @@ export interface CatalogModelEvidence {
|
|
|
23
28
|
features: string[];
|
|
24
29
|
intelligenceIndex?: number;
|
|
25
30
|
intelligenceIndexVersion?: string;
|
|
31
|
+
codingIndex?: number;
|
|
32
|
+
agenticCodingScore?: number;
|
|
33
|
+
codingScore?: number;
|
|
34
|
+
evidenceDate?: string;
|
|
26
35
|
}
|
|
27
36
|
type CatalogResponse = {
|
|
28
37
|
ok: boolean;
|
|
29
38
|
status: number;
|
|
30
39
|
json: () => Promise<unknown>;
|
|
40
|
+
text: () => Promise<string>;
|
|
31
41
|
};
|
|
32
42
|
export type CatalogFetcher = (input: string | URL | Request, init?: RequestInit) => Promise<CatalogResponse>;
|
|
33
43
|
export interface RefreshCatalogOptions {
|
|
@@ -2,7 +2,13 @@ import { randomUUID } from "node:crypto";
|
|
|
2
2
|
import { mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
|
|
3
3
|
import { dirname, join } from "node:path";
|
|
4
4
|
export const MODELS_DEV_CATALOG_URL = "https://models.dev/api.json";
|
|
5
|
-
export const ARTIFICIAL_ANALYSIS_CATALOG_URL = "https://artificialanalysis.ai/api/v2/
|
|
5
|
+
export const ARTIFICIAL_ANALYSIS_CATALOG_URL = "https://artificialanalysis.ai/api/v2/data/llms/models";
|
|
6
|
+
// LiveBench ships one dated CSV per question-set release plus the category->task map that defines
|
|
7
|
+
// the aggregates. Fetch the deployed site, NEVER raw.githubusercontent.com: same filename, 15 fewer
|
|
8
|
+
// models on raw. ponytail: hand-bumped constant; doctor lints its age against the release listing.
|
|
9
|
+
export const LIVEBENCH_TABLE_DATE = "2026_06_25";
|
|
10
|
+
export const LIVEBENCH_TABLE_URL = `https://livebench.ai/table_${LIVEBENCH_TABLE_DATE}.csv`;
|
|
11
|
+
export const LIVEBENCH_CATEGORIES_URL = `https://livebench.ai/categories_${LIVEBENCH_TABLE_DATE}.json`;
|
|
6
12
|
export const CATALOG_CACHE_MAX_AGE_MS = 30 * 86_400_000;
|
|
7
13
|
export const CATALOG_REFRESH_TIMEOUT_MS = 10_000;
|
|
8
14
|
const ARTIFICIAL_ANALYSIS_PAGE_SIZE = 100;
|
|
@@ -94,28 +100,31 @@ function providerModels(modelsDev, preferred) {
|
|
|
94
100
|
// no provider at all) used to ZERO the search space and blanket-uncover the whole adapter.
|
|
95
101
|
// Fail open to the full scan instead — advisory evidence with a visible models.dev id beats none.
|
|
96
102
|
const selected = preferred && preferredEntries.length > 0 ? preferredEntries : entries;
|
|
97
|
-
return selected
|
|
98
|
-
|
|
99
|
-
|
|
103
|
+
return selected.flatMap(([providerKey, provider]) => {
|
|
104
|
+
const models = record(record(provider)?.models);
|
|
105
|
+
return models ? [{ providerKey, models }] : [];
|
|
106
|
+
});
|
|
100
107
|
}
|
|
101
108
|
function findModelsDevModel(catalog, provider, modelId) {
|
|
102
109
|
// CLI namespaces prefix their catalog ids (kimi-code/k3 vs catalog key k3; omp's openai/gpt-4):
|
|
103
110
|
// after exact key/id misses, retry with the bare segment after the last "/". Deterministic
|
|
104
111
|
// provider order; first hit wins — acceptable for advisory evidence, never routing.
|
|
105
112
|
const bare = modelId.includes("/") ? modelId.slice(modelId.lastIndexOf("/") + 1) : undefined;
|
|
106
|
-
for (const models of providerModels(catalog.modelsDev, provider)) {
|
|
113
|
+
for (const { providerKey, models } of providerModels(catalog.modelsDev, provider)) {
|
|
107
114
|
const direct = record(models[modelId]);
|
|
108
115
|
if (direct)
|
|
109
|
-
return direct;
|
|
110
|
-
const
|
|
111
|
-
|
|
112
|
-
|
|
116
|
+
return { providerKey, models, recordId: modelId, model: direct };
|
|
117
|
+
for (const [recordId, value] of Object.entries(models)) {
|
|
118
|
+
const candidate = record(value);
|
|
119
|
+
if (candidate?.id === modelId)
|
|
120
|
+
return { providerKey, models, recordId, model: candidate };
|
|
121
|
+
}
|
|
113
122
|
}
|
|
114
123
|
if (bare) {
|
|
115
|
-
for (const models of providerModels(catalog.modelsDev, provider)) {
|
|
124
|
+
for (const { providerKey, models } of providerModels(catalog.modelsDev, provider)) {
|
|
116
125
|
const direct = record(models[bare]);
|
|
117
126
|
if (direct)
|
|
118
|
-
return direct;
|
|
127
|
+
return { providerKey, models, recordId: bare, model: direct };
|
|
119
128
|
}
|
|
120
129
|
}
|
|
121
130
|
return undefined;
|
|
@@ -160,27 +169,129 @@ function artificialAnalysisIndex(value, identities, provider) {
|
|
|
160
169
|
const n = nonNegative(candidate);
|
|
161
170
|
if (n !== undefined) {
|
|
162
171
|
const version = root?.intelligence_index_version;
|
|
172
|
+
const coding = [
|
|
173
|
+
evaluations?.artificial_analysis_coding_index,
|
|
174
|
+
row.artificial_analysis_coding_index,
|
|
175
|
+
].map(nonNegative).find((value) => value !== undefined);
|
|
163
176
|
return {
|
|
164
177
|
index: n,
|
|
165
178
|
...((typeof version === "string" || typeof version === "number") ? { version: String(version) } : {}),
|
|
179
|
+
...(coding !== undefined ? { codingIndex: coding } : {}),
|
|
166
180
|
};
|
|
167
181
|
}
|
|
168
182
|
}
|
|
169
183
|
}
|
|
170
184
|
return undefined;
|
|
171
185
|
}
|
|
186
|
+
/**
|
|
187
|
+
* LiveBench ships the table as CSV and the category->task map as JSON. Every aggregate is derived
|
|
188
|
+
* from that map, never from a task-column list written here: a category reshuffle upstream must
|
|
189
|
+
* re-score us, not silently mis-score us.
|
|
190
|
+
*/
|
|
191
|
+
function parseLiveBenchTable(csv) {
|
|
192
|
+
const lines = csv.split(/\r?\n/).filter((line) => line.trim().length > 0);
|
|
193
|
+
const header = lines.shift()?.split(",").map((cell) => cell.trim());
|
|
194
|
+
if (header?.[0] !== "model")
|
|
195
|
+
return [];
|
|
196
|
+
return lines.flatMap((line) => {
|
|
197
|
+
const cells = line.split(",").map((cell) => cell.trim());
|
|
198
|
+
const model = cells[0];
|
|
199
|
+
if (!model)
|
|
200
|
+
return [];
|
|
201
|
+
const row = { model };
|
|
202
|
+
header.forEach((column, i) => {
|
|
203
|
+
const score = i > 0 && cells[i] !== "" ? finite(Number(cells[i])) : undefined;
|
|
204
|
+
if (score !== undefined)
|
|
205
|
+
row[column] = score;
|
|
206
|
+
});
|
|
207
|
+
// A scoreless row (truncated payload, blank cells) carries no evidence — it is not a fleet member.
|
|
208
|
+
return Object.keys(row).length > 1 ? [row] : [];
|
|
209
|
+
});
|
|
210
|
+
}
|
|
211
|
+
/**
|
|
212
|
+
* A probe is usable only if it can actually score the two categories we read: both must be arrays
|
|
213
|
+
* naming at least one task column the table really scored. Otherwise throw — an unusable probe must
|
|
214
|
+
* never overwrite the cached section (a truncated table plus `{}` categories is a failed fetch).
|
|
215
|
+
*/
|
|
216
|
+
function assertUsableLiveBench(rows, categories) {
|
|
217
|
+
const scored = new Set(rows.flatMap((row) => Object.keys(row).filter((key) => key !== "model")));
|
|
218
|
+
for (const name of ["Agentic Coding", "Coding"]) {
|
|
219
|
+
const tasks = categories[name];
|
|
220
|
+
if (!Array.isArray(tasks) || !tasks.some((task) => typeof task === "string" && scored.has(task))) {
|
|
221
|
+
throw new Error(`LiveBench categories name no scored "${name}" task`);
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
// LiveBench splits one model across reasoning-effort rows (`-max-effort`, `-xhigh`,
|
|
226
|
+
// `-thinking-auto-medium-effort`); the highest effort is the model at its best. Rank by
|
|
227
|
+
// hyphen-delimited token so `xhigh` never reads as `high`.
|
|
228
|
+
const LIVEBENCH_EFFORT_RANK = { max: 5, xhigh: 4, high: 3, medium: 2, low: 1 };
|
|
229
|
+
const liveBenchEffortRank = (residue) => residue.split("-").reduce((rank, token) => Math.max(rank, LIVEBENCH_EFFORT_RANK[token] ?? 0), 0);
|
|
230
|
+
const liveBenchCategoryMean = (row, tasks) => {
|
|
231
|
+
const scores = (Array.isArray(tasks) ? tasks : [])
|
|
232
|
+
.map((task) => (typeof task === "string" ? finite(row[task]) : undefined))
|
|
233
|
+
.filter((score) => score !== undefined);
|
|
234
|
+
return scores.length > 0 ? scores.reduce((sum, score) => sum + score, 0) / scores.length : undefined;
|
|
235
|
+
};
|
|
236
|
+
/** LiveBench hyphenates where models.dev spaces: `Kimi K3` is `kimi-k3` in the table. */
|
|
237
|
+
const liveBenchIdentity = (value) => value.trim().toLowerCase().replace(/\s+/g, "-");
|
|
238
|
+
function liveBenchIndex(value, identities) {
|
|
239
|
+
const root = record(value);
|
|
240
|
+
if (!root || !Array.isArray(root.rows))
|
|
241
|
+
return undefined;
|
|
242
|
+
const wanted = identities.map(liveBenchIdentity).filter(Boolean);
|
|
243
|
+
let best;
|
|
244
|
+
for (const raw of root.rows) {
|
|
245
|
+
const row = record(raw);
|
|
246
|
+
const model = typeof row?.model === "string" ? liveBenchIdentity(row.model) : undefined;
|
|
247
|
+
if (!row || !model)
|
|
248
|
+
continue;
|
|
249
|
+
// Bare startsWith is a false-positive machine: `glm-5` would claim `glm-5.2`. The residue after
|
|
250
|
+
// the fleet id must be empty or an effort suffix.
|
|
251
|
+
const residue = wanted.map((identity) => model.startsWith(identity) ? model.slice(identity.length) : undefined)
|
|
252
|
+
.find((rest) => rest === "" || rest?.startsWith("-"));
|
|
253
|
+
if (residue === undefined)
|
|
254
|
+
continue;
|
|
255
|
+
const rank = liveBenchEffortRank(residue);
|
|
256
|
+
if (!best || rank > best.rank)
|
|
257
|
+
best = { row, rank };
|
|
258
|
+
}
|
|
259
|
+
if (!best)
|
|
260
|
+
return undefined;
|
|
261
|
+
const categories = record(root.categories);
|
|
262
|
+
const agenticCodingScore = liveBenchCategoryMean(best.row, categories?.["Agentic Coding"]);
|
|
263
|
+
const codingScore = liveBenchCategoryMean(best.row, categories?.["Coding"]);
|
|
264
|
+
return {
|
|
265
|
+
...(agenticCodingScore !== undefined ? { agenticCodingScore } : {}),
|
|
266
|
+
...(codingScore !== undefined ? { codingScore } : {}),
|
|
267
|
+
...(typeof root.tableDate === "string" ? { tableDate: root.tableDate } : {}),
|
|
268
|
+
};
|
|
269
|
+
}
|
|
172
270
|
/** Resolve the alias identity first; the floating alias is only a fallback when identity is unknown. */
|
|
173
271
|
export function resolveCatalogModel(catalog, query) {
|
|
174
272
|
const modelId = query.resolvedModel ?? query.model;
|
|
175
|
-
const
|
|
176
|
-
if (!
|
|
273
|
+
const match = findModelsDevModel(catalog, query.provider, modelId);
|
|
274
|
+
if (!match)
|
|
177
275
|
return undefined;
|
|
276
|
+
const { model } = match;
|
|
277
|
+
const catalogId = `${match.providerKey}/${match.recordId}`;
|
|
178
278
|
const cost = record(model.cost);
|
|
179
279
|
const limit = record(model.limit);
|
|
280
|
+
// LiveBench spells rows with the vendor-prefixed model name (`k3` is `kimi-k3` there), so
|
|
281
|
+
// models.dev `name` bridges a fleet id that is only the bare suffix. `family` must NEVER join
|
|
282
|
+
// this list: it is a LINEAGE label many models share — real models.dev has `glm` on every GLM and
|
|
283
|
+
// `gpt` on gpt-4o — so through it glm-5 would claim glm-5.2's row and an unbenchmarked gpt-4o
|
|
284
|
+
// would inherit a gpt-5.6 row's scores and evidenceDate. `name` is per-model; `family` is not.
|
|
285
|
+
const identities = [
|
|
286
|
+
modelId,
|
|
287
|
+
query.model,
|
|
288
|
+
...(typeof model.name === "string" ? [model.name] : []),
|
|
289
|
+
];
|
|
180
290
|
const intelligence = artificialAnalysisIndex(catalog.artificialAnalysis, [
|
|
181
291
|
modelId,
|
|
182
292
|
...(typeof model.name === "string" ? [model.name] : []),
|
|
183
293
|
], query.provider);
|
|
294
|
+
const liveBench = liveBenchIndex(catalog.liveBench, identities);
|
|
184
295
|
const features = [
|
|
185
296
|
["reasoning", model.reasoning],
|
|
186
297
|
["tool-call", model.tool_call],
|
|
@@ -190,6 +301,7 @@ export function resolveCatalogModel(catalog, query) {
|
|
|
190
301
|
].filter((entry) => entry[1] === true).map((entry) => entry[0]);
|
|
191
302
|
return {
|
|
192
303
|
modelId,
|
|
304
|
+
catalogId,
|
|
193
305
|
...(nonNegative(cost?.input) !== undefined ? { inputCostPerMtok: nonNegative(cost?.input) } : {}),
|
|
194
306
|
...(nonNegative(cost?.output) !== undefined ? { outputCostPerMtok: nonNegative(cost?.output) } : {}),
|
|
195
307
|
...(positive(limit?.context) !== undefined ? { contextWindow: positive(limit?.context) } : {}),
|
|
@@ -198,10 +310,19 @@ export function resolveCatalogModel(catalog, query) {
|
|
|
198
310
|
...(intelligence !== undefined ? {
|
|
199
311
|
intelligenceIndex: intelligence.index,
|
|
200
312
|
...(intelligence.version ? { intelligenceIndexVersion: intelligence.version } : {}),
|
|
313
|
+
...(intelligence.codingIndex !== undefined ? { codingIndex: intelligence.codingIndex } : {}),
|
|
314
|
+
} : {}),
|
|
315
|
+
...(liveBench !== undefined ? {
|
|
316
|
+
...(liveBench.agenticCodingScore !== undefined ? { agenticCodingScore: liveBench.agenticCodingScore } : {}),
|
|
317
|
+
...(liveBench.codingScore !== undefined ? { codingScore: liveBench.codingScore } : {}),
|
|
318
|
+
...(liveBench.tableDate !== undefined ? { evidenceDate: liveBench.tableDate } : {}),
|
|
201
319
|
} : {}),
|
|
202
320
|
};
|
|
203
321
|
}
|
|
204
|
-
|
|
322
|
+
const catalogSourceLabel = (url) => url.startsWith(MODELS_DEV_CATALOG_URL) ? "models.dev"
|
|
323
|
+
: url.startsWith(LIVEBENCH_TABLE_URL) || url.startsWith(LIVEBENCH_CATEGORIES_URL) ? "LiveBench"
|
|
324
|
+
: "Artificial Analysis";
|
|
325
|
+
async function fetchCatalog(fetcher, url, init, timeoutMs, read) {
|
|
205
326
|
const controller = new AbortController();
|
|
206
327
|
let timer;
|
|
207
328
|
const timeout = new Promise((_resolve, reject) => {
|
|
@@ -215,8 +336,8 @@ async function fetchJson(fetcher, url, init, timeoutMs) {
|
|
|
215
336
|
const request = (async () => {
|
|
216
337
|
const response = await fetcher(url, { ...init, signal: controller.signal });
|
|
217
338
|
if (!response.ok)
|
|
218
|
-
throw new Error(`${url
|
|
219
|
-
return response
|
|
339
|
+
throw new Error(`${catalogSourceLabel(url)} HTTP ${response.status}`);
|
|
340
|
+
return read(response);
|
|
220
341
|
})();
|
|
221
342
|
return await Promise.race([request, timeout]);
|
|
222
343
|
}
|
|
@@ -245,7 +366,7 @@ async function fetchArtificialAnalysis(fetcher, apiKey, timeoutMs) {
|
|
|
245
366
|
let intelligenceIndexVersion;
|
|
246
367
|
for (let page = 1; page <= ARTIFICIAL_ANALYSIS_MAX_PAGES; page++) {
|
|
247
368
|
const url = `${ARTIFICIAL_ANALYSIS_CATALOG_URL}?page=${page}&page_size=${ARTIFICIAL_ANALYSIS_PAGE_SIZE}`;
|
|
248
|
-
const value = await
|
|
369
|
+
const value = await fetchCatalog(fetcher, url, { headers: { "x-api-key": apiKey } }, timeoutMs, (r) => r.json());
|
|
249
370
|
const root = record(value);
|
|
250
371
|
if (!root || !Array.isArray(root.data))
|
|
251
372
|
throw new Error("Artificial Analysis catalog schema is invalid");
|
|
@@ -266,6 +387,19 @@ async function fetchArtificialAnalysis(fetcher, apiKey, timeoutMs) {
|
|
|
266
387
|
}
|
|
267
388
|
throw new Error(`Artificial Analysis catalog exceeds ${ARTIFICIAL_ANALYSIS_MAX_PAGES} pages`);
|
|
268
389
|
}
|
|
390
|
+
/** Keyless leg: one CSV plus the category map. Both URLs are pinned to the deployed livebench.ai. */
|
|
391
|
+
async function fetchLiveBench(fetcher, timeoutMs) {
|
|
392
|
+
const csv = await fetchCatalog(fetcher, LIVEBENCH_TABLE_URL, {}, timeoutMs, (r) => r.text());
|
|
393
|
+
const categories = record(await fetchCatalog(fetcher, LIVEBENCH_CATEGORIES_URL, {}, timeoutMs, (r) => r.json()));
|
|
394
|
+
if (!categories)
|
|
395
|
+
throw new Error("LiveBench categories schema is invalid");
|
|
396
|
+
const rows = parseLiveBenchTable(csv);
|
|
397
|
+
// Orca import #3, write site: an empty probe is a failed probe, never an empty fleet.
|
|
398
|
+
if (rows.length === 0)
|
|
399
|
+
throw new Error("LiveBench table parsed to zero rows");
|
|
400
|
+
assertUsableLiveBench(rows, categories);
|
|
401
|
+
return { tableDate: LIVEBENCH_TABLE_DATE, categories, rows };
|
|
402
|
+
}
|
|
269
403
|
/**
|
|
270
404
|
* The named, explicit refresh path. No other function in this module can reach fetch.
|
|
271
405
|
* A failed refresh preserves the previous cache byte-for-byte and returns it fail-open.
|
|
@@ -276,21 +410,37 @@ export async function refreshCatalogCommand(opts) {
|
|
|
276
410
|
try {
|
|
277
411
|
const fetcher = opts.fetcher ?? globalThis.fetch.bind(globalThis);
|
|
278
412
|
const timeoutMs = opts.timeoutMs ?? CATALOG_REFRESH_TIMEOUT_MS;
|
|
279
|
-
const modelsDev = await
|
|
413
|
+
const modelsDev = await fetchCatalog(fetcher, MODELS_DEV_CATALOG_URL, {}, timeoutMs, (r) => r.json());
|
|
280
414
|
if (!validModelsDevCatalog(modelsDev))
|
|
281
415
|
throw new Error("models.dev catalog schema is invalid");
|
|
282
416
|
const apiKey = opts.artificialAnalysisKey ?? process.env.ARTIFICIAL_ANALYSIS_API_KEY?.trim();
|
|
283
417
|
const artificialAnalysis = apiKey
|
|
284
418
|
? await fetchArtificialAnalysis(fetcher, apiKey, timeoutMs)
|
|
285
419
|
: undefined;
|
|
420
|
+
// The LiveBench leg is keyless and never costs the models.dev refresh: a failure keeps the
|
|
421
|
+
// previous section verbatim and names the leg in the warning.
|
|
422
|
+
let liveBench = current.catalog.liveBench;
|
|
423
|
+
let warning;
|
|
424
|
+
try {
|
|
425
|
+
liveBench = await fetchLiveBench(fetcher, timeoutMs);
|
|
426
|
+
}
|
|
427
|
+
catch (error) {
|
|
428
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
429
|
+
warning = message.startsWith("LiveBench") ? message : `LiveBench refresh failed: ${message}`;
|
|
430
|
+
}
|
|
286
431
|
const catalog = {
|
|
287
432
|
schemaVersion: 1,
|
|
288
433
|
fetchedAt: now().toISOString(),
|
|
289
434
|
modelsDev,
|
|
290
435
|
...(artificialAnalysis !== undefined ? { artificialAnalysis } : {}),
|
|
436
|
+
...(liveBench !== undefined ? { liveBench } : {}),
|
|
291
437
|
};
|
|
292
438
|
writeCatalogCache(opts.repoRoot, catalog);
|
|
293
|
-
return {
|
|
439
|
+
return {
|
|
440
|
+
updated: true,
|
|
441
|
+
catalog: readCachedCatalog(opts.repoRoot, { now }),
|
|
442
|
+
...(warning !== undefined ? { warning } : {}),
|
|
443
|
+
};
|
|
294
444
|
}
|
|
295
445
|
catch (error) {
|
|
296
446
|
return {
|
package/dist/adapters/catalog.js
CHANGED
|
@@ -147,6 +147,65 @@ export const CLI_CATALOG = [
|
|
|
147
147
|
listModels: { argv: ["models", "ls", "--json"], parser: "json", path: "models", field: "selector" },
|
|
148
148
|
},
|
|
149
149
|
},
|
|
150
|
+
{
|
|
151
|
+
id: "agy",
|
|
152
|
+
binary: "agy",
|
|
153
|
+
// PROBE-agy-v190.md, recorded by hand 2026-08-13: `agy --version` → `1.1.12`. A bare semver
|
|
154
|
+
// banner, so the identity pins the shape, prefix-matched to survive patch bumps.
|
|
155
|
+
identity: "^\\d+\\.\\d+\\.\\d+",
|
|
156
|
+
// agy (Antigravity) is a multi-provider gateway: its model list spans google, anthropic and
|
|
157
|
+
// openai ids, so a selected model carries its real vendor — same posture as omp.
|
|
158
|
+
vendor: "mixed",
|
|
159
|
+
drive: {
|
|
160
|
+
// PROBE-agy-v190.md: three recorded traps shape every byte of this template. (1) `-p`
|
|
161
|
+
// consumes the NEXT argv as the prompt, so every flag precedes it. (2) tool writes land in
|
|
162
|
+
// cwd ONLY when cwd is workspace-added by ABSOLUTE path — without `--add-dir "$PWD"` (and
|
|
163
|
+
// with a relative `.`) the write is silently redirected to ~/.gemini/antigravity-cli/scratch
|
|
164
|
+
// while the model still claims DONE; dispatch runs through bash -lc, so the literal "$PWD"
|
|
165
|
+
// expands to the worktree at runtime. (3) --print-timeout defaults to 5m0s, which would kill
|
|
166
|
+
// any real worker task. --dangerously-skip-permissions: a permission prompt in print mode
|
|
167
|
+
// has no answerer.
|
|
168
|
+
headless: `agy --add-dir "$PWD" --dangerously-skip-permissions --print-timeout 240m --model {model} -p "$(cat {promptFile})"`,
|
|
169
|
+
// No interactive probe is recorded (PROBE-agy-v190.md names the falsifier) — null keeps the
|
|
170
|
+
// claim honest and the print fallback drives visible panes.
|
|
171
|
+
interactive: null,
|
|
172
|
+
trustDialog: {
|
|
173
|
+
kind: "none",
|
|
174
|
+
reason: "agy 1.1.12 rendered no workspace-trust prompt across five fresh-temp-repo print-mode probes (PROBE-agy-v190.md, 2026-08-13); interactive is null, so no dialog can reach a pane",
|
|
175
|
+
},
|
|
176
|
+
// `agy models` emits `id<TAB>label` rows behind a fetch banner — no parser projects that
|
|
177
|
+
// shape without admitting label words, so the list surface stays undeclared and doctor
|
|
178
|
+
// reports "no model-list surface" (claude-code posture).
|
|
179
|
+
},
|
|
180
|
+
},
|
|
181
|
+
{
|
|
182
|
+
id: "prime-agent",
|
|
183
|
+
binary: "prime-agent",
|
|
184
|
+
// PROBE-prime-agent-v190.md, recorded by hand 2026-08-13: `prime-agent --version` → `0.7.1`.
|
|
185
|
+
// Bare semver banner — shape-pinned, prefix-matched (agy precedent).
|
|
186
|
+
identity: "^\\d+\\.\\d+\\.\\d+",
|
|
187
|
+
// Multi-provider gateway (anthropic, google, openai model list) — the joined provider/model
|
|
188
|
+
// id carries the real vendor, same posture as omp and agy.
|
|
189
|
+
vendor: "mixed",
|
|
190
|
+
drive: {
|
|
191
|
+
// PROBE-prime-agent-v190.md: print mode executes tools unattended in cwd with NO approval
|
|
192
|
+
// flag needed (hello.txt probe: file in cwd, exit 0, 5.6s) and no trust prompt in fresh
|
|
193
|
+
// repos. `--model` accepts the joined provider/model form the listModels contract emits
|
|
194
|
+
// (verified: --model google/gemini-3.6-flash). Prompt rides "$(cat …)" — the `@file`
|
|
195
|
+
// attach syntax exists but its semantics were not probed, so the contract does not use it.
|
|
196
|
+
headless: `prime-agent -p --model {model} "$(cat {promptFile})"`,
|
|
197
|
+
// No interactive probe is recorded (PROBE-prime-agent-v190.md names the falsifier) — null
|
|
198
|
+
// keeps the claim honest and the print fallback drives visible panes.
|
|
199
|
+
interactive: null,
|
|
200
|
+
trustDialog: {
|
|
201
|
+
kind: "none",
|
|
202
|
+
reason: "prime-agent 0.7.1 rendered no workspace-trust or tool-approval prompt across three fresh-temp-repo print-mode probes (PROBE-prime-agent-v190.md, 2026-08-13); interactive is null, so no dialog can reach a pane",
|
|
203
|
+
},
|
|
204
|
+
// `prime-agent model list` is the pi-table shape: header `provider model …`, first two
|
|
205
|
+
// columns join to the id `--model` accepts (live-verified 2026-08-13).
|
|
206
|
+
listModels: { argv: ["model", "list"], parser: "pi-table" },
|
|
207
|
+
},
|
|
208
|
+
},
|
|
150
209
|
].flatMap((entry) => typeof entry === "string"
|
|
151
210
|
? [{ id: entry, binary: entry, identity: ".+", vendor: null }]
|
|
152
211
|
: [entry]);
|
|
@@ -4,19 +4,52 @@ import { type AuthHealth, type WorkerAdapter } from "./types.js";
|
|
|
4
4
|
import { type CatalogModelEvidence, type CatalogReadResult } from "./catalog-remote.js";
|
|
5
5
|
export declare const SEED_STAMPED = "2026-07-09";
|
|
6
6
|
export declare const MODEL_STALE_DAYS = 30;
|
|
7
|
+
/**
|
|
8
|
+
* Operator directive 2026-08-13 ("we should exclude all retired models"): the fleet models
|
|
9
|
+
* screen hides these classes BY DEFAULT — omp alone reports 218 ids, most of them dated
|
|
10
|
+
* snapshots, previews, and SKUs that can never carry a worker. Shape-based on purpose: a
|
|
11
|
+
* knowledge list of retired families rots, a suffix grammar does not. Hidden is never gone —
|
|
12
|
+
* the screen counts what it hid and one key reveals it, and CLASSIFIED models are never hidden
|
|
13
|
+
* regardless of shape (an operator who tiered a dated snapshot meant it).
|
|
14
|
+
*/
|
|
15
|
+
export declare function retiredModelReason(model: string): "dated snapshot" | "preview" | "non-worker" | "legacy family" | null;
|
|
7
16
|
export declare const ttyVisual: () => boolean;
|
|
17
|
+
/** Bases that band FLEET-RELATIVELY, most-trusted first (2026-08-13 ranking-sites assessment §4.3). */
|
|
18
|
+
declare const RANKED_BASES: readonly ["agentic-coding", "intelligence"];
|
|
19
|
+
type RankedBasis = typeof RANKED_BASES[number];
|
|
20
|
+
export type CatalogSuggestionBasis = RankedBasis | "price";
|
|
8
21
|
export interface CatalogModelAdvisory {
|
|
9
22
|
coverage: "covered" | "uncovered";
|
|
10
23
|
evidence?: CatalogModelEvidence;
|
|
11
24
|
suggestion?: {
|
|
12
25
|
tier: Tier;
|
|
13
26
|
kind: "inference";
|
|
14
|
-
basis:
|
|
27
|
+
basis: CatalogSuggestionBasis;
|
|
15
28
|
provenanceNote: string;
|
|
16
29
|
};
|
|
17
30
|
display: string;
|
|
18
31
|
}
|
|
19
|
-
|
|
32
|
+
/** A (adapter, model) pair under advisory at a call site; it joins that call's ranking universe. */
|
|
33
|
+
export interface CatalogAdvisoryRow {
|
|
34
|
+
adapter: string;
|
|
35
|
+
model: string;
|
|
36
|
+
resolvedModel?: string;
|
|
37
|
+
}
|
|
38
|
+
/** Per-basis descending scores; a basis below MIN_RANKED_MODELS never appears, so it yields. */
|
|
39
|
+
export type CatalogTierRanking = ReadonlyArray<{
|
|
40
|
+
basis: RankedBasis;
|
|
41
|
+
scores: number[];
|
|
42
|
+
}>;
|
|
43
|
+
/**
|
|
44
|
+
* The ranking universe: configured tier models ∪ the rows under advisory at THIS call site,
|
|
45
|
+
* restricted per basis to those that resolve it. Build it once per call — a band that depended on
|
|
46
|
+
* which adapter happened to be iterated first would not be a band. `resolvedModel` is the SAME
|
|
47
|
+
* resolver the call site hands its advisory rows: a configured claude alias (`opus`) is absent from
|
|
48
|
+
* models.dev under that spelling, so without it the fleet's own frontier models silently drop out
|
|
49
|
+
* of the universe they are supposed to anchor.
|
|
50
|
+
*/
|
|
51
|
+
export declare function catalogTierRanking(cfg: TickmarkrConfig, catalog: CatalogReadResult, rows?: readonly CatalogAdvisoryRow[], resolvedModel?: (adapter: string, model: string) => string | undefined): CatalogTierRanking;
|
|
52
|
+
export declare function catalogModelAdvisory(cfg: TickmarkrConfig, catalog: CatalogReadResult, adapter: string, model: string, resolvedModel?: string, ranking?: CatalogTierRanking): CatalogModelAdvisory;
|
|
20
53
|
/**
|
|
21
54
|
* Whether the doctor should add its optional window column. Keep non-TTY default output stable for
|
|
22
55
|
* machine consumers; an interactive seed-only matrix shows T14's fleet windows, and any explicit
|
|
@@ -55,3 +88,4 @@ export declare function fleetUnclassifiedModels(cfg: TickmarkrConfig, health: Re
|
|
|
55
88
|
model: string;
|
|
56
89
|
detectedAt?: string;
|
|
57
90
|
}[];
|
|
91
|
+
export {};
|