@phnx-labs/agents-cli 1.21.3 → 1.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +136 -0
- package/README.md +32 -3
- package/dist/bin/agents +0 -0
- package/dist/commands/computer-actions.js +1 -0
- package/dist/commands/doctor.js +34 -2
- package/dist/commands/exec.d.ts +27 -0
- package/dist/commands/exec.js +123 -6
- package/dist/commands/models.js +36 -1
- package/dist/commands/projects.js +120 -62
- package/dist/commands/sessions-backfill.d.ts +32 -0
- package/dist/commands/sessions-backfill.js +186 -0
- package/dist/commands/sessions.d.ts +17 -1
- package/dist/commands/sessions.js +317 -18
- package/dist/commands/teams.js +1 -1
- package/dist/commands/worktree.d.ts +3 -3
- package/dist/commands/worktree.js +35 -4
- package/dist/lib/app-bundle-install.d.ts +17 -0
- package/dist/lib/app-bundle-install.js +94 -0
- package/dist/lib/daemon.d.ts +5 -1
- package/dist/lib/daemon.js +63 -14
- package/dist/lib/devices/doctor-findings.js +12 -4
- package/dist/lib/devices/doctor-overview-cache.d.ts +45 -0
- package/dist/lib/devices/doctor-overview-cache.js +168 -0
- package/dist/lib/devices/fleet.js +7 -2
- package/dist/lib/devices/resolve-target.d.ts +6 -0
- package/dist/lib/devices/resolve-target.js +9 -3
- package/dist/lib/devices/self-host.d.ts +9 -0
- package/dist/lib/devices/self-host.js +61 -0
- package/dist/lib/exec.js +39 -8
- package/dist/lib/hosts/dispatch.d.ts +12 -0
- package/dist/lib/hosts/dispatch.js +23 -6
- package/dist/lib/hosts/passthrough.js +8 -6
- package/dist/lib/hosts/reconnect.d.ts +38 -0
- package/dist/lib/hosts/reconnect.js +85 -4
- package/dist/lib/hosts/run-target.js +14 -2
- package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
- package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
- package/dist/lib/menubar/install-menubar.js +27 -23
- package/dist/lib/model-tiers.d.ts +54 -0
- package/dist/lib/model-tiers.js +229 -0
- package/dist/lib/models.d.ts +3 -0
- package/dist/lib/models.js +44 -7
- package/dist/lib/pricing/prices.json +16 -1
- package/dist/lib/project-focus.d.ts +42 -0
- package/dist/lib/project-focus.js +80 -0
- package/dist/lib/project-schedule.d.ts +75 -0
- package/dist/lib/project-schedule.js +110 -0
- package/dist/lib/project-status.d.ts +7 -0
- package/dist/lib/project-status.js +9 -0
- package/dist/lib/projects.d.ts +11 -2
- package/dist/lib/projects.js +57 -9
- package/dist/lib/redact.d.ts +2 -0
- package/dist/lib/redact.js +22 -0
- package/dist/lib/remote-agents-json.d.ts +2 -0
- package/dist/lib/remote-agents-json.js +3 -3
- package/dist/lib/rotate.d.ts +84 -1
- package/dist/lib/rotate.js +155 -5
- package/dist/lib/runner.d.ts +4 -2
- package/dist/lib/runner.js +13 -4
- package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
- package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
- package/dist/lib/secrets/install-helper.js +28 -31
- package/dist/lib/session/bash-command.js +60 -9
- package/dist/lib/session/db.d.ts +7 -1
- package/dist/lib/session/db.js +301 -32
- package/dist/lib/session/discover.d.ts +40 -7
- package/dist/lib/session/discover.js +144 -83
- package/dist/lib/session/parse.d.ts +8 -1
- package/dist/lib/session/parse.js +83 -32
- package/dist/lib/session/remote-list.d.ts +71 -0
- package/dist/lib/session/remote-list.js +410 -2
- package/dist/lib/session/shell-programs.d.ts +15 -0
- package/dist/lib/session/shell-programs.js +359 -0
- package/dist/lib/session/tool-calls.d.ts +88 -0
- package/dist/lib/session/tool-calls.js +612 -0
- package/dist/lib/session/tool-index.d.ts +100 -0
- package/dist/lib/session/tool-index.js +773 -0
- package/dist/lib/session/tool-store.d.ts +15 -0
- package/dist/lib/session/tool-store.js +198 -0
- package/dist/lib/session/types.d.ts +7 -0
- package/dist/lib/state.d.ts +10 -1
- package/dist/lib/state.js +11 -2
- package/dist/lib/teams/remoteWorktree.d.ts +3 -4
- package/dist/lib/teams/remoteWorktree.js +3 -4
- package/dist/lib/teams/worktree.d.ts +11 -1
- package/dist/lib/teams/worktree.js +42 -4
- package/dist/lib/types.d.ts +17 -0
- package/dist/lib/types.js +17 -0
- package/dist/lib/versions.js +69 -22
- package/package.json +3 -1
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
import { getModelCatalog } from './models.js';
|
|
2
|
+
import { getModelPricing } from './pricing/index.js';
|
|
3
|
+
/** The four cross-harness cost tiers, cheapest -> most capable. */
|
|
4
|
+
export const MODEL_TIERS = ['cheap', 'default', 'best', 'ultra'];
|
|
5
|
+
/** True if `s` is one of the four tier tokens (not a concrete model id). */
|
|
6
|
+
export function isTierToken(s) {
|
|
7
|
+
return !!s && MODEL_TIERS.includes(s);
|
|
8
|
+
}
|
|
9
|
+
// --- single-model harnesses: the tier is reasoning effort, not a model ---------
|
|
10
|
+
const TIER_EFFORT = {
|
|
11
|
+
cheap: 'low',
|
|
12
|
+
default: 'medium',
|
|
13
|
+
best: 'high',
|
|
14
|
+
ultra: 'xhigh',
|
|
15
|
+
};
|
|
16
|
+
// --- Droid: no live catalog; prices in credit multipliers. Curated map, capped
|
|
17
|
+
// at 2x (no 4x models like Fable 5 / Fast modes). Ids are Factory -m values.
|
|
18
|
+
const DROID_TIERS = {
|
|
19
|
+
cheap: 'glm-5.2', // 0.55x (Droid Core)
|
|
20
|
+
default: 'kimi-k3', // 0.6x (Droid Core)
|
|
21
|
+
best: 'claude-opus-5', // 2x
|
|
22
|
+
ultra: 'claude-opus-5', // clamp to best; avoid 4x
|
|
23
|
+
};
|
|
24
|
+
/** Router / pseudo models that are not a concrete choice and never a tier target. */
|
|
25
|
+
const PSEUDO = /(^|[-/])(auto|auto-review|router|dynamic)([-/]|$)/i;
|
|
26
|
+
/** Effort / speed suffixes aggregator harnesses (Cursor) bake into ids. */
|
|
27
|
+
const AGGREGATOR_SUFFIX = /-(low|medium|high|xhigh|thinking|fast|reasoning)\b/gi;
|
|
28
|
+
/** Anthropic capability family -> rank (cheapest 0 -> dearest 3). */
|
|
29
|
+
function anthropicFamilyRank(id) {
|
|
30
|
+
if (/(^|[-/])claude-haiku|(^|[-/])haiku/.test(id))
|
|
31
|
+
return 0;
|
|
32
|
+
if (/claude-sonnet|(^|[-/])sonnet/.test(id))
|
|
33
|
+
return 1;
|
|
34
|
+
if (/claude-opus|(^|[-/])opus/.test(id))
|
|
35
|
+
return 2;
|
|
36
|
+
if (/claude-(fable|mythos)|(^|[-/])(fable|mythos)/.test(id))
|
|
37
|
+
return 3;
|
|
38
|
+
return null;
|
|
39
|
+
}
|
|
40
|
+
/** Rank from the provider's own description keywords (e.g. Codex Sol/Terra/Luna). */
|
|
41
|
+
function descriptionRank(desc) {
|
|
42
|
+
if (!desc)
|
|
43
|
+
return null;
|
|
44
|
+
const d = desc.toLowerCase();
|
|
45
|
+
if (/(fast|affordable|cost-efficient|small|cheap|mini|nano|lightweight|spark)/.test(d))
|
|
46
|
+
return 0;
|
|
47
|
+
if (/(balanced|everyday|strong|general)/.test(d))
|
|
48
|
+
return 1;
|
|
49
|
+
if (/(frontier|latest|flagship|most capable|complex|advanced|professional)/.test(d))
|
|
50
|
+
return 2;
|
|
51
|
+
return null;
|
|
52
|
+
}
|
|
53
|
+
/** Last-resort ordering from size tokens embedded in the id. */
|
|
54
|
+
function sizeTokenRank(id) {
|
|
55
|
+
if (/(nano|mini|lite|flash|highspeed|small|air|spark)/.test(id))
|
|
56
|
+
return 0;
|
|
57
|
+
if (/(pro|max|ultra|opus|sol|large|frontier|heavy|thinking)/.test(id))
|
|
58
|
+
return 2;
|
|
59
|
+
return 1;
|
|
60
|
+
}
|
|
61
|
+
const blended = (id) => {
|
|
62
|
+
const p = getModelPricing(id);
|
|
63
|
+
return p ? p.inputPerToken + p.outputPerToken : null;
|
|
64
|
+
};
|
|
65
|
+
/** Strip an aggregator's effort/speed suffixes down to a base provider id. */
|
|
66
|
+
function normalizeAggregatorId(id) {
|
|
67
|
+
return id.replace(AGGREGATOR_SUFFIX, '').replace(/-+$/, '');
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* Compare two concrete ids so the newest wins within a family. Strips a trailing
|
|
71
|
+
* date stamp (`-20251101`), rebuild marker (`-v1`), and `-fast` first, so a dated
|
|
72
|
+
* `opus-4-5-20251101` doesn't out-rank the genuinely newer `opus-4-8` (a bare
|
|
73
|
+
* `compareVersions` reads the date as a huge version component).
|
|
74
|
+
*/
|
|
75
|
+
function cleanForCompare(id) {
|
|
76
|
+
return id
|
|
77
|
+
.replace(/-\d{8}(?=($|-))/, '')
|
|
78
|
+
.replace(/-v\d+$/, '')
|
|
79
|
+
.replace(/-fast$/, '');
|
|
80
|
+
}
|
|
81
|
+
/** Numeric segments of a (cleaned) id, splitting on BOTH dashes and dots. */
|
|
82
|
+
function versionSegments(id) {
|
|
83
|
+
const m = cleanForCompare(id).match(/\d+/g);
|
|
84
|
+
return m ? m.map((n) => parseInt(n, 10)) : [];
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Newest concrete id within a family wins. `compareVersions` only splits on `.`,
|
|
88
|
+
* so it degenerates to a single `[0]` segment for a dash-separated model id and
|
|
89
|
+
* mis-ranks e.g. `claude-sonnet-5` below `claude-sonnet-4-6`. Compare the numeric
|
|
90
|
+
* segments directly instead.
|
|
91
|
+
*/
|
|
92
|
+
function newer(a, b) {
|
|
93
|
+
const A = versionSegments(a);
|
|
94
|
+
const B = versionSegments(b);
|
|
95
|
+
for (let i = 0; i < Math.max(A.length, B.length); i++) {
|
|
96
|
+
const d = (A[i] ?? 0) - (B[i] ?? 0);
|
|
97
|
+
if (d !== 0)
|
|
98
|
+
return d;
|
|
99
|
+
}
|
|
100
|
+
return 0;
|
|
101
|
+
}
|
|
102
|
+
/**
|
|
103
|
+
* Rank a harness's catalog models cheapest -> dearest and collapse variants of
|
|
104
|
+
* one model to a single rung (keeping the newest concrete id). Strategy depends
|
|
105
|
+
* on the harness class: aggregator (Cursor) ranks by price of the normalized
|
|
106
|
+
* base id; single-provider harnesses rank by the provider lineup with price and
|
|
107
|
+
* size tokens as fallbacks.
|
|
108
|
+
*/
|
|
109
|
+
function rankCatalog(agent, models) {
|
|
110
|
+
const usable = models.filter((m) => !PSEUDO.test(m.id));
|
|
111
|
+
const aggregator = agent === 'cursor';
|
|
112
|
+
const scored = usable.map((m) => {
|
|
113
|
+
const rawId = m.id;
|
|
114
|
+
const baseId = aggregator ? normalizeAggregatorId(rawId) : rawId;
|
|
115
|
+
const lc = baseId.toLowerCase();
|
|
116
|
+
const price = blended(baseId) ?? blended(rawId);
|
|
117
|
+
let rank;
|
|
118
|
+
let family;
|
|
119
|
+
if (aggregator) {
|
|
120
|
+
// cross-provider: price is the unifying signal; family = base id
|
|
121
|
+
rank = price != null ? price * 1e6 : 100 + sizeTokenRank(lc);
|
|
122
|
+
family = baseId;
|
|
123
|
+
}
|
|
124
|
+
else {
|
|
125
|
+
const fam = anthropicFamilyRank(lc);
|
|
126
|
+
const desc = descriptionRank(m.description);
|
|
127
|
+
if (fam != null) {
|
|
128
|
+
rank = fam;
|
|
129
|
+
family = `anthropic-${fam}`;
|
|
130
|
+
}
|
|
131
|
+
else if (desc != null) {
|
|
132
|
+
rank = desc;
|
|
133
|
+
family = `desc-${desc}-${baseId.replace(/[0-9].*$/, '')}`;
|
|
134
|
+
}
|
|
135
|
+
else if (price != null) {
|
|
136
|
+
rank = 10 + price * 1e6;
|
|
137
|
+
// Collapse only true re-releases of ONE model (same base id, differing
|
|
138
|
+
// date/rebuild suffix). Keying on price would merge two DIFFERENT models
|
|
139
|
+
// that happen to cost the same (e.g. gpt-5.5 and gpt-5.6-sol), dropping
|
|
140
|
+
// one from every tier.
|
|
141
|
+
family = cleanForCompare(baseId);
|
|
142
|
+
}
|
|
143
|
+
else {
|
|
144
|
+
rank = 20 + sizeTokenRank(lc);
|
|
145
|
+
family = cleanForCompare(baseId);
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
return { id: rawId, baseId, rank, family, price };
|
|
149
|
+
});
|
|
150
|
+
// collapse by family: keep the newest concrete id, lowest (cheapest) rank
|
|
151
|
+
const byFamily = new Map();
|
|
152
|
+
for (const s of scored) {
|
|
153
|
+
const prev = byFamily.get(s.family);
|
|
154
|
+
if (!prev) {
|
|
155
|
+
byFamily.set(s.family, s);
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
// keep the newer id; keep the cheaper rank
|
|
159
|
+
if (newer(s.id, prev.id) > 0)
|
|
160
|
+
prev.id = s.id;
|
|
161
|
+
if (s.rank < prev.rank)
|
|
162
|
+
prev.rank = s.rank;
|
|
163
|
+
if (prev.price == null && s.price != null)
|
|
164
|
+
prev.price = s.price;
|
|
165
|
+
}
|
|
166
|
+
return [...byFamily.values()].sort((a, b) => a.rank - b.rank || newer(b.id, a.id));
|
|
167
|
+
}
|
|
168
|
+
/** Which ranked rung each tier index maps to, collapsing when there are < 4 rungs. */
|
|
169
|
+
function rungIndexFor(tierIndex, n) {
|
|
170
|
+
return n >= 4 ? Math.round((tierIndex / 3) * (n - 1)) : Math.min(tierIndex, n - 1);
|
|
171
|
+
}
|
|
172
|
+
/**
|
|
173
|
+
* Resolve all four tiers for an (agent, version). The map is what `agents models`
|
|
174
|
+
* prints and what `resolveTier` indexes into.
|
|
175
|
+
*/
|
|
176
|
+
export function resolveTierMap(agent, version) {
|
|
177
|
+
// Droid: curated credit-multiplier map (no live catalog).
|
|
178
|
+
if (agent === 'droid') {
|
|
179
|
+
return {
|
|
180
|
+
cheap: { tier: 'cheap', model: DROID_TIERS.cheap, note: 'Droid Core 0.55x' },
|
|
181
|
+
default: { tier: 'default', model: DROID_TIERS.default, note: 'Droid Core 0.6x' },
|
|
182
|
+
best: { tier: 'best', model: DROID_TIERS.best, note: '2x' },
|
|
183
|
+
ultra: { tier: 'ultra', model: DROID_TIERS.ultra, clampedFrom: 'best', note: 'capped at 2x (4x models excluded)' },
|
|
184
|
+
};
|
|
185
|
+
}
|
|
186
|
+
const catalog = getModelCatalog(agent, version);
|
|
187
|
+
return tierizeModels(agent, catalog?.models ?? []);
|
|
188
|
+
}
|
|
189
|
+
/**
|
|
190
|
+
* Map a harness's catalog models onto the four tiers. Pure (no catalog lookup)
|
|
191
|
+
* so it is directly testable with synthetic inputs. Ranks the models, collapses
|
|
192
|
+
* variants, buckets onto cheap/default/best/ultra, and clamps absent tiers down
|
|
193
|
+
* to the nearest lower one. A single-model harness maps the tiers to reasoning
|
|
194
|
+
* effort instead of models.
|
|
195
|
+
*/
|
|
196
|
+
export function tierizeModels(agent, models) {
|
|
197
|
+
const rungs = rankCatalog(agent, models);
|
|
198
|
+
// Single-model harness (e.g. Grok): the tier is reasoning effort, not a model.
|
|
199
|
+
if (rungs.length === 1) {
|
|
200
|
+
const only = rungs[0].id;
|
|
201
|
+
const map = {};
|
|
202
|
+
for (const t of MODEL_TIERS)
|
|
203
|
+
map[t] = { tier: t, model: only, effort: TIER_EFFORT[t], note: 'single model — tier maps to reasoning effort' };
|
|
204
|
+
return map;
|
|
205
|
+
}
|
|
206
|
+
const n = rungs.length;
|
|
207
|
+
const map = {};
|
|
208
|
+
if (n === 0) {
|
|
209
|
+
// Fail-safe: no catalog -> every tier null, caller drops the --model flag.
|
|
210
|
+
for (const t of MODEL_TIERS)
|
|
211
|
+
map[t] = { tier: t, model: null };
|
|
212
|
+
return map;
|
|
213
|
+
}
|
|
214
|
+
// Map each tier onto a rung; a tier that shares the rung of the tier below it
|
|
215
|
+
// has no distinct rung of its own, so mark it clamped for an honest display.
|
|
216
|
+
for (let i = 0; i < MODEL_TIERS.length; i++) {
|
|
217
|
+
const t = MODEL_TIERS[i];
|
|
218
|
+
const idx = rungIndexFor(i, n);
|
|
219
|
+
const shared = i > 0 && rungIndexFor(i - 1, n) === idx;
|
|
220
|
+
map[t] = shared
|
|
221
|
+
? { tier: t, model: rungs[idx].id, clampedFrom: MODEL_TIERS[i - 1], note: `no distinct ${t} rung; using ${MODEL_TIERS[i - 1]}` }
|
|
222
|
+
: { tier: t, model: rungs[idx].id };
|
|
223
|
+
}
|
|
224
|
+
return map;
|
|
225
|
+
}
|
|
226
|
+
/** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
|
|
227
|
+
export function resolveTier(agent, version, tier) {
|
|
228
|
+
return resolveTierMap(agent, version)[tier];
|
|
229
|
+
}
|
package/dist/lib/models.d.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import type { AgentId } from './types.js';
|
|
2
|
+
import { type ModelPricing } from './pricing/index.js';
|
|
2
3
|
/** Model identifiers per cloud provider (used by Claude's multi-cloud routing). */
|
|
3
4
|
export interface ModelPerCloud {
|
|
4
5
|
firstParty: string;
|
|
@@ -28,6 +29,8 @@ export interface ModelInfo {
|
|
|
28
29
|
reasoningLevels?: ReasoningLevel[];
|
|
29
30
|
/** Default reasoning level if applicable */
|
|
30
31
|
defaultReasoningLevel?: string;
|
|
32
|
+
/** Per-token USD pricing when known (from prices.json); absent for subscription/unpriced models. */
|
|
33
|
+
pricing?: ModelPricing;
|
|
31
34
|
}
|
|
32
35
|
/** The complete model catalog for a specific (agent, version) pair. */
|
|
33
36
|
export interface ModelCatalog {
|
package/dist/lib/models.js
CHANGED
|
@@ -15,6 +15,7 @@ import { getVersionDir, getVersionHomePath, getBinaryPath } from './versions.js'
|
|
|
15
15
|
import { getModelsCachePath } from './state.js';
|
|
16
16
|
import { agentConfigDirName } from './agents.js';
|
|
17
17
|
import { resolveRunDefaults } from './run-defaults.js';
|
|
18
|
+
import { getModelPricing } from './pricing/index.js';
|
|
18
19
|
const CACHE_PATH = getModelsCachePath();
|
|
19
20
|
/**
|
|
20
21
|
* Bump when the extractor logic changes shape in an incompatible way so cached
|
|
@@ -342,17 +343,12 @@ function extractClaudeCatalog(text) {
|
|
|
342
343
|
mantle: m[6] ?? null,
|
|
343
344
|
};
|
|
344
345
|
}
|
|
345
|
-
const allIds = new Set([
|
|
346
|
-
...Object.values(aliases),
|
|
347
|
-
...Object.keys(displayNames),
|
|
348
|
-
...Object.keys(perCloud),
|
|
349
|
-
]);
|
|
350
346
|
const aliasReverse = {};
|
|
351
347
|
for (const [a, id] of Object.entries(aliases))
|
|
352
348
|
aliasReverse[id] = a;
|
|
353
349
|
const defaults = new Set(Object.values(aliases));
|
|
354
|
-
const
|
|
355
|
-
.filter((id) => /^claude-(opus|sonnet|haiku)-/.test(id))
|
|
350
|
+
const build = (ids) => Array.from(new Set(ids))
|
|
351
|
+
.filter((id) => /^claude-(opus|sonnet|haiku|fable|mythos)-/.test(id))
|
|
356
352
|
.sort()
|
|
357
353
|
.map((id) => ({
|
|
358
354
|
id,
|
|
@@ -361,6 +357,32 @@ function extractClaudeCatalog(text) {
|
|
|
361
357
|
isDefault: defaults.has(id),
|
|
362
358
|
perCloud: perCloud[id],
|
|
363
359
|
}));
|
|
360
|
+
// The structured maps (alias/perCloud/const) are the curated, accurate
|
|
361
|
+
// supported set. Prefer them.
|
|
362
|
+
let models = build([
|
|
363
|
+
...Object.values(aliases),
|
|
364
|
+
...Object.keys(displayNames),
|
|
365
|
+
...Object.keys(perCloud),
|
|
366
|
+
]);
|
|
367
|
+
// Fallback id scan. The structured maps fail on the newest native-binary
|
|
368
|
+
// format (verified: claude@2.1.219 leaks only a stray id, so the curated set
|
|
369
|
+
// is effectively empty). Only when the curated catalog is that thin do we scan
|
|
370
|
+
// the raw strings for canonical ids -- so an older version keeps its precise
|
|
371
|
+
// catalog while a newer one still gets a real catalog (incl. fable/mythos and
|
|
372
|
+
// the opus-5/sonnet-5 line) rather than an empty or single-model one.
|
|
373
|
+
if (models.length < 2) {
|
|
374
|
+
const scanned = new Set();
|
|
375
|
+
// Dash-separated segments only. Real ids are `claude-sonnet-4-6`; the dotted
|
|
376
|
+
// form `claude-sonnet-4.6` appears only inside the binary's own "Typo in
|
|
377
|
+
// model ID" troubleshooting text, so a `.`-permitting pattern would scrape a
|
|
378
|
+
// non-model string as if it were real.
|
|
379
|
+
const idRe = /claude-(?:opus|sonnet|haiku|fable|mythos)-\d+(?:-\d+)*(?:-(?:fast|v\d+))?/g;
|
|
380
|
+
let sm;
|
|
381
|
+
while ((sm = idRe.exec(text)) !== null)
|
|
382
|
+
scanned.add(sm[0]);
|
|
383
|
+
if (scanned.size >= 2)
|
|
384
|
+
models = build(scanned);
|
|
385
|
+
}
|
|
364
386
|
return { models, aliases };
|
|
365
387
|
}
|
|
366
388
|
/**
|
|
@@ -929,6 +951,14 @@ export function getModelCatalog(agent, version) {
|
|
|
929
951
|
else if (agent === 'grok')
|
|
930
952
|
({ models, aliases } = extractGrokCatalog(src.path));
|
|
931
953
|
}
|
|
954
|
+
// Attach per-token pricing where the offline table knows the model, so the
|
|
955
|
+
// catalog carries $/token for the tier display and budgeting. Subscription /
|
|
956
|
+
// unknown models keep `pricing` undefined (surfaced as "--", never faked).
|
|
957
|
+
for (const m of models) {
|
|
958
|
+
const p = getModelPricing(m.id);
|
|
959
|
+
if (p)
|
|
960
|
+
m.pricing = p;
|
|
961
|
+
}
|
|
932
962
|
const catalog = {
|
|
933
963
|
agent,
|
|
934
964
|
version,
|
|
@@ -1123,5 +1153,12 @@ export function buildReasoningFlags(agent, level) {
|
|
|
1123
1153
|
const droidLevel = (normalized === 'xhigh' || normalized === 'max') ? 'high' : normalized;
|
|
1124
1154
|
return ['-r', droidLevel];
|
|
1125
1155
|
}
|
|
1156
|
+
if (agent === 'grok') {
|
|
1157
|
+
// Grok: `--reasoning-effort <low|medium|high>` (alias --effort). xhigh/max
|
|
1158
|
+
// clamp to high. This is the effort dial cost tiers steer for Grok, whose
|
|
1159
|
+
// catalog exposes a single model.
|
|
1160
|
+
const grokLevel = (normalized === 'xhigh' || normalized === 'max') ? 'high' : normalized;
|
|
1161
|
+
return ['--reasoning-effort', grokLevel];
|
|
1162
|
+
}
|
|
1126
1163
|
return [];
|
|
1127
1164
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"version": "2026-
|
|
2
|
+
"version": "2026-08-03",
|
|
3
3
|
"models": {
|
|
4
4
|
"claude-opus-4": {
|
|
5
5
|
"inputPerToken": 0.000005,
|
|
@@ -49,6 +49,21 @@
|
|
|
49
49
|
"cacheReadPerToken": 0.0000015,
|
|
50
50
|
"cacheWritePerToken": 0.00001875
|
|
51
51
|
},
|
|
52
|
+
"gpt-5.6-sol": {
|
|
53
|
+
"inputPerToken": 0.000005,
|
|
54
|
+
"outputPerToken": 0.00003,
|
|
55
|
+
"cacheReadPerToken": 0.0000005
|
|
56
|
+
},
|
|
57
|
+
"gpt-5.6-terra": {
|
|
58
|
+
"inputPerToken": 0.0000025,
|
|
59
|
+
"outputPerToken": 0.000015,
|
|
60
|
+
"cacheReadPerToken": 0.00000025
|
|
61
|
+
},
|
|
62
|
+
"gpt-5.6-luna": {
|
|
63
|
+
"inputPerToken": 0.000001,
|
|
64
|
+
"outputPerToken": 0.000006,
|
|
65
|
+
"cacheReadPerToken": 0.0000001
|
|
66
|
+
},
|
|
52
67
|
"gpt-5.5": {
|
|
53
68
|
"inputPerToken": 0.000005,
|
|
54
69
|
"outputPerToken": 0.00003,
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Where a project's work actually went, from the local git history.
|
|
3
|
+
*
|
|
4
|
+
* The card can say how many agents are running and how many PRs merged, but not
|
|
5
|
+
* *what was worked on*. That answer is already sitting in the checkout — every
|
|
6
|
+
* merged commit names the files it touched — so it costs no API call, no
|
|
7
|
+
* credential, and no rate-limit budget. Measured at 0.23s on this repo's own
|
|
8
|
+
* 7-day window (897 commits), which is why it runs unconditionally rather than
|
|
9
|
+
* behind a flag.
|
|
10
|
+
*
|
|
11
|
+
* Deliberately NOT from `gh`: the GitHub API would spend a request per PR to
|
|
12
|
+
* learn what `git log --name-only` already knows locally, and it would be wrong
|
|
13
|
+
* on a monorepo whose interesting unit is a subdirectory rather than a repo.
|
|
14
|
+
*/
|
|
15
|
+
/** One directory and how many file-touches landed in it during the window. */
|
|
16
|
+
export interface FocusArea {
|
|
17
|
+
path: string;
|
|
18
|
+
touches: number;
|
|
19
|
+
}
|
|
20
|
+
/** Areas shown on the card before the tail is dropped. */
|
|
21
|
+
export declare const FOCUS_LIMIT = 4;
|
|
22
|
+
/**
|
|
23
|
+
* Bucket a file path to its area. Files shallower than {@link DEPTH} bucket to
|
|
24
|
+
* their own directory, so a repo-root `README.md` does not vanish.
|
|
25
|
+
*/
|
|
26
|
+
export declare function focusBucket(file: string): string | undefined;
|
|
27
|
+
/**
|
|
28
|
+
* Rank areas by file-touches, descending, ties broken by path so the order is
|
|
29
|
+
* stable across runs. Pure — the caller supplies the file list, so this is
|
|
30
|
+
* testable without a git repo.
|
|
31
|
+
*/
|
|
32
|
+
export declare function rankFocusAreas(files: string[], limit?: number): FocusArea[];
|
|
33
|
+
/**
|
|
34
|
+
* Read the window's changed files from a checkout. Best-effort in the same shape
|
|
35
|
+
* as the rest of the card's enrichment: a missing checkout, a shallow clone, or
|
|
36
|
+
* a repo with no commits in the window yields an empty list, never a throw.
|
|
37
|
+
*
|
|
38
|
+
* Reads the LOCAL default branch ref rather than fetching — a status command
|
|
39
|
+
* must not mutate the repo it is describing, so the answer is only as fresh as
|
|
40
|
+
* the user's last fetch, which is the correct trade for a read-only card.
|
|
41
|
+
*/
|
|
42
|
+
export declare function readFocusAreas(root: string, windowDays: number): Promise<FocusArea[]>;
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Where a project's work actually went, from the local git history.
|
|
3
|
+
*
|
|
4
|
+
* The card can say how many agents are running and how many PRs merged, but not
|
|
5
|
+
* *what was worked on*. That answer is already sitting in the checkout — every
|
|
6
|
+
* merged commit names the files it touched — so it costs no API call, no
|
|
7
|
+
* credential, and no rate-limit budget. Measured at 0.23s on this repo's own
|
|
8
|
+
* 7-day window (897 commits), which is why it runs unconditionally rather than
|
|
9
|
+
* behind a flag.
|
|
10
|
+
*
|
|
11
|
+
* Deliberately NOT from `gh`: the GitHub API would spend a request per PR to
|
|
12
|
+
* learn what `git log --name-only` already knows locally, and it would be wrong
|
|
13
|
+
* on a monorepo whose interesting unit is a subdirectory rather than a repo.
|
|
14
|
+
*/
|
|
15
|
+
import { execFile } from 'child_process';
|
|
16
|
+
import { promisify } from 'util';
|
|
17
|
+
const execFileAsync = promisify(execFile);
|
|
18
|
+
/** How deep a bucket goes: `apps/cli/src`, not `apps` and not every leaf file. */
|
|
19
|
+
const DEPTH = 3;
|
|
20
|
+
/** Areas shown on the card before the tail is dropped. */
|
|
21
|
+
export const FOCUS_LIMIT = 4;
|
|
22
|
+
/**
|
|
23
|
+
* Paths whose churn is process, not engineering.
|
|
24
|
+
*
|
|
25
|
+
* This repo files one changelog fragment per PR, so `.changelog` ranks second by
|
|
26
|
+
* raw file-touches — presenting it as an "area of focus" would read as a signal
|
|
27
|
+
* while measuring nothing but the number of PRs. Same for the generated
|
|
28
|
+
* CHANGELOG and lockfiles.
|
|
29
|
+
*/
|
|
30
|
+
const NOISE = /(^|\/)(\.changelog|CHANGELOG\.md|bun\.lock|package-lock\.json|yarn\.lock)(\/|$)/;
|
|
31
|
+
/**
|
|
32
|
+
* Bucket a file path to its area. Files shallower than {@link DEPTH} bucket to
|
|
33
|
+
* their own directory, so a repo-root `README.md` does not vanish.
|
|
34
|
+
*/
|
|
35
|
+
export function focusBucket(file) {
|
|
36
|
+
if (NOISE.test(file))
|
|
37
|
+
return undefined;
|
|
38
|
+
const parts = file.split('/').filter(Boolean);
|
|
39
|
+
if (parts.length === 0)
|
|
40
|
+
return undefined;
|
|
41
|
+
if (parts.length === 1)
|
|
42
|
+
return parts[0];
|
|
43
|
+
return parts.slice(0, Math.min(DEPTH, parts.length - 1)).join('/');
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Rank areas by file-touches, descending, ties broken by path so the order is
|
|
47
|
+
* stable across runs. Pure — the caller supplies the file list, so this is
|
|
48
|
+
* testable without a git repo.
|
|
49
|
+
*/
|
|
50
|
+
export function rankFocusAreas(files, limit = FOCUS_LIMIT) {
|
|
51
|
+
const counts = new Map();
|
|
52
|
+
for (const f of files) {
|
|
53
|
+
const bucket = focusBucket(f.trim());
|
|
54
|
+
if (!bucket)
|
|
55
|
+
continue;
|
|
56
|
+
counts.set(bucket, (counts.get(bucket) ?? 0) + 1);
|
|
57
|
+
}
|
|
58
|
+
return [...counts.entries()]
|
|
59
|
+
.map(([path, touches]) => ({ path, touches }))
|
|
60
|
+
.sort((a, b) => b.touches - a.touches || a.path.localeCompare(b.path))
|
|
61
|
+
.slice(0, Math.max(1, limit));
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Read the window's changed files from a checkout. Best-effort in the same shape
|
|
65
|
+
* as the rest of the card's enrichment: a missing checkout, a shallow clone, or
|
|
66
|
+
* a repo with no commits in the window yields an empty list, never a throw.
|
|
67
|
+
*
|
|
68
|
+
* Reads the LOCAL default branch ref rather than fetching — a status command
|
|
69
|
+
* must not mutate the repo it is describing, so the answer is only as fresh as
|
|
70
|
+
* the user's last fetch, which is the correct trade for a read-only card.
|
|
71
|
+
*/
|
|
72
|
+
export async function readFocusAreas(root, windowDays) {
|
|
73
|
+
try {
|
|
74
|
+
const { stdout } = await execFileAsync('git', ['-C', root, 'log', `--since=${windowDays} days ago`, '--name-only', '--pretty=format:'], { timeout: 5000, encoding: 'utf8', maxBuffer: 32 * 1024 * 1024 });
|
|
75
|
+
return rankFocusAreas(stdout.split('\n').filter((l) => l.trim().length > 0));
|
|
76
|
+
}
|
|
77
|
+
catch {
|
|
78
|
+
return [];
|
|
79
|
+
}
|
|
80
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What a project's milestone dates prove about its schedule — and nothing more.
|
|
3
|
+
*
|
|
4
|
+
* The tempting version of this file computes "on track / at risk". It cannot be
|
|
5
|
+
* written honestly against this data. Probed live against a real workspace:
|
|
6
|
+
*
|
|
7
|
+
* health: null projectUpdates: []
|
|
8
|
+
* startDate: null targetDate: null
|
|
9
|
+
* scopeHistory: [] completedScopeHistory: []
|
|
10
|
+
*
|
|
11
|
+
* "Behind schedule" needs either start+target dates to interpolate an expected
|
|
12
|
+
* progress line, or a history series to extrapolate a finish date. Both are
|
|
13
|
+
* absent, so any on-track/at-risk chip would be invented — and a confident wrong
|
|
14
|
+
* answer on a status card is worse than a blank one, because it is unfalsifiable
|
|
15
|
+
* from the card itself.
|
|
16
|
+
*
|
|
17
|
+
* So every verdict here is arithmetic on a stored date or a count:
|
|
18
|
+
*
|
|
19
|
+
* overdue a milestone's targetDate has passed and it is unfinished
|
|
20
|
+
* due-soon the next one lands within DUE_SOON_DAYS
|
|
21
|
+
* untracked milestones exist but nothing is filed against any of them,
|
|
22
|
+
* so their progress is not measurable
|
|
23
|
+
* scheduled dated milestones ahead, none due soon, work is filed
|
|
24
|
+
* no-dates milestones exist, none carries a date
|
|
25
|
+
* none the project declares no milestones
|
|
26
|
+
*
|
|
27
|
+
* `declared` relays Linear's OWN health when a human has posted one. It is
|
|
28
|
+
* passed through and attributed, never synthesized — if the user starts posting
|
|
29
|
+
* project updates, their answer wins over anything derived here.
|
|
30
|
+
*/
|
|
31
|
+
import type { LinearMilestone } from './linear-project-counts.js';
|
|
32
|
+
/** How far ahead counts as "due soon" — one sprint's notice. */
|
|
33
|
+
export declare const DUE_SOON_DAYS = 14;
|
|
34
|
+
/** A verdict about the schedule, as a tagged union so `--json` stays stable. */
|
|
35
|
+
export type ProjectVerdict = {
|
|
36
|
+
kind: 'declared';
|
|
37
|
+
health: string;
|
|
38
|
+
} | {
|
|
39
|
+
kind: 'overdue';
|
|
40
|
+
milestone: string;
|
|
41
|
+
days: number;
|
|
42
|
+
} | {
|
|
43
|
+
kind: 'due-soon';
|
|
44
|
+
milestone: string;
|
|
45
|
+
days: number;
|
|
46
|
+
} | {
|
|
47
|
+
kind: 'untracked';
|
|
48
|
+
milestones: number;
|
|
49
|
+
} | {
|
|
50
|
+
kind: 'scheduled';
|
|
51
|
+
milestone: string;
|
|
52
|
+
days: number;
|
|
53
|
+
} | {
|
|
54
|
+
kind: 'no-dates';
|
|
55
|
+
milestones: number;
|
|
56
|
+
} | {
|
|
57
|
+
kind: 'none';
|
|
58
|
+
};
|
|
59
|
+
/** Whole days from `nowMs` to a `YYYY-MM-DD` date, compared at LOCAL midnight. */
|
|
60
|
+
export declare function daysUntil(targetDate: string, nowMs: number): number | undefined;
|
|
61
|
+
/**
|
|
62
|
+
* Decide what the dates prove. Precedence is time-sensitivity first:
|
|
63
|
+
*
|
|
64
|
+
* declared > overdue > due-soon > untracked > no-dates > scheduled
|
|
65
|
+
*
|
|
66
|
+
* An overdue milestone outranks an approaching one, and both outrank the
|
|
67
|
+
* observation that nothing is filed — a deadline moves, that observation does
|
|
68
|
+
* not. `declared` overrides everything, because a human said it.
|
|
69
|
+
*/
|
|
70
|
+
export declare function scheduleVerdict(milestones: LinearMilestone[], nowMs: number, declaredHealth?: string | null): ProjectVerdict;
|
|
71
|
+
/** One line for the card. Returns undefined for `none` — an empty row says nothing. */
|
|
72
|
+
export declare function formatVerdict(v: ProjectVerdict): {
|
|
73
|
+
text: string;
|
|
74
|
+
warn: boolean;
|
|
75
|
+
} | undefined;
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* What a project's milestone dates prove about its schedule — and nothing more.
|
|
3
|
+
*
|
|
4
|
+
* The tempting version of this file computes "on track / at risk". It cannot be
|
|
5
|
+
* written honestly against this data. Probed live against a real workspace:
|
|
6
|
+
*
|
|
7
|
+
* health: null projectUpdates: []
|
|
8
|
+
* startDate: null targetDate: null
|
|
9
|
+
* scopeHistory: [] completedScopeHistory: []
|
|
10
|
+
*
|
|
11
|
+
* "Behind schedule" needs either start+target dates to interpolate an expected
|
|
12
|
+
* progress line, or a history series to extrapolate a finish date. Both are
|
|
13
|
+
* absent, so any on-track/at-risk chip would be invented — and a confident wrong
|
|
14
|
+
* answer on a status card is worse than a blank one, because it is unfalsifiable
|
|
15
|
+
* from the card itself.
|
|
16
|
+
*
|
|
17
|
+
* So every verdict here is arithmetic on a stored date or a count:
|
|
18
|
+
*
|
|
19
|
+
* overdue a milestone's targetDate has passed and it is unfinished
|
|
20
|
+
* due-soon the next one lands within DUE_SOON_DAYS
|
|
21
|
+
* untracked milestones exist but nothing is filed against any of them,
|
|
22
|
+
* so their progress is not measurable
|
|
23
|
+
* scheduled dated milestones ahead, none due soon, work is filed
|
|
24
|
+
* no-dates milestones exist, none carries a date
|
|
25
|
+
* none the project declares no milestones
|
|
26
|
+
*
|
|
27
|
+
* `declared` relays Linear's OWN health when a human has posted one. It is
|
|
28
|
+
* passed through and attributed, never synthesized — if the user starts posting
|
|
29
|
+
* project updates, their answer wins over anything derived here.
|
|
30
|
+
*/
|
|
31
|
+
/** How far ahead counts as "due soon" — one sprint's notice. */
|
|
32
|
+
export const DUE_SOON_DAYS = 14;
|
|
33
|
+
/** Whole days from `nowMs` to a `YYYY-MM-DD` date, compared at LOCAL midnight. */
|
|
34
|
+
export function daysUntil(targetDate, nowMs) {
|
|
35
|
+
const m = /^(\d{4})-(\d{2})-(\d{2})$/.exec(targetDate.trim());
|
|
36
|
+
if (!m)
|
|
37
|
+
return undefined;
|
|
38
|
+
const due = new Date(Number(m[1]), Number(m[2]) - 1, Number(m[3]));
|
|
39
|
+
if (Number.isNaN(due.getTime()))
|
|
40
|
+
return undefined;
|
|
41
|
+
const now = new Date(nowMs);
|
|
42
|
+
const today = new Date(now.getFullYear(), now.getMonth(), now.getDate());
|
|
43
|
+
return Math.round((due.getTime() - today.getTime()) / 86_400_000);
|
|
44
|
+
}
|
|
45
|
+
const unfinished = (m) => m.total === 0 || m.done < m.total;
|
|
46
|
+
/**
|
|
47
|
+
* Decide what the dates prove. Precedence is time-sensitivity first:
|
|
48
|
+
*
|
|
49
|
+
* declared > overdue > due-soon > untracked > no-dates > scheduled
|
|
50
|
+
*
|
|
51
|
+
* An overdue milestone outranks an approaching one, and both outrank the
|
|
52
|
+
* observation that nothing is filed — a deadline moves, that observation does
|
|
53
|
+
* not. `declared` overrides everything, because a human said it.
|
|
54
|
+
*/
|
|
55
|
+
export function scheduleVerdict(milestones, nowMs, declaredHealth) {
|
|
56
|
+
if (declaredHealth)
|
|
57
|
+
return { kind: 'declared', health: declaredHealth };
|
|
58
|
+
if (milestones.length === 0)
|
|
59
|
+
return { kind: 'none' };
|
|
60
|
+
const open = milestones.filter(unfinished);
|
|
61
|
+
const dated = open
|
|
62
|
+
.map((m) => ({ m, days: m.targetDate ? daysUntil(m.targetDate, nowMs) : undefined }))
|
|
63
|
+
.filter((x) => x.days !== undefined)
|
|
64
|
+
.sort((a, b) => a.days - b.days);
|
|
65
|
+
const worst = dated[0];
|
|
66
|
+
if (worst && worst.days < 0)
|
|
67
|
+
return { kind: 'overdue', milestone: worst.m.name, days: -worst.days };
|
|
68
|
+
// A date bearing down is time-sensitive; "nothing is filed" is a standing
|
|
69
|
+
// condition that will still be true tomorrow. So an approaching deadline is
|
|
70
|
+
// reported even when the milestone has no issues against it — reversing these
|
|
71
|
+
// hid a milestone due in two days behind "3 milestones, no issues filed".
|
|
72
|
+
if (worst && worst.days <= DUE_SOON_DAYS)
|
|
73
|
+
return { kind: 'due-soon', milestone: worst.m.name, days: worst.days };
|
|
74
|
+
// Nothing is filed against ANY milestone, so no progress can be computed for
|
|
75
|
+
// them — the useful thing to say, and the actual state of a project whose
|
|
76
|
+
// milestones were created before its issues. Checked against the full list,
|
|
77
|
+
// not just the open ones: a COMPLETED milestone necessarily has issues, so a
|
|
78
|
+
// project with one cannot honestly be called untracked.
|
|
79
|
+
if (milestones.every((m) => m.total === 0))
|
|
80
|
+
return { kind: 'untracked', milestones: milestones.length };
|
|
81
|
+
if (!worst)
|
|
82
|
+
return { kind: 'no-dates', milestones: open.length };
|
|
83
|
+
return { kind: 'scheduled', milestone: worst.m.name, days: worst.days };
|
|
84
|
+
}
|
|
85
|
+
/** One line for the card. Returns undefined for `none` — an empty row says nothing. */
|
|
86
|
+
export function formatVerdict(v) {
|
|
87
|
+
switch (v.kind) {
|
|
88
|
+
case 'declared':
|
|
89
|
+
// Attributed, so nobody mistakes a human's call for a derived one.
|
|
90
|
+
return { text: `per Linear: ${v.health}`, warn: v.health.toLowerCase() !== 'ontrack' };
|
|
91
|
+
case 'overdue':
|
|
92
|
+
return { text: `${v.milestone} overdue by ${v.days} day${v.days === 1 ? '' : 's'}`, warn: true };
|
|
93
|
+
case 'due-soon':
|
|
94
|
+
return {
|
|
95
|
+
text: `${v.milestone} due ${v.days === 0 ? 'today' : v.days === 1 ? 'tomorrow' : `in ${v.days} days`}`,
|
|
96
|
+
warn: false,
|
|
97
|
+
};
|
|
98
|
+
case 'untracked':
|
|
99
|
+
return {
|
|
100
|
+
text: `${v.milestones} milestone${v.milestones === 1 ? '' : 's'}, no issues filed against any — progress is not measurable`,
|
|
101
|
+
warn: true,
|
|
102
|
+
};
|
|
103
|
+
case 'scheduled':
|
|
104
|
+
return { text: `${v.milestone} in ${v.days} days`, warn: false };
|
|
105
|
+
case 'no-dates':
|
|
106
|
+
return { text: `${v.milestones} open milestone${v.milestones === 1 ? '' : 's'}, none dated`, warn: false };
|
|
107
|
+
case 'none':
|
|
108
|
+
return undefined;
|
|
109
|
+
}
|
|
110
|
+
}
|