@phnx-labs/agents-cli 1.21.2 → 1.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/CHANGELOG.md +101 -0
  2. package/README.md +32 -3
  3. package/dist/bin/agents +0 -0
  4. package/dist/commands/computer-actions.d.ts +4 -0
  5. package/dist/commands/computer-actions.js +35 -0
  6. package/dist/commands/computer.js +4 -2
  7. package/dist/commands/exec.d.ts +27 -0
  8. package/dist/commands/exec.js +123 -6
  9. package/dist/commands/models.js +36 -1
  10. package/dist/commands/perf.d.ts +16 -0
  11. package/dist/commands/perf.js +11 -1
  12. package/dist/commands/projects.d.ts +11 -1
  13. package/dist/commands/projects.js +38 -4
  14. package/dist/commands/sessions-backfill.d.ts +32 -0
  15. package/dist/commands/sessions-backfill.js +186 -0
  16. package/dist/commands/sessions-picker.js +17 -2
  17. package/dist/commands/sessions.d.ts +22 -1
  18. package/dist/commands/sessions.js +331 -18
  19. package/dist/commands/teams.js +1 -1
  20. package/dist/commands/worktree.d.ts +3 -3
  21. package/dist/commands/worktree.js +35 -4
  22. package/dist/index.js +8 -0
  23. package/dist/lib/browser/service.js +13 -0
  24. package/dist/lib/computer/dispatch.d.ts +3 -1
  25. package/dist/lib/computer/dispatch.js +10 -2
  26. package/dist/lib/daemon.d.ts +5 -1
  27. package/dist/lib/daemon.js +63 -14
  28. package/dist/lib/devices/resolve-target.d.ts +6 -0
  29. package/dist/lib/devices/resolve-target.js +9 -3
  30. package/dist/lib/event-stream.d.ts +2 -0
  31. package/dist/lib/event-stream.js +3 -0
  32. package/dist/lib/events.d.ts +3 -1
  33. package/dist/lib/events.js +4 -2
  34. package/dist/lib/exec.js +39 -8
  35. package/dist/lib/git.d.ts +14 -0
  36. package/dist/lib/git.js +36 -0
  37. package/dist/lib/hooks/profile.js +1 -14
  38. package/dist/lib/hosts/dispatch.d.ts +12 -0
  39. package/dist/lib/hosts/dispatch.js +23 -6
  40. package/dist/lib/hosts/reconnect.d.ts +38 -0
  41. package/dist/lib/hosts/reconnect.js +85 -4
  42. package/dist/lib/hosts/run-target.js +14 -2
  43. package/dist/lib/menubar/MenubarHelper.app/Contents/CodeResources +0 -0
  44. package/dist/lib/menubar/MenubarHelper.app/Contents/MacOS/MenubarHelper +0 -0
  45. package/dist/lib/model-tiers.d.ts +54 -0
  46. package/dist/lib/model-tiers.js +229 -0
  47. package/dist/lib/models.d.ts +3 -0
  48. package/dist/lib/models.js +44 -7
  49. package/dist/lib/percentile.d.ts +12 -0
  50. package/dist/lib/percentile.js +24 -0
  51. package/dist/lib/perf/db.d.ts +1 -2
  52. package/dist/lib/perf/db.js +2 -14
  53. package/dist/lib/plugins.js +12 -1
  54. package/dist/lib/pricing/prices.json +16 -1
  55. package/dist/lib/project-focus.d.ts +42 -0
  56. package/dist/lib/project-focus.js +80 -0
  57. package/dist/lib/project-schedule.d.ts +75 -0
  58. package/dist/lib/project-schedule.js +110 -0
  59. package/dist/lib/redact.d.ts +2 -0
  60. package/dist/lib/redact.js +22 -0
  61. package/dist/lib/remote-agents-json.d.ts +2 -0
  62. package/dist/lib/remote-agents-json.js +3 -3
  63. package/dist/lib/resources.d.ts +16 -0
  64. package/dist/lib/resources.js +25 -14
  65. package/dist/lib/rotate.d.ts +84 -1
  66. package/dist/lib/rotate.js +155 -5
  67. package/dist/lib/routines.js +1 -14
  68. package/dist/lib/runner.d.ts +4 -2
  69. package/dist/lib/runner.js +21 -5
  70. package/dist/lib/secrets/Agents CLI.app/Contents/CodeResources +0 -0
  71. package/dist/lib/secrets/Agents CLI.app/Contents/MacOS/Agents CLI +0 -0
  72. package/dist/lib/session/bash-command.js +60 -9
  73. package/dist/lib/session/db.d.ts +22 -1
  74. package/dist/lib/session/db.js +516 -24
  75. package/dist/lib/session/discover.d.ts +68 -7
  76. package/dist/lib/session/discover.js +186 -84
  77. package/dist/lib/session/highlights.d.ts +24 -4
  78. package/dist/lib/session/highlights.js +52 -7
  79. package/dist/lib/session/parse.d.ts +8 -1
  80. package/dist/lib/session/parse.js +102 -35
  81. package/dist/lib/session/prompt.d.ts +19 -0
  82. package/dist/lib/session/prompt.js +43 -0
  83. package/dist/lib/session/remote-list.d.ts +71 -0
  84. package/dist/lib/session/remote-list.js +410 -2
  85. package/dist/lib/session/shell-programs.d.ts +15 -0
  86. package/dist/lib/session/shell-programs.js +359 -0
  87. package/dist/lib/session/tool-calls.d.ts +88 -0
  88. package/dist/lib/session/tool-calls.js +612 -0
  89. package/dist/lib/session/tool-index.d.ts +100 -0
  90. package/dist/lib/session/tool-index.js +773 -0
  91. package/dist/lib/session/tool-store.d.ts +15 -0
  92. package/dist/lib/session/tool-store.js +198 -0
  93. package/dist/lib/session/types.d.ts +49 -0
  94. package/dist/lib/state.d.ts +10 -1
  95. package/dist/lib/state.js +11 -2
  96. package/dist/lib/teams/remoteWorktree.d.ts +3 -4
  97. package/dist/lib/teams/remoteWorktree.js +3 -4
  98. package/dist/lib/teams/worktree.d.ts +11 -1
  99. package/dist/lib/teams/worktree.js +42 -4
  100. package/dist/lib/types.d.ts +31 -0
  101. package/dist/lib/types.js +17 -0
  102. package/package.json +3 -1
@@ -28,12 +28,50 @@
28
28
  * The retry policy is a pure state machine (`reconnectStep`) so it is unit-tested
29
29
  * without touching SSH; the loop (`reconnectInteractiveSession`) only adds the real
30
30
  * preflight + `sshStream` re-attach and the wait.
31
+ *
32
+ * **255 from the REMOTE side should never be trusted as "the link dropped."**
33
+ * `reattachRemoteSession`'s `connected` flag is set as soon as the fast preflight
34
+ * probe succeeds — it says nothing about whether the interactive attach that
35
+ * follows actually reattached a live pane. If the REMOTE command (`agents
36
+ * sessions focus <id> --local --attach-only`) itself ever happened to exit 255
37
+ * for a reason that has nothing to do with the ssh transport, `sshStream` would
38
+ * return that same 255, `reconnectStep` couldn't tell it apart from a genuine
39
+ * drop, and `connected: true` would refill the retry budget forever — "attempt
40
+ * 1/N" printed on every single cycle, the terminal filling with aborted-TTY
41
+ * escape-code garbage, `MAX_ATTEMPTS` never actually bounding anything.
42
+ *
43
+ * Two candidate producers of that scenario were investigated —
44
+ * `refuseFallback`'s login-shell fallback (commands/go.ts) and `jumpTo`'s
45
+ * nested remote-tmux hop — and both turned out to be UNREACHABLE through this
46
+ * exact remote command: under `--local`, `gatherLiveTargets` never sets a
47
+ * foreign `.machine` (go.ts:61), so `remote` is always `undefined` in both,
48
+ * and neither branch can fire (the same reasoning that made an earlier,
49
+ * narrower fix here dead code — see git history). So this fix does not close
50
+ * a confirmed incident cause; what it closes is the underlying channel-level
51
+ * flaw that would make *any* future remote-side 255 producer — reachable today
52
+ * or not — indistinguishable from a real drop. {@link wrapRemoteExitCode} wraps
53
+ * the entire remote command so that whatever exit code it decides on, a 255 is
54
+ * remapped to {@link REMOTE_EXIT_255_REMAPPED} before `sshStream` ever sees it,
55
+ * regardless of which internal branch produced it and regardless of the peer's
56
+ * `agents` version (the remap happens in the shell wrapper THIS process sends).
57
+ *
58
+ * This does NOT close every way `reconnectStep` can loop on a real transport
59
+ * 255: a genuinely recurring LOCAL ssh failure (a fast-flapping link, an
60
+ * attach that dies at the TTY stage on every reconnect) still refills the
61
+ * budget every time by design (see "Why the budget refills on `connected`"
62
+ * above) and can still print "attempt 1/N" indefinitely. That's an accepted,
63
+ * pre-existing tradeoff of the original feature, not something this fix
64
+ * changes either way — tracked separately (agents-cli#1884), not fixed here.
31
65
  */
32
66
  import { sshExec, sshStream, shellQuote } from '../ssh-exec.js';
33
67
  import { sshTargetFor } from './types.js';
34
68
  /** ssh's connection-layer failure code — the signal that the link dropped rather
35
69
  * than the remote command exiting on its own. Mirrors ssh-exec.ts `sshStream`. */
36
70
  export const SSH_CONN_FAILURE = 255;
71
+ /** What a would-be-255 remote-origin exit code is remapped to by
72
+ * {@link wrapRemoteExitCode} — see the file header. Never produced by the ssh
73
+ * transport itself, so it can never be confused with {@link SSH_CONN_FAILURE}. */
74
+ export const REMOTE_EXIT_255_REMAPPED = 254;
37
75
  /** Consecutive failed-to-connect reattaches before giving up. Backoff is capped at
38
76
  * {@link MAX_BACKOFF_MS}. A reattach that actually reconnected (then dropped again)
39
77
  * refills the budget, so a long session that blinks all day reconnects every time
@@ -77,6 +115,50 @@ export function reconnectNotice(sessionId, host, attempt, waitMs) {
77
115
  export function exhaustedNotice(sessionId, host) {
78
116
  return `\nCouldn't reconnect to ${host} after ${MAX_ATTEMPTS} attempts. The agent may still be running — reattach when the network is back:\n agents sessions focus ${sessionId.slice(0, 8)}\n`;
79
117
  }
118
+ /** Notice shown when a reattach stops on a remapped remote-side exit
119
+ * ({@link REMOTE_EXIT_255_REMAPPED} — a would-be-255 the remote command decided
120
+ * on for its own reasons, not the ssh transport dropping; see
121
+ * {@link wrapRemoteExitCode}). Distinct from {@link exhaustedNotice}, which is
122
+ * only for a genuinely spent retry budget. */
123
+ export function remoteExitNotice(sessionId, host) {
124
+ return `\nReattach to ${sessionId.slice(0, 8)} on ${host} ended (not a network drop) — check whether it's still live:\n agents sessions ${sessionId.slice(0, 8)}\n`;
125
+ }
126
+ /**
127
+ * Wrap `cmd` in `bash -lc` (the login-shell pattern `buildRemoteAgentsInvocation`
128
+ * in remote-cmd.ts uses for its own POSIX callers — see its doc comment for why
129
+ * a login shell at all; the sibling interactive dispatch in dispatch.ts sends a
130
+ * bare `agents …` with no shell wrapper, so this is a NEW login-shell hop on the
131
+ * reattach path specifically, not something already universal here) with a
132
+ * trailing exit-code remap: whatever `cmd` itself exits with, a 255 becomes
133
+ * {@link REMOTE_EXIT_255_REMAPPED} before the wrapper exits — see the file
134
+ * header for why. Every other code (0, 1, …) passes through unchanged. This
135
+ * carries no PATH bootstrap of its own — `ensureHostReady`/`readyProbe` already
136
+ * gates every `--host` dispatch on `bash -lc 'agents --version'` succeeding
137
+ * before a run is attempted at all, so the peer's login shell resolving `agents`
138
+ * is an established precondition here too. Pure string-building, so it is
139
+ * unit-tested without SSH (and, since the constructed script is ordinary POSIX,
140
+ * also exercised by actually running it through a real shell in the test — no
141
+ * mock needed).
142
+ */
143
+ export function wrapRemoteExitCode(cmd) {
144
+ const guarded = `${cmd}; rc=$?; [ "$rc" = "${SSH_CONN_FAILURE}" ] && rc=${REMOTE_EXIT_255_REMAPPED}; exit "$rc"`;
145
+ return `bash -lc ${shellQuote(guarded)}`;
146
+ }
147
+ /**
148
+ * The remote command a reattach runs — the peer's own reconnect verb
149
+ * (`agents sessions focus <id> --local --attach-only`), wrapped by
150
+ * {@link wrapRemoteExitCode} so a stray remote-origin 255 (from this command,
151
+ * whatever produces it — see the file header) can never masquerade as a
152
+ * network drop. Split out from {@link reattachRemoteSession} so it is
153
+ * unit-tested without SSH — mirrors `remoteAgentsJsonCommand` in
154
+ * lib/remote-agents-json.ts.
155
+ */
156
+ export function reattachRemoteCommand(sessionId) {
157
+ const inner = ['agents', 'sessions', 'focus', sessionId, '--local', '--attach-only']
158
+ .map(shellQuote)
159
+ .join(' ');
160
+ return wrapRemoteExitCode(inner);
161
+ }
80
162
  /**
81
163
  * Re-attach the live remote tmux pane for `sessionId` by driving the peer's own
82
164
  * `agents sessions focus`. A fast, un-multiplexed preflight probe (`ssh … true`)
@@ -94,10 +176,7 @@ export function reattachRemoteSession(host, sessionId) {
94
176
  const probe = sshExec(target, 'true', { multiplex: false });
95
177
  if (probe.code !== 0)
96
178
  return { code: SSH_CONN_FAILURE, connected: false };
97
- const remoteCmd = ['agents', 'sessions', 'focus', sessionId, '--local', '--attach-only']
98
- .map(shellQuote)
99
- .join(' ');
100
- return { code: sshStream(target, remoteCmd, { tty: true }), connected: true };
179
+ return { code: sshStream(target, reattachRemoteCommand(sessionId), { tty: true }), connected: true };
101
180
  }
102
181
  const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
103
182
  /**
@@ -117,6 +196,8 @@ export async function reconnectInteractiveSession(opts) {
117
196
  if (decision.action === 'stop') {
118
197
  if (decision.code === SSH_CONN_FAILURE)
119
198
  write(exhaustedNotice(opts.sessionId, opts.host.name));
199
+ else if (decision.code === REMOTE_EXIT_255_REMAPPED)
200
+ write(remoteExitNotice(opts.sessionId, opts.host.name));
120
201
  return decision.code;
121
202
  }
122
203
  write(reconnectNotice(opts.sessionId, opts.host.name, decision.state.attempt, decision.waitMs));
@@ -55,7 +55,16 @@ export async function resolveHostRunTarget(name, opts = {}) {
55
55
  }
56
56
  /** Resolve the id the remote host will adopt for a fresh Claude session. */
57
57
  export function resolveHostSessionId(agent, resume, sessionId) {
58
- return agent === 'claude' && !resume ? sessionId ?? randomUUID() : undefined;
58
+ if (resume)
59
+ return undefined;
60
+ if (agent === 'claude')
61
+ return sessionId ?? randomUUID();
62
+ // `run auto`: the harness is picked on the REMOTE. Forward an explicit id so
63
+ // a claude pick adopts it — but never mint one: minting would suppress the
64
+ // --emit-session-id marker a non-claude pick needs to register its session.
65
+ if (agent === 'auto')
66
+ return sessionId;
67
+ return undefined;
59
68
  }
60
69
  /**
61
70
  * Dispatch a headless prompt run onto a resolved host, then relate the run's
@@ -77,7 +86,10 @@ export async function dispatchPromptToHost(host, opts) {
77
86
  // Ask the remote to print its resolved id whenever we did NOT force one (every
78
87
  // non-Claude agent, and Claude-on-resume where the id is already known). No-op
79
88
  // when the run isn't followed — nothing tails the log to catch the marker.
80
- const emitSessionId = !forcedSessionId && !opts.resume && opts.follow !== false;
89
+ // `run auto` with an explicit --session-id is the exception: the id is only
90
+ // ADOPTED when the remote picks claude, so a non-claude pick must still emit
91
+ // the marker for its own coined id.
92
+ const emitSessionId = (!forcedSessionId || opts.agent === 'auto') && !opts.resume && opts.follow !== false;
81
93
  const result = await dispatchToHost(host, {
82
94
  agent: opts.agent,
83
95
  prompt: opts.prompt,
@@ -0,0 +1,54 @@
1
+ /**
2
+ * Cost tiers for model selection: cheap / default / best / ultra.
3
+ *
4
+ * An orchestrating agent picks a teammate's model by a stable, cost-first tier
5
+ * instead of a concrete id that churns per release and varies per harness. A
6
+ * tier resolves, per (harness, installed version), to a model that version
7
+ * actually ships — so `--model cheap|default|best|ultra` works on `agents run`
8
+ * and `agents teams add` alike, funnelling through `resolveModel()`.
9
+ *
10
+ * Ranking signal, in priority (see apps/cli/docs — model ranking mechanisms):
11
+ * 1. Provider-declared lineup — the catalog's own family names / descriptions
12
+ * (opus/sonnet/haiku/fable; Codex "frontier / balanced / fast"). Most
13
+ * drift-proof: the provider tells us its own ranking.
14
+ * 2. Per-token price (prices.json) — cross-check + the $/Mtok display + budget.
15
+ * 3. Size-token heuristic (nano|mini|lite|flash cheaper; pro|max|opus dearer).
16
+ * 4. Reasoning effort for single-model harnesses (Grok) — tiers steer --effort.
17
+ *
18
+ * The mechanism differs per provider and drifts across versions, so tiers always
19
+ * resolve against the installed version's own catalog.
20
+ */
21
+ import type { AgentId } from './types.js';
22
+ import { type ModelInfo } from './models.js';
23
+ /** The four cross-harness cost tiers, cheapest -> most capable. */
24
+ export declare const MODEL_TIERS: readonly ["cheap", "default", "best", "ultra"];
25
+ export type ModelTier = (typeof MODEL_TIERS)[number];
26
+ /** True if `s` is one of the four tier tokens (not a concrete model id). */
27
+ export declare function isTierToken(s: string | undefined | null): s is ModelTier;
28
+ /** How a single tier resolved for a given (agent, version). */
29
+ export interface TierResolution {
30
+ tier: ModelTier;
31
+ /** Concrete model id to forward, or null when nothing resolves (fail-safe). */
32
+ model: string | null;
33
+ /** Reasoning effort to forward, for single-model harnesses where the tier is effort, not model. */
34
+ effort?: string;
35
+ /** Set when this tier has no rung of its own: the lower tier whose model it borrowed. */
36
+ clampedFrom?: ModelTier;
37
+ /** Human note (e.g. why it clamped, or that it is a curated/subscription mapping). */
38
+ note?: string;
39
+ }
40
+ /**
41
+ * Resolve all four tiers for an (agent, version). The map is what `agents models`
42
+ * prints and what `resolveTier` indexes into.
43
+ */
44
+ export declare function resolveTierMap(agent: AgentId, version: string): Record<ModelTier, TierResolution>;
45
+ /**
46
+ * Map a harness's catalog models onto the four tiers. Pure (no catalog lookup)
47
+ * so it is directly testable with synthetic inputs. Ranks the models, collapses
48
+ * variants, buckets onto cheap/default/best/ultra, and clamps absent tiers down
49
+ * to the nearest lower one. A single-model harness maps the tiers to reasoning
50
+ * effort instead of models.
51
+ */
52
+ export declare function tierizeModels(agent: AgentId, models: ModelInfo[]): Record<ModelTier, TierResolution>;
53
+ /** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
54
+ export declare function resolveTier(agent: AgentId, version: string, tier: ModelTier): TierResolution;
@@ -0,0 +1,229 @@
1
+ import { getModelCatalog } from './models.js';
2
+ import { getModelPricing } from './pricing/index.js';
3
+ /** The four cross-harness cost tiers, cheapest -> most capable. */
4
+ export const MODEL_TIERS = ['cheap', 'default', 'best', 'ultra'];
5
+ /** True if `s` is one of the four tier tokens (not a concrete model id). */
6
+ export function isTierToken(s) {
7
+ return !!s && MODEL_TIERS.includes(s);
8
+ }
9
+ // --- single-model harnesses: the tier is reasoning effort, not a model ---------
10
+ const TIER_EFFORT = {
11
+ cheap: 'low',
12
+ default: 'medium',
13
+ best: 'high',
14
+ ultra: 'xhigh',
15
+ };
16
+ // --- Droid: no live catalog; prices in credit multipliers. Curated map, capped
17
+ // at 2x (no 4x models like Fable 5 / Fast modes). Ids are Factory -m values.
18
+ const DROID_TIERS = {
19
+ cheap: 'glm-5.2', // 0.55x (Droid Core)
20
+ default: 'kimi-k3', // 0.6x (Droid Core)
21
+ best: 'claude-opus-5', // 2x
22
+ ultra: 'claude-opus-5', // clamp to best; avoid 4x
23
+ };
24
+ /** Router / pseudo models that are not a concrete choice and never a tier target. */
25
+ const PSEUDO = /(^|[-/])(auto|auto-review|router|dynamic)([-/]|$)/i;
26
+ /** Effort / speed suffixes aggregator harnesses (Cursor) bake into ids. */
27
+ const AGGREGATOR_SUFFIX = /-(low|medium|high|xhigh|thinking|fast|reasoning)\b/gi;
28
+ /** Anthropic capability family -> rank (cheapest 0 -> dearest 3). */
29
+ function anthropicFamilyRank(id) {
30
+ if (/(^|[-/])claude-haiku|(^|[-/])haiku/.test(id))
31
+ return 0;
32
+ if (/claude-sonnet|(^|[-/])sonnet/.test(id))
33
+ return 1;
34
+ if (/claude-opus|(^|[-/])opus/.test(id))
35
+ return 2;
36
+ if (/claude-(fable|mythos)|(^|[-/])(fable|mythos)/.test(id))
37
+ return 3;
38
+ return null;
39
+ }
40
+ /** Rank from the provider's own description keywords (e.g. Codex Sol/Terra/Luna). */
41
+ function descriptionRank(desc) {
42
+ if (!desc)
43
+ return null;
44
+ const d = desc.toLowerCase();
45
+ if (/(fast|affordable|cost-efficient|small|cheap|mini|nano|lightweight|spark)/.test(d))
46
+ return 0;
47
+ if (/(balanced|everyday|strong|general)/.test(d))
48
+ return 1;
49
+ if (/(frontier|latest|flagship|most capable|complex|advanced|professional)/.test(d))
50
+ return 2;
51
+ return null;
52
+ }
53
+ /** Last-resort ordering from size tokens embedded in the id. */
54
+ function sizeTokenRank(id) {
55
+ if (/(nano|mini|lite|flash|highspeed|small|air|spark)/.test(id))
56
+ return 0;
57
+ if (/(pro|max|ultra|opus|sol|large|frontier|heavy|thinking)/.test(id))
58
+ return 2;
59
+ return 1;
60
+ }
61
+ const blended = (id) => {
62
+ const p = getModelPricing(id);
63
+ return p ? p.inputPerToken + p.outputPerToken : null;
64
+ };
65
+ /** Strip an aggregator's effort/speed suffixes down to a base provider id. */
66
+ function normalizeAggregatorId(id) {
67
+ return id.replace(AGGREGATOR_SUFFIX, '').replace(/-+$/, '');
68
+ }
69
+ /**
70
+ * Compare two concrete ids so the newest wins within a family. Strips a trailing
71
+ * date stamp (`-20251101`), rebuild marker (`-v1`), and `-fast` first, so a dated
72
+ * `opus-4-5-20251101` doesn't out-rank the genuinely newer `opus-4-8` (a bare
73
+ * `compareVersions` reads the date as a huge version component).
74
+ */
75
+ function cleanForCompare(id) {
76
+ return id
77
+ .replace(/-\d{8}(?=($|-))/, '')
78
+ .replace(/-v\d+$/, '')
79
+ .replace(/-fast$/, '');
80
+ }
81
+ /** Numeric segments of a (cleaned) id, splitting on BOTH dashes and dots. */
82
+ function versionSegments(id) {
83
+ const m = cleanForCompare(id).match(/\d+/g);
84
+ return m ? m.map((n) => parseInt(n, 10)) : [];
85
+ }
86
+ /**
87
+ * Newest concrete id within a family wins. `compareVersions` only splits on `.`,
88
+ * so it degenerates to a single `[0]` segment for a dash-separated model id and
89
+ * mis-ranks e.g. `claude-sonnet-5` below `claude-sonnet-4-6`. Compare the numeric
90
+ * segments directly instead.
91
+ */
92
+ function newer(a, b) {
93
+ const A = versionSegments(a);
94
+ const B = versionSegments(b);
95
+ for (let i = 0; i < Math.max(A.length, B.length); i++) {
96
+ const d = (A[i] ?? 0) - (B[i] ?? 0);
97
+ if (d !== 0)
98
+ return d;
99
+ }
100
+ return 0;
101
+ }
102
+ /**
103
+ * Rank a harness's catalog models cheapest -> dearest and collapse variants of
104
+ * one model to a single rung (keeping the newest concrete id). Strategy depends
105
+ * on the harness class: aggregator (Cursor) ranks by price of the normalized
106
+ * base id; single-provider harnesses rank by the provider lineup with price and
107
+ * size tokens as fallbacks.
108
+ */
109
+ function rankCatalog(agent, models) {
110
+ const usable = models.filter((m) => !PSEUDO.test(m.id));
111
+ const aggregator = agent === 'cursor';
112
+ const scored = usable.map((m) => {
113
+ const rawId = m.id;
114
+ const baseId = aggregator ? normalizeAggregatorId(rawId) : rawId;
115
+ const lc = baseId.toLowerCase();
116
+ const price = blended(baseId) ?? blended(rawId);
117
+ let rank;
118
+ let family;
119
+ if (aggregator) {
120
+ // cross-provider: price is the unifying signal; family = base id
121
+ rank = price != null ? price * 1e6 : 100 + sizeTokenRank(lc);
122
+ family = baseId;
123
+ }
124
+ else {
125
+ const fam = anthropicFamilyRank(lc);
126
+ const desc = descriptionRank(m.description);
127
+ if (fam != null) {
128
+ rank = fam;
129
+ family = `anthropic-${fam}`;
130
+ }
131
+ else if (desc != null) {
132
+ rank = desc;
133
+ family = `desc-${desc}-${baseId.replace(/[0-9].*$/, '')}`;
134
+ }
135
+ else if (price != null) {
136
+ rank = 10 + price * 1e6;
137
+ // Collapse only true re-releases of ONE model (same base id, differing
138
+ // date/rebuild suffix). Keying on price would merge two DIFFERENT models
139
+ // that happen to cost the same (e.g. gpt-5.5 and gpt-5.6-sol), dropping
140
+ // one from every tier.
141
+ family = cleanForCompare(baseId);
142
+ }
143
+ else {
144
+ rank = 20 + sizeTokenRank(lc);
145
+ family = cleanForCompare(baseId);
146
+ }
147
+ }
148
+ return { id: rawId, baseId, rank, family, price };
149
+ });
150
+ // collapse by family: keep the newest concrete id, lowest (cheapest) rank
151
+ const byFamily = new Map();
152
+ for (const s of scored) {
153
+ const prev = byFamily.get(s.family);
154
+ if (!prev) {
155
+ byFamily.set(s.family, s);
156
+ continue;
157
+ }
158
+ // keep the newer id; keep the cheaper rank
159
+ if (newer(s.id, prev.id) > 0)
160
+ prev.id = s.id;
161
+ if (s.rank < prev.rank)
162
+ prev.rank = s.rank;
163
+ if (prev.price == null && s.price != null)
164
+ prev.price = s.price;
165
+ }
166
+ return [...byFamily.values()].sort((a, b) => a.rank - b.rank || newer(b.id, a.id));
167
+ }
168
+ /** Which ranked rung each tier index maps to, collapsing when there are < 4 rungs. */
169
+ function rungIndexFor(tierIndex, n) {
170
+ return n >= 4 ? Math.round((tierIndex / 3) * (n - 1)) : Math.min(tierIndex, n - 1);
171
+ }
172
+ /**
173
+ * Resolve all four tiers for an (agent, version). The map is what `agents models`
174
+ * prints and what `resolveTier` indexes into.
175
+ */
176
+ export function resolveTierMap(agent, version) {
177
+ // Droid: curated credit-multiplier map (no live catalog).
178
+ if (agent === 'droid') {
179
+ return {
180
+ cheap: { tier: 'cheap', model: DROID_TIERS.cheap, note: 'Droid Core 0.55x' },
181
+ default: { tier: 'default', model: DROID_TIERS.default, note: 'Droid Core 0.6x' },
182
+ best: { tier: 'best', model: DROID_TIERS.best, note: '2x' },
183
+ ultra: { tier: 'ultra', model: DROID_TIERS.ultra, clampedFrom: 'best', note: 'capped at 2x (4x models excluded)' },
184
+ };
185
+ }
186
+ const catalog = getModelCatalog(agent, version);
187
+ return tierizeModels(agent, catalog?.models ?? []);
188
+ }
189
+ /**
190
+ * Map a harness's catalog models onto the four tiers. Pure (no catalog lookup)
191
+ * so it is directly testable with synthetic inputs. Ranks the models, collapses
192
+ * variants, buckets onto cheap/default/best/ultra, and clamps absent tiers down
193
+ * to the nearest lower one. A single-model harness maps the tiers to reasoning
194
+ * effort instead of models.
195
+ */
196
+ export function tierizeModels(agent, models) {
197
+ const rungs = rankCatalog(agent, models);
198
+ // Single-model harness (e.g. Grok): the tier is reasoning effort, not a model.
199
+ if (rungs.length === 1) {
200
+ const only = rungs[0].id;
201
+ const map = {};
202
+ for (const t of MODEL_TIERS)
203
+ map[t] = { tier: t, model: only, effort: TIER_EFFORT[t], note: 'single model — tier maps to reasoning effort' };
204
+ return map;
205
+ }
206
+ const n = rungs.length;
207
+ const map = {};
208
+ if (n === 0) {
209
+ // Fail-safe: no catalog -> every tier null, caller drops the --model flag.
210
+ for (const t of MODEL_TIERS)
211
+ map[t] = { tier: t, model: null };
212
+ return map;
213
+ }
214
+ // Map each tier onto a rung; a tier that shares the rung of the tier below it
215
+ // has no distinct rung of its own, so mark it clamped for an honest display.
216
+ for (let i = 0; i < MODEL_TIERS.length; i++) {
217
+ const t = MODEL_TIERS[i];
218
+ const idx = rungIndexFor(i, n);
219
+ const shared = i > 0 && rungIndexFor(i - 1, n) === idx;
220
+ map[t] = shared
221
+ ? { tier: t, model: rungs[idx].id, clampedFrom: MODEL_TIERS[i - 1], note: `no distinct ${t} rung; using ${MODEL_TIERS[i - 1]}` }
222
+ : { tier: t, model: rungs[idx].id };
223
+ }
224
+ return map;
225
+ }
226
+ /** Resolve one tier for an (agent, version). Null model => caller drops the flag. */
227
+ export function resolveTier(agent, version, tier) {
228
+ return resolveTierMap(agent, version)[tier];
229
+ }
@@ -1,4 +1,5 @@
1
1
  import type { AgentId } from './types.js';
2
+ import { type ModelPricing } from './pricing/index.js';
2
3
  /** Model identifiers per cloud provider (used by Claude's multi-cloud routing). */
3
4
  export interface ModelPerCloud {
4
5
  firstParty: string;
@@ -28,6 +29,8 @@ export interface ModelInfo {
28
29
  reasoningLevels?: ReasoningLevel[];
29
30
  /** Default reasoning level if applicable */
30
31
  defaultReasoningLevel?: string;
32
+ /** Per-token USD pricing when known (from prices.json); absent for subscription/unpriced models. */
33
+ pricing?: ModelPricing;
31
34
  }
32
35
  /** The complete model catalog for a specific (agent, version) pair. */
33
36
  export interface ModelCatalog {
@@ -15,6 +15,7 @@ import { getVersionDir, getVersionHomePath, getBinaryPath } from './versions.js'
15
15
  import { getModelsCachePath } from './state.js';
16
16
  import { agentConfigDirName } from './agents.js';
17
17
  import { resolveRunDefaults } from './run-defaults.js';
18
+ import { getModelPricing } from './pricing/index.js';
18
19
  const CACHE_PATH = getModelsCachePath();
19
20
  /**
20
21
  * Bump when the extractor logic changes shape in an incompatible way so cached
@@ -342,17 +343,12 @@ function extractClaudeCatalog(text) {
342
343
  mantle: m[6] ?? null,
343
344
  };
344
345
  }
345
- const allIds = new Set([
346
- ...Object.values(aliases),
347
- ...Object.keys(displayNames),
348
- ...Object.keys(perCloud),
349
- ]);
350
346
  const aliasReverse = {};
351
347
  for (const [a, id] of Object.entries(aliases))
352
348
  aliasReverse[id] = a;
353
349
  const defaults = new Set(Object.values(aliases));
354
- const models = Array.from(allIds)
355
- .filter((id) => /^claude-(opus|sonnet|haiku)-/.test(id))
350
+ const build = (ids) => Array.from(new Set(ids))
351
+ .filter((id) => /^claude-(opus|sonnet|haiku|fable|mythos)-/.test(id))
356
352
  .sort()
357
353
  .map((id) => ({
358
354
  id,
@@ -361,6 +357,32 @@ function extractClaudeCatalog(text) {
361
357
  isDefault: defaults.has(id),
362
358
  perCloud: perCloud[id],
363
359
  }));
360
+ // The structured maps (alias/perCloud/const) are the curated, accurate
361
+ // supported set. Prefer them.
362
+ let models = build([
363
+ ...Object.values(aliases),
364
+ ...Object.keys(displayNames),
365
+ ...Object.keys(perCloud),
366
+ ]);
367
+ // Fallback id scan. The structured maps fail on the newest native-binary
368
+ // format (verified: claude@2.1.219 leaks only a stray id, so the curated set
369
+ // is effectively empty). Only when the curated catalog is that thin do we scan
370
+ // the raw strings for canonical ids -- so an older version keeps its precise
371
+ // catalog while a newer one still gets a real catalog (incl. fable/mythos and
372
+ // the opus-5/sonnet-5 line) rather than an empty or single-model one.
373
+ if (models.length < 2) {
374
+ const scanned = new Set();
375
+ // Dash-separated segments only. Real ids are `claude-sonnet-4-6`; the dotted
376
+ // form `claude-sonnet-4.6` appears only inside the binary's own "Typo in
377
+ // model ID" troubleshooting text, so a `.`-permitting pattern would scrape a
378
+ // non-model string as if it were real.
379
+ const idRe = /claude-(?:opus|sonnet|haiku|fable|mythos)-\d+(?:-\d+)*(?:-(?:fast|v\d+))?/g;
380
+ let sm;
381
+ while ((sm = idRe.exec(text)) !== null)
382
+ scanned.add(sm[0]);
383
+ if (scanned.size >= 2)
384
+ models = build(scanned);
385
+ }
364
386
  return { models, aliases };
365
387
  }
366
388
  /**
@@ -929,6 +951,14 @@ export function getModelCatalog(agent, version) {
929
951
  else if (agent === 'grok')
930
952
  ({ models, aliases } = extractGrokCatalog(src.path));
931
953
  }
954
+ // Attach per-token pricing where the offline table knows the model, so the
955
+ // catalog carries $/token for the tier display and budgeting. Subscription /
956
+ // unknown models keep `pricing` undefined (surfaced as "--", never faked).
957
+ for (const m of models) {
958
+ const p = getModelPricing(m.id);
959
+ if (p)
960
+ m.pricing = p;
961
+ }
932
962
  const catalog = {
933
963
  agent,
934
964
  version,
@@ -1123,5 +1153,12 @@ export function buildReasoningFlags(agent, level) {
1123
1153
  const droidLevel = (normalized === 'xhigh' || normalized === 'max') ? 'high' : normalized;
1124
1154
  return ['-r', droidLevel];
1125
1155
  }
1156
+ if (agent === 'grok') {
1157
+ // Grok: `--reasoning-effort <low|medium|high>` (alias --effort). xhigh/max
1158
+ // clamp to high. This is the effort dial cost tiers steer for Grok, whose
1159
+ // catalog exposes a single model.
1160
+ const grokLevel = (normalized === 'xhigh' || normalized === 'max') ? 'high' : normalized;
1161
+ return ['--reasoning-effort', grokLevel];
1162
+ }
1126
1163
  return [];
1127
1164
  }
@@ -0,0 +1,12 @@
1
+ /**
2
+ * Percentile of a sorted-ascending array, linear interpolation. p in [0,100].
3
+ *
4
+ * Deliberately its own file with zero imports: `perf/db.ts` (the SQLite
5
+ * warehouse), `hooks/profile.ts` (the legacy JSONL hook profile), and
6
+ * `routines.ts` (routineStats) all need this exact formula, but `routines.ts`
7
+ * and `hooks/profile.ts` must NOT pull in `perf/db.ts`'s `../sqlite.js`
8
+ * dependency just to round a percentile — sqlite is a heavier, perf-warehouse-
9
+ * specific dependency that has no business loading into every routines- or
10
+ * hooks-touching code path.
11
+ */
12
+ export declare function percentile(sorted: number[], p: number): number;
@@ -0,0 +1,24 @@
1
+ /**
2
+ * Percentile of a sorted-ascending array, linear interpolation. p in [0,100].
3
+ *
4
+ * Deliberately its own file with zero imports: `perf/db.ts` (the SQLite
5
+ * warehouse), `hooks/profile.ts` (the legacy JSONL hook profile), and
6
+ * `routines.ts` (routineStats) all need this exact formula, but `routines.ts`
7
+ * and `hooks/profile.ts` must NOT pull in `perf/db.ts`'s `../sqlite.js`
8
+ * dependency just to round a percentile — sqlite is a heavier, perf-warehouse-
9
+ * specific dependency that has no business loading into every routines- or
10
+ * hooks-touching code path.
11
+ */
12
+ export function percentile(sorted, p) {
13
+ if (sorted.length === 0)
14
+ return 0;
15
+ if (sorted.length === 1)
16
+ return sorted[0];
17
+ const rank = (p / 100) * (sorted.length - 1);
18
+ const lo = Math.floor(rank);
19
+ const hi = Math.ceil(rank);
20
+ if (lo === hi)
21
+ return sorted[lo];
22
+ const frac = rank - lo;
23
+ return sorted[lo] * (1 - frac) + sorted[hi] * frac;
24
+ }
@@ -8,14 +8,13 @@ import Database from '../sqlite.js';
8
8
  import type { AggregateOptions, PerfAggregateRow } from './types.js';
9
9
  export type { AggregateOptions, PerfAggregateRow, PerfSample } from './types.js';
10
10
  export { recordSample, shortSessionId, resolveSpoolPath } from './spool.js';
11
+ export { percentile } from '../percentile.js';
11
12
  export declare const PERF_SCHEMA_VERSION = 1;
12
13
  export declare const DEFAULT_RETENTION_DAYS = 30;
13
14
  /** Test seam — redirect the warehouse path (like AGENTS_EVENTS_PATH). */
14
15
  export declare function _resetPerfDbForTest(overridePath?: string | null): void;
15
16
  /** Drain the NDJSON spool into samples. Idempotent; truncates on success. */
16
17
  export declare function drainSpool(db?: Database.Database): number;
17
- /** Percentile of a sorted-ascending array. p in [0,100]. */
18
- export declare function percentile(sorted: number[], p: number): number;
19
18
  /**
20
19
  * Aggregate samples by (kind, label) with p50/p95/p99. Drains the spool first.
21
20
  *
@@ -10,8 +10,10 @@ import Database from '../sqlite.js';
10
10
  import { getPerfDbPath, getPerfDir } from '../state.js';
11
11
  import { localMachineId } from '../session/origin-machine.js';
12
12
  import { resolveProjectKey } from '../project-key.js';
13
+ import { percentile } from '../percentile.js';
13
14
  import { resolveSpoolPath, shortSessionId, _resetPerfSpoolForTest } from './spool.js';
14
15
  export { recordSample, shortSessionId, resolveSpoolPath } from './spool.js';
16
+ export { percentile } from '../percentile.js';
15
17
  export const PERF_SCHEMA_VERSION = 1;
16
18
  export const DEFAULT_RETENTION_DAYS = 30;
17
19
  const SCHEMA = `
@@ -186,20 +188,6 @@ function maybeRetain(db) {
186
188
  // ignore
187
189
  }
188
190
  }
189
- /** Percentile of a sorted-ascending array. p in [0,100]. */
190
- export function percentile(sorted, p) {
191
- if (sorted.length === 0)
192
- return 0;
193
- if (sorted.length === 1)
194
- return sorted[0];
195
- const rank = (p / 100) * (sorted.length - 1);
196
- const lo = Math.floor(rank);
197
- const hi = Math.ceil(rank);
198
- if (lo === hi)
199
- return sorted[lo];
200
- const frac = rank - lo;
201
- return sorted[lo] * (1 - frac) + sorted[hi] * frac;
202
- }
203
191
  /**
204
192
  * Aggregate samples by (kind, label) with p50/p95/p99. Drains the spool first.
205
193
  *