claude-usage-limits 1.9.2 → 1.11.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -23,6 +23,7 @@ const readline = require('readline');
23
23
 
24
24
  const host = require('./host.js');
25
25
  const codex = require('./codex.js');
26
+ const live = require('./live.js');
26
27
 
27
28
  // Which agent's meter to read. Resolved once from the command line or the
28
29
  // environment, because a process that changed its mind halfway through would
@@ -69,6 +70,11 @@ const RATES = {
69
70
  // An unrecognised family falls back to Opus rates on purpose: over-estimating
70
71
  // cost understates headroom, and that is the safe direction for a budget.
71
72
  const FAMILIES = ['fable', 'mythos', 'opus', 'sonnet', 'haiku'];
73
+
74
+ // Settings that name a strategy rather than a model. `opusplan` plans on Opus
75
+ // and executes on Sonnet, so it spends into both families and neither of them
76
+ // is what a substring match would find on its own.
77
+ const MODEL_ALIASES = { opusplan: ['opus', 'sonnet'] };
72
78
  const FALLBACK_RATE = { input: 5, output: 25 };
73
79
 
74
80
  function familyOf(model) {
@@ -117,13 +123,85 @@ const CACHE_WRITE_5M = 1.25;
117
123
  const CACHE_WRITE_1H = 2;
118
124
  const CACHE_READ = 0.1;
119
125
 
126
+ // `family` marks a window that caps one model family rather than the account
127
+ // as a whole. It is what tells the rest of the file that a window cannot stop
128
+ // work which does not use that family.
120
129
  const WINDOWS = [
121
130
  { key: 'five_hour', label: '5-hour', span: 5 * HOUR },
122
131
  { key: 'seven_day', label: 'weekly', span: 7 * DAY },
123
- { key: 'seven_day_opus', label: 'weekly (Opus)', span: 7 * DAY },
124
- { key: 'seven_day_sonnet', label: 'weekly (Sonnet)', span: 7 * DAY },
132
+ { key: 'seven_day_opus', label: 'weekly (Opus)', span: 7 * DAY, family: 'opus' },
133
+ { key: 'seven_day_sonnet', label: 'weekly (Sonnet)', span: 7 * DAY, family: 'sonnet' },
125
134
  ];
126
135
 
136
+ // Which model families this agent can actually spend into.
137
+ //
138
+ // A per-model weekly caps one family's spend and nothing else, so it can only
139
+ // ever stop work that uses that family. Reported without that qualification it
140
+ // becomes the loudest number in the brief for a limit the session cannot move:
141
+ // a session running Opus was told to weigh a Fable weekly at 88%, and no
142
+ // amount of work it did would have moved it a single point.
143
+ //
144
+ // The configured model is the floor. A session's own turns are added on top,
145
+ // because subagents and a mid-session /model both spend into families the
146
+ // setting never mentions. Other sessions' turns are deliberately not counted:
147
+ // what another window is burning is not this one's constraint.
148
+ //
149
+ // An empty set means the model could not be worked out at all, and nothing is
150
+ // suppressed on the strength of a guess.
151
+ function familiesInUse(events, sessionId, models) {
152
+ const families = new Set();
153
+ const hints = Array.isArray(models) ? models : [models];
154
+ for (const hint of hints) {
155
+ const name = normalizeModel(hint);
156
+ // Some settings name more than one model. `opusplan` plans on Opus and
157
+ // executes on Sonnet, so a session set to it spends into both, and reading
158
+ // only the first would suppress a Sonnet weekly while Sonnet is running.
159
+ const alias = MODEL_ALIASES[name];
160
+ if (alias) {
161
+ for (const family of alias) families.add(family);
162
+ continue;
163
+ }
164
+ const family = familyOf(name);
165
+ if (family) families.add(family);
166
+ }
167
+ // This session's own turns only. What another window is burning is not this
168
+ // one's constraint, and counting it is how a weekly for a model this agent
169
+ // never runs gets weighed against work that cannot move it. With no session
170
+ // to scan - the CLI report - the configured model is the whole answer, and
171
+ // when that says nothing usable, nothing is suppressed.
172
+ if (sessionId) {
173
+ for (const event of events || []) {
174
+ if (!event || event.sessionId !== sessionId) continue;
175
+ const family = familyOf(event.model);
176
+ if (family) families.add(family);
177
+ }
178
+ }
179
+ return families;
180
+ }
181
+
182
+ // Whether a window is one this agent can spend into. Windows that cap the whole
183
+ // account always are; a per-model one only when that model is in use.
184
+ function appliesTo(window, families) {
185
+ if (!window || !window.family) return true;
186
+ if (!families || !families.size) return true;
187
+ // Only ever suppress on a family this file recognises on both sides. A
188
+ // scoped weekly names its model by display name, and one for a model
189
+ // released after this table was written falls back to that raw name - which
190
+ // familyOf() will never return for the setting either, so the window would
191
+ // be suppressed permanently, including while it is the thing being spent.
192
+ if (FAMILIES.indexOf(window.family) === -1) return true;
193
+ return families.has(window.family);
194
+ }
195
+
196
+ // Stamps the answer onto each window so every reader - the binding choice, the
197
+ // critical warning, the report table - makes the same call from the same field.
198
+ function markApplicable(windows, families) {
199
+ for (const window of windows || []) {
200
+ if (window) window.applies = appliesTo(window, families);
201
+ }
202
+ return windows;
203
+ }
204
+
127
205
  // organizationType gives the family; the rate limit tier is what separates
128
206
  // Max 5x from Max 20x. Both come out of oauthAccount.
129
207
  // Abbreviations for the status line, where there is no room to spell it out.
@@ -428,6 +506,29 @@ function freshFiles(dir, since) {
428
506
  .filter((file) => fresh(file, since));
429
507
  }
430
508
 
509
+ // Every transcript under a session's subagents directory. Plain subagents
510
+ // write straight into it; the agents a Workflow runs write under
511
+ // subagents/workflows/<run id>/, and eight of those spending in parallel is
512
+ // exactly the burst that empties a window between two readings, so they must
513
+ // be counted. Two levels down is as deep as Claude Code goes today; a bounded
514
+ // walk copes if that changes.
515
+ function subagentTranscripts(dir, since, depth) {
516
+ const left = Number.isFinite(depth) ? depth : 3;
517
+ const files = freshFiles(dir, since);
518
+ if (left <= 0) return files;
519
+ let entries;
520
+ try {
521
+ entries = fs.readdirSync(dir, { withFileTypes: true });
522
+ } catch (err) {
523
+ return files;
524
+ }
525
+ for (const entry of entries) {
526
+ if (!entry.isDirectory()) continue;
527
+ for (const file of subagentTranscripts(path.join(dir, entry.name), since, left - 1)) files.push(file);
528
+ }
529
+ return files;
530
+ }
531
+
431
532
  async function readClaudeEvents(since) {
432
533
  const root = path.join(configDir(), 'projects');
433
534
  let dirs = [];
@@ -452,7 +553,7 @@ async function readClaudeEvents(since) {
452
553
  // A session's subagents write their transcripts under
453
554
  // <project>/<session id>/subagents/. Same budget, different file, and
454
555
  // for a long time an Explore or Plan agent's whole spend went unseen.
455
- for (const file of freshFiles(path.join(full, entry.name, 'subagents'), since)) {
556
+ for (const file of subagentTranscripts(path.join(full, entry.name, 'subagents'), since)) {
456
557
  files.push({ file, project: dir.name });
457
558
  }
458
559
  continue;
@@ -701,8 +802,8 @@ function creditsFrom(utilization) {
701
802
  // Turn cost is not a single number, it is a spread: a turn that reads three
702
803
  // files costs many times one that answers from context. A median alone
703
804
  // under-promises on the expensive half, so carry a high end too.
704
- function costPercentiles(events) {
705
- const costs = mainThread(events)
805
+ function callPercentiles(events) {
806
+ const costs = (events || [])
706
807
  .map((event) => event.cost)
707
808
  .filter((cost) => Number.isFinite(cost) && cost > 0)
708
809
  .sort((a, b) => a - b);
@@ -712,6 +813,124 @@ function costPercentiles(events) {
712
813
  return { median: at(0.5), high: at(0.8), sample: costs.length };
713
814
  }
714
815
 
816
+ function costPercentiles(events) {
817
+ return callPercentiles(mainThread(events));
818
+ }
819
+
820
+ // What a turn of each model family has actually cost on this machine.
821
+ //
822
+ // The account's snapshot has no model dimension at all. It says a window is at
823
+ // 88 per cent and never says whose turns put it there, so the report could say
824
+ // how much room was left and not what that room would buy. Every transcript
825
+ // line carries its model, which is the half the snapshot is missing.
826
+ //
827
+ // Cost is measured over main-thread turns, because a turn of headroom means a
828
+ // main-thread turn everywhere else in this file. A family that has only ever
829
+ // run as a subagent has none to measure - Sonnet on this machine had 114 calls
830
+ // and not one turn - and pricing it at nothing would hand back an unlimited
831
+ // budget, so it is priced per call instead and the row says so.
832
+ function modelSpend(events) {
833
+ const byFamily = new Map();
834
+ for (const event of events || []) {
835
+ const family = familyOf(event.model);
836
+ if (!family) continue;
837
+ if (!byFamily.has(family)) byFamily.set(family, []);
838
+ byFamily.get(family).push(event);
839
+ }
840
+
841
+ const rows = [];
842
+ for (const [family, own] of byFamily) {
843
+ const spent = totals(own);
844
+ const main = costPercentiles(own);
845
+ const rates = main || callPercentiles(own);
846
+ rows.push({
847
+ family,
848
+ calls: own.length,
849
+ turns: spent.turns,
850
+ usd: spent.cost,
851
+ tokens: spent.tokens,
852
+ usdPerTurn: rates ? rates.median : null,
853
+ sample: rates ? rates.sample : 0,
854
+ // True when the figure prices a subagent call rather than a turn.
855
+ perCall: !main,
856
+ });
857
+ }
858
+ return rows.sort((a, b) => b.usd - a.usd);
859
+ }
860
+
861
+ // Which window a family's spend lands in: its own weekly where the account
862
+ // gives it one, and the shared weekly otherwise.
863
+ function windowForFamily(windows, family) {
864
+ const own = (windows || []).find((w) => w && w.family === family);
865
+ if (own) return own;
866
+ return (windows || []).find((w) => w && w.key === 'seven_day') || null;
867
+ }
868
+
869
+ // Too few turns to price one. The figure is divided into the whole remaining
870
+ // budget, so an error in it is multiplied up rather than averaged away.
871
+ const MIN_MODEL_SAMPLE = 5;
872
+
873
+ // What the room that is left buys, counted in turns of each model.
874
+ //
875
+ // Families that share the weekly window are alternatives, not additions: each
876
+ // row says what the same remaining room would buy if it all went on that model.
877
+ // A family with a weekly of its own is the exception, and it is the case worth
878
+ // knowing about, because it is the one where changing model changes which wall
879
+ // the work is walking towards.
880
+ function modelHeadroom(windows, events, families, remembered) {
881
+ return modelSpend(events).map((row) => {
882
+ const window = windowForFamily(windows, row.family);
883
+ const learned = remembered && remembered[row.family];
884
+ // A handful of turns prices a turn badly. What was measured when there was
885
+ // a proper sample is better evidence than what this week happens to hold.
886
+ const thin = row.sample < MIN_MODEL_SAMPLE;
887
+ const canRemember =
888
+ thin && Boolean(learned) && Number.isFinite(learned.usdPerTurn) && learned.usdPerTurn > 0 &&
889
+ // A remembered per-call price is not a turn price either, whatever it is
890
+ // worth for delegation. See below.
891
+ !learned.perCall;
892
+ const usdPerTurn = canRemember ? learned.usdPerTurn : row.usdPerTurn;
893
+ const perCall = canRemember ? Boolean(learned.perCall) : row.perCall;
894
+ // No turn count for a model that has never taken a turn.
895
+ //
896
+ // A family that has only ever run as a subagent has errands to price, not
897
+ // turns: 114 Sonnet calls here averaged under two cents because they were
898
+ // one-shot lookups, and dividing the remaining budget by that promised
899
+ // twenty-two thousand Sonnet turns. Those turns would not be doing the work
900
+ // the Opus turns are doing, and the error is in the direction that promises
901
+ // room, which is the direction that gets a session cut off mid-edit. The
902
+ // row still says what the model has cost; it does not project from it.
903
+ //
904
+ // Nor against a window that has already rolled over. Its remaining money
905
+ // describes the allowance the stale reading was taken from, not the one
906
+ // running now, and dividing by a turn price turns that into a confident
907
+ // count of turns nobody has: a weekly past its reset offered five hundred.
908
+ const turnsLeft =
909
+ !perCall &&
910
+ window &&
911
+ !window.stale &&
912
+ Number.isFinite(window.remainingUSD) &&
913
+ Number.isFinite(usdPerTurn) &&
914
+ usdPerTurn > 0
915
+ ? Math.max(0, Math.floor(window.remainingUSD / usdPerTurn))
916
+ : null;
917
+ return Object.assign({}, row, {
918
+ windowKey: window ? window.key : null,
919
+ windowLabel: window ? window.label : null,
920
+ // True when this family has a weekly of its own rather than sharing.
921
+ ownWindow: Boolean(window && window.family),
922
+ // False when the window is one this agent cannot spend into anyway.
923
+ windowApplies: !window || window.applies !== false,
924
+ inUse: !families || !families.size || families.has(row.family),
925
+ usdPerTurn,
926
+ perCall,
927
+ // True when the price came off the record rather than this week's turns.
928
+ remembered: Boolean(canRemember),
929
+ turnsLeft,
930
+ });
931
+ });
932
+ }
933
+
715
934
  // What a job of this many turns would take out of one window.
716
935
  function forecastWindow(window, turns, rates) {
717
936
  if (!window || !rates || !window.usdPerPercent || window.stale) return null;
@@ -816,6 +1035,70 @@ function writeCalibration(all) {
816
1035
  }
817
1036
  }
818
1037
 
1038
+ // What each model has cost, kept so the next session does not have to have
1039
+ // spent anything to know.
1040
+ //
1041
+ // The transcripts only answer for as long as they are on disk and as far back
1042
+ // as the scan reaches, which is eight days. A session that opens on a model it
1043
+ // has not used this week would otherwise have no price for it at all, and what
1044
+ // Sonnet would buy is worth answering before the first Sonnet turn rather than
1045
+ // after. One entry per family, so it cannot grow.
1046
+ //
1047
+ // It is stamped with the plan and read back through the same guard as the
1048
+ // window calibration. What a turn costs is a fact about the model; what it buys
1049
+ // is a fact about the allowance, and that moves when the plan does.
1050
+ function modelRecordFile() {
1051
+ const dir = isCodex() ? codex.homeDir() : configDir();
1052
+ return path.join(dir, 'usage-limits-models.json');
1053
+ }
1054
+
1055
+ function readModelRecord() {
1056
+ try {
1057
+ const parsed = JSON.parse(fs.readFileSync(modelRecordFile(), 'utf8'));
1058
+ return parsed && typeof parsed === 'object' && !Array.isArray(parsed) ? parsed : {};
1059
+ } catch (err) {
1060
+ return {};
1061
+ }
1062
+ }
1063
+
1064
+ function writeModelRecord(all) {
1065
+ try {
1066
+ fs.mkdirSync(path.dirname(modelRecordFile()), { recursive: true });
1067
+ fs.writeFileSync(modelRecordFile(), JSON.stringify(all), 'utf8');
1068
+ } catch (err) {
1069
+ // The record is a convenience for a thin week, not a source of truth.
1070
+ }
1071
+ }
1072
+
1073
+ // The measurement to keep. Deliberately the freshest adequate one rather than
1074
+ // the largest: what a turn costs drifts as a session's context grows, so an old
1075
+ // figure with a big sample behind it is not the better answer, only the better
1076
+ // attested one.
1077
+ function modelSamples(headroom, previous) {
1078
+ const kept = Object.assign({}, previous || {});
1079
+ let changed = false;
1080
+ for (const row of headroom || []) {
1081
+ if (row.remembered) continue;
1082
+ if (!Number.isFinite(row.usdPerTurn) || row.usdPerTurn <= 0) continue;
1083
+ if (row.sample < MIN_MODEL_SAMPLE) continue;
1084
+ // Rounded before comparing, so a fraction of a cent of drift does not
1085
+ // rewrite the file on every prompt.
1086
+ const usdPerTurn = Math.round(row.usdPerTurn * 1e6) / 1e6;
1087
+ const before = kept[row.family];
1088
+ if (
1089
+ before &&
1090
+ before.usdPerTurn === usdPerTurn &&
1091
+ before.perCall === Boolean(row.perCall) &&
1092
+ before.sample === row.sample
1093
+ ) {
1094
+ continue;
1095
+ }
1096
+ kept[row.family] = { usdPerTurn, sample: row.sample, perCall: Boolean(row.perCall) };
1097
+ changed = true;
1098
+ }
1099
+ return { models: kept, changed };
1100
+ }
1101
+
819
1102
  // Everything learned about a budget belongs to the plan it was learned on.
820
1103
  //
821
1104
  // A point of a window is a share of an allowance, so changing the allowance
@@ -946,7 +1229,14 @@ function buildWindow(spec, snapshot, events, now, options) {
946
1229
  severity: null,
947
1230
  isActive: false,
948
1231
  scoped: false,
949
- family: null,
1232
+ // The model family this window caps, when it caps one. Set from the spec
1233
+ // here for the bucket-table weeklies, and again by the caller for the
1234
+ // per-model limits that only exist in the account's own `limits` list.
1235
+ family: spec.family || null,
1236
+ // Whether this session's models spend into it. Filled in by
1237
+ // markApplicable once the models in use are known; assume they do until
1238
+ // then, so a reader that never marks them behaves exactly as before.
1239
+ applies: true,
950
1240
  verdict: 'unknown',
951
1241
  };
952
1242
 
@@ -1149,6 +1439,10 @@ function criticalOthers(windows, bindingKey, threshold) {
1149
1439
  w &&
1150
1440
  w.key !== bindingKey &&
1151
1441
  !w.stale &&
1442
+ // A full window that this session cannot spend into is not a warning, it
1443
+ // is someone else's news. Told to weigh it, the only thing an agent can
1444
+ // do about it is less work, against a limit its work never touches.
1445
+ w.applies !== false &&
1152
1446
  w.percentUsed !== null &&
1153
1447
  w.percentUsed >= limit
1154
1448
  );
@@ -1156,7 +1450,18 @@ function criticalOthers(windows, bindingKey, threshold) {
1156
1450
 
1157
1451
  // The window that will stop the work first.
1158
1452
  function bindingWindow(windows) {
1159
- const known = windows.filter((w) => w.percentUsed !== null);
1453
+ // A per-model weekly for a model that is not running cannot be the window
1454
+ // that stops the work, however full it is - and the account's own is_active
1455
+ // flag says nothing about which model this session happens to be using, so
1456
+ // it must not promote one either. Kept as a fallback in the impossible case
1457
+ // that every window is a model's, so this never returns nothing.
1458
+ // Ordered this way round on purpose: the fallback has to fire when there is
1459
+ // no readable window left after suppression, not merely no window. Testing
1460
+ // the unfiltered list first returned nothing at all where a suppressed
1461
+ // per-model weekly was the only window carrying a reading.
1462
+ const readable = windows.filter((w) => w && w.percentUsed !== null);
1463
+ const relevant = readable.filter((w) => w.applies !== false);
1464
+ const known = relevant.length ? relevant : readable;
1160
1465
  // Prefer windows we can still trust; fall back only if every one is stale.
1161
1466
  const fresh = known.filter((w) => !w.stale);
1162
1467
  const live = fresh.length ? fresh : known;
@@ -1345,12 +1650,39 @@ function collect(now) {
1345
1650
  return collectClaude(now);
1346
1651
  }
1347
1652
 
1653
+ // Whether the plugin's own live reading should stand in for Claude Code's
1654
+ // cache. Both describe the same account; the newer one is simply the more
1655
+ // recent fact. A reading for a different account, or one stamped from a clock
1656
+ // that is ahead, is not a fresher reading of this account.
1657
+ function preferLive(cache, fresh, accountUuid, now) {
1658
+ if (!fresh || !fresh.utilization || typeof fresh.utilization !== 'object') return false;
1659
+ if (!Number.isFinite(fresh.fetchedAtMs)) return false;
1660
+ if (fresh.fetchedAtMs > now + MINUTE) return false;
1661
+ if (fresh.accountUuid && accountUuid && fresh.accountUuid !== accountUuid) return false;
1662
+ if (!cache || !Number.isFinite(cache.fetchedAtMs)) return true;
1663
+ return fresh.fetchedAtMs > cache.fetchedAtMs;
1664
+ }
1665
+
1666
+ // The account the login belongs to, so a live reading can be stamped with it.
1667
+ function accountUuid() {
1668
+ const account = readJson(accountFile());
1669
+ return account && account.oauthAccount && account.oauthAccount.accountUuid
1670
+ ? account.oauthAccount.accountUuid
1671
+ : null;
1672
+ }
1673
+
1348
1674
  function collectClaude(now) {
1349
1675
  const account = readJson(accountFile()) || {};
1350
1676
  const settings = readJson(path.join(configDir(), 'settings.json')) || {};
1351
1677
  const cache = account.cachedUsageUtilization || null;
1352
- const utilization = cache && cache.utilization ? cache.utilization : null;
1353
1678
  const oauth = account.oauthAccount || {};
1679
+ // The panel and the status line take the same reading Claude Code takes for
1680
+ // /usage and keep it in a file of their own. When that is newer than what
1681
+ // Claude Code cached, it is the better description of the same account.
1682
+ const fresh = live.readLive();
1683
+ const useLive = preferLive(cache, fresh, oauth.accountUuid, now);
1684
+ const snapshot = useLive ? fresh : cache;
1685
+ const utilization = snapshot && snapshot.utilization ? snapshot.utilization : null;
1354
1686
  const plan = detectPlan(oauth);
1355
1687
 
1356
1688
  return {
@@ -1362,8 +1694,9 @@ function collectClaude(now) {
1362
1694
  planId: plan.id,
1363
1695
  planTier: plan.tier,
1364
1696
  planAdvice: plan.advice,
1365
- snapshotAgeMs: cache && cache.fetchedAtMs ? now - cache.fetchedAtMs : null,
1366
- snapshotFetchedAt: cache && cache.fetchedAtMs ? cache.fetchedAtMs : null,
1697
+ snapshotAgeMs: snapshot && snapshot.fetchedAtMs ? now - snapshot.fetchedAtMs : null,
1698
+ snapshotFetchedAt: snapshot && snapshot.fetchedAtMs ? snapshot.fetchedAtMs : null,
1699
+ snapshotSource: utilization ? (useLive ? 'live' : 'cache') : null,
1367
1700
  utilization,
1368
1701
  settings: {
1369
1702
  model: settings.model || 'default',
@@ -1669,6 +2002,28 @@ async function report(now, options) {
1669
2002
  calibrated.planChanged ? new Map() : rejections
1670
2003
  );
1671
2004
 
2005
+ // Which of those windows this agent can actually spend into. Everything that
2006
+ // ranks or warns about a window reads the answer off the window itself.
2007
+ const families = familiesInUse(
2008
+ events,
2009
+ options && options.sessionId,
2010
+ [base.settings && base.settings.model, process.env.ANTHROPIC_MODEL]
2011
+ );
2012
+ markApplicable(windows, families);
2013
+
2014
+ // What the room left buys in turns of each model, and the record that lets a
2015
+ // later session answer that for a model it has not run yet. Scoped to a week
2016
+ // because every window a family draws on here is a weekly one.
2017
+ const remembered = calibrationForPlan(readModelRecord(), base.planId).learned;
2018
+ const headroom = modelHeadroom(
2019
+ windows,
2020
+ events.filter((event) => event.at >= now - 7 * DAY),
2021
+ families,
2022
+ remembered
2023
+ );
2024
+ const sampled = modelSamples(headroom, remembered);
2025
+ if (sampled.changed) writeModelRecord(stampPlan(sampled.models, base.planId));
2026
+
1672
2027
  // Keep the best sample seen so far, so a thin baseline never has to guess.
1673
2028
  const updated = Object.assign({}, learned);
1674
2029
  for (const window of windows) {
@@ -1710,10 +2065,14 @@ async function report(now, options) {
1710
2065
  credits: base.codexCredits || creditsFrom(base.utilization),
1711
2066
  sessions: activeSessions(events, now, CONCURRENT_WINDOW_MS),
1712
2067
  session: sessionSpend(events, options && options.sessionId),
1713
- staleWindows: windows.filter((w) => w.stale).length,
2068
+ // A per-model weekly for a model that is not running could be a week past
2069
+ // its reset without that saying anything about the numbers this agent is
2070
+ // working from, and it should not put "run /usage" on every prompt.
2071
+ staleWindows: windows.filter((w) => w.stale && w.applies !== false).length,
1714
2072
  rates: costPercentiles(recentEvents.length >= 5 ? recentEvents : scoped),
1715
2073
  resumeAt: binding ? binding.resetsAt : null,
1716
2074
  models: scopedModels,
2075
+ modelHeadroom: headroom,
1717
2076
  projects: byProject(scoped),
1718
2077
  tokens: scopedTotals.parts,
1719
2078
  reasoning: reasoningSpend(scopedModels, scopedTotals.parts),
@@ -1780,6 +2139,13 @@ function statusLine(collected) {
1780
2139
  if (!utilization) return '';
1781
2140
 
1782
2141
  const now = collected.now || Date.now();
2142
+ // No transcripts here on purpose, so the only thing that says which model is
2143
+ // running is the setting. That is enough to keep a weekly for a model this
2144
+ // agent is not using out of a line that is meant to read as "your room".
2145
+ const families = familiesInUse(null, null, [
2146
+ collected.settings && collected.settings.model,
2147
+ process.env.ANTHROPIC_MODEL,
2148
+ ]);
1783
2149
  const parts = [];
1784
2150
  for (const spec of collected.windowSpecs && collected.windowSpecs.length
1785
2151
  ? collected.windowSpecs
@@ -1792,6 +2158,11 @@ function statusLine(collected) {
1792
2158
  label: SHORT_LABELS[spec.key] || spec.label,
1793
2159
  percent: snapshot.utilization,
1794
2160
  msToReset,
2161
+ // A per-model weekly for a model that is not running is shown - hiding a
2162
+ // limit outright is the one failure worse than over-reporting one, and
2163
+ // the only thing telling this line which model is running is the
2164
+ // setting, which can be behind. It just does not raise the alarm.
2165
+ idle: !appliesTo(spec, families),
1795
2166
  stale: msToReset !== null && msToReset <= 0,
1796
2167
  // Zero with no reset time is not an empty window, it is a bucket that is
1797
2168
  // not reporting: a real window at 0% has just reset and says when it will
@@ -1814,6 +2185,7 @@ function statusLine(collected) {
1814
2185
  label: limit.family,
1815
2186
  percent: limit.percent,
1816
2187
  msToReset,
2188
+ idle: !appliesTo(limit, families),
1817
2189
  stale: msToReset !== null && msToReset <= 0,
1818
2190
  unreported: false,
1819
2191
  });
@@ -1821,7 +2193,11 @@ function statusLine(collected) {
1821
2193
 
1822
2194
  if (!parts.length) return '';
1823
2195
 
1824
- const trusted = parts.filter((part) => !part.stale && !part.unreported);
2196
+ // Shown, but never the alarm. Hiding a limit outright is the one failure
2197
+ // worse than over-reporting one, and the only thing telling this line which
2198
+ // model is running is the setting, which can be behind a /model. So an idle
2199
+ // per-model weekly stays on the line and is left out of the worst-of.
2200
+ const trusted = parts.filter((part) => !part.stale && !part.unreported && !part.idle);
1825
2201
  const worst = trusted.length
1826
2202
  ? trusted.reduce((a, b) => (b.percent > a.percent ? b : a))
1827
2203
  : null;
@@ -1850,7 +2226,9 @@ function render(data) {
1850
2226
  lines.push(' Plan ' + data.plan);
1851
2227
  lines.push(
1852
2228
  ' Snapshot ' +
1853
- (data.snapshotAgeMs === null ? 'none on disk' : formatDuration(data.snapshotAgeMs) + ' old')
2229
+ (data.snapshotAgeMs === null
2230
+ ? 'none on disk'
2231
+ : formatDuration(data.snapshotAgeMs) + ' old' + (data.snapshotSource === 'live' ? ' (live reading)' : ''))
1854
2232
  );
1855
2233
  lines.push(' Settings model=' + data.settings.model + ' effort=' + data.settings.effortLevel);
1856
2234
  if (data.planChanged) {
@@ -1913,11 +2291,18 @@ function render(data) {
1913
2291
  const bound = data.binding && window.key === data.binding.key;
1914
2292
  // The account's own severity, when it says critical, is worth a word.
1915
2293
  const critical = window.severity === 'critical';
2294
+ // A per-model weekly for a model that is not running still belongs in the
2295
+ // table - it is real, and switching to that model would make it bite - but
2296
+ // it is not this agent's room, and a bare percentage next to the others
2297
+ // reads as though it were. The row says which it is.
2298
+ const idle = window.applies === false;
1916
2299
  const marker = bound
1917
2300
  ? ' <- binding' + (critical ? ', critical' : '')
1918
- : critical
1919
- ? ' critical'
1920
- : '';
2301
+ : idle
2302
+ ? ' not in use' + (critical ? ', critical for that model' : '')
2303
+ : critical
2304
+ ? ' critical'
2305
+ : '';
1921
2306
  lines.push(
1922
2307
  ' ' + pad(window.label, 15) +
1923
2308
  padLeft(
@@ -1934,8 +2319,62 @@ function render(data) {
1934
2319
  marker
1935
2320
  );
1936
2321
  }
2322
+ if (data.windows.some((window) => window.applies === false)) {
2323
+ // Named from what is actually running rather than from the setting, which
2324
+ // can say 'default', or name a strategy like 'opusplan', or be behind a
2325
+ // /model - and the legend would then assert a model nothing was decided
2326
+ // from.
2327
+ const running = (data.modelHeadroom || [])
2328
+ .filter((row) => row.inUse)
2329
+ .map((row) => row.family);
2330
+ lines.push(
2331
+ ' not in use caps one model, and this agent is running ' +
2332
+ (running.length ? running.join(' and ') : 'another model') +
2333
+ ', so nothing here spends into it'
2334
+ );
2335
+ }
1937
2336
  lines.push('');
1938
2337
 
2338
+ // How much room is left is only half the question. The other half is what
2339
+ // that room buys, and the answer is different for every model: the same
2340
+ // weekly holds a few hundred Fable turns or several thousand Sonnet ones.
2341
+ // The account's own figures cannot say this - they have no model in them.
2342
+ const headroom = money ? (data.modelHeadroom || []) : [];
2343
+ if (headroom.some((row) => row.turnsLeft !== null)) {
2344
+ lines.push(' Model headroom, what the room left buys in turns of each model');
2345
+ lines.push(
2346
+ ' ' + pad(' Model', 12) + pad('Window', 17) + padLeft('Turns', 7) +
2347
+ padLeft('Spent', 9) + padLeft('Per turn', 10) + padLeft('Turns left', 12)
2348
+ );
2349
+ for (const row of headroom) {
2350
+ lines.push(
2351
+ ' ' + pad(' ' + row.family, 12) +
2352
+ pad(row.windowLabel || '-', 17) +
2353
+ padLeft(formatCount(row.sample), 7) +
2354
+ padLeft(formatUSD(row.usd), 9) +
2355
+ padLeft(formatUSD(row.usdPerTurn) + (row.perCall ? '*' : ''), 10) +
2356
+ padLeft(row.turnsLeft === null ? '-' : '~' + formatCount(row.turnsLeft), 12) +
2357
+ (row.inUse ? ' <- running' : '')
2358
+ );
2359
+ }
2360
+ if (headroom.some((row) => row.perCall)) {
2361
+ lines.push(' * a subagent call, not a turn: this model has taken no turns of its own,');
2362
+ lines.push(' so there is nothing here to project a turn count from. The price is');
2363
+ lines.push(' still what delegating to it has cost.');
2364
+ }
2365
+ if (headroom.some((row) => row.remembered)) {
2366
+ lines.push(' A row with too few turns this week is priced from the record of what');
2367
+ lines.push(' that model cost when there were enough.');
2368
+ }
2369
+ // The two tables price the same window differently on purpose, and someone
2370
+ // is going to notice, so say why before it gets read as a bug.
2371
+ lines.push(' Turns left in the table above is a blend of every model on record;');
2372
+ lines.push(" these are each model's own measured cost per turn, and rows sharing a");
2373
+ lines.push(' window are alternatives rather than additions: the same room, spent on');
2374
+ lines.push(' a different model.');
2375
+ lines.push('');
2376
+ }
2377
+
1939
2378
  // A bucket with no span cannot be priced or projected, but saying nothing
1940
2379
  // about one that is nearly full would be the worse failure.
1941
2380
  if (data.otherLimits && data.otherLimits.length) {
@@ -2141,6 +2580,9 @@ function renderForecast(data, turns) {
2141
2580
  }
2142
2581
 
2143
2582
  const rows = data.windows
2583
+ // A limit this agent cannot spend into is not what a job fails to fit in,
2584
+ // so it is never offered as a reason to cut the work down.
2585
+ .filter((window) => window.applies !== false)
2144
2586
  .map((window) => forecastWindow(window, turns, data.rates))
2145
2587
  .filter(Boolean);
2146
2588
 
@@ -2360,7 +2802,10 @@ async function main(argv) {
2360
2802
  const turns = Number(argv[forecastAt + 1]);
2361
2803
  if (wantsJson) {
2362
2804
  const rows = data.windows
2363
- .map((window) => forecastWindow(window, turns, data.rates))
2805
+ // A limit this agent cannot spend into is not what a job fails to fit in,
2806
+ // so it is never offered as a reason to cut the work down.
2807
+ .filter((window) => window.applies !== false)
2808
+ .map((window) => forecastWindow(window, turns, data.rates))
2364
2809
  .filter(Boolean);
2365
2810
  process.stdout.write(JSON.stringify({ turns, rates: data.rates, windows: rows }, null, 2) + '\n');
2366
2811
  } else {
@@ -2402,12 +2847,27 @@ module.exports = {
2402
2847
  isCodex,
2403
2848
  otherLimits,
2404
2849
  collectClaude,
2850
+ preferLive,
2851
+ accountUuid,
2852
+ subagentTranscripts,
2405
2853
  readClaudeEvents,
2406
2854
  RATES,
2407
2855
  WINDOWS,
2408
2856
  rateFor,
2409
2857
  familyOf,
2410
2858
  familyAverage,
2859
+ familiesInUse,
2860
+ appliesTo,
2861
+ markApplicable,
2862
+ callPercentiles,
2863
+ modelSpend,
2864
+ windowForFamily,
2865
+ modelHeadroom,
2866
+ modelSamples,
2867
+ modelRecordFile,
2868
+ readModelRecord,
2869
+ writeModelRecord,
2870
+ MIN_MODEL_SAMPLE,
2411
2871
  isKnownModel,
2412
2872
  costOf,
2413
2873
  tokensOf,