claude-usage-limits 1.26.0 → 1.39.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -124,9 +124,28 @@ function slotFrom(input, previous, now) {
124
124
  cwd: typeof input.cwd === 'string' ? input.cwd : prior.cwd || null,
125
125
  version: typeof input.version === 'string' ? input.version : prior.version || null,
126
126
  fastMode: input.fast_mode === true,
127
+ cacheMiss: missFrom(input.prompt_cache, prior.cacheMiss),
127
128
  };
128
129
  }
129
130
 
131
+ // The last prompt-cache miss and its diagnosed causes, from the status line's
132
+ // prompt_cache object: last_miss_at is epoch seconds and last_miss_cause is
133
+ // null until the first miss and whenever no cause was found (Claude Code
134
+ // 2.1.260 and later; both per code.claude.com/docs/en/statusline). The moment
135
+ // is what lets the brief tell a new miss from the same one arriving on every
136
+ // refresh. A refresh without the object, or without a diagnosed cause, keeps
137
+ // what the last one said.
138
+ function missFrom(cache, prior) {
139
+ if (!cache || typeof cache !== 'object') return prior || null;
140
+ const at = typeof cache.last_miss_at === 'number' && Number.isFinite(cache.last_miss_at) ? cache.last_miss_at * 1000 : null;
141
+ const cause = cache.last_miss_cause;
142
+ const causes = cause && typeof cause === 'object' && Array.isArray(cause.causes)
143
+ ? cause.causes.filter((name) => typeof name === 'string')
144
+ : [];
145
+ if (at === null || !causes.length) return prior || null;
146
+ return { at, causes };
147
+ }
148
+
130
149
  function trim(all, keep) {
131
150
  const ordered = Object.keys(all).sort((a, b) => (all[b].at || 0) - (all[a].at || 0));
132
151
  const kept = {};
@@ -378,7 +397,13 @@ function runPrevious(command, raw, env, budgetMs) {
378
397
  windowsHide: true,
379
398
  stdio: ['pipe', 'pipe', 'ignore'],
380
399
  });
381
- if (result.error || result.status !== 0) return '';
400
+ // A command that never reads its stdin - echo, most one-line status lines -
401
+ // can exit before the JSON is written, and the write then fails with EPIPE.
402
+ // That says nothing about the command, which ran and printed. Measured on a
403
+ // Linux runner: the whole test took 85 ms, so no timeout was involved, and
404
+ // the line was dropped anyway. Only the command's own failure drops it.
405
+ if (result.status !== 0) return '';
406
+ if (result.error && result.error.code !== 'EPIPE') return '';
382
407
  return String(result.stdout || '').replace(/\s+$/, '');
383
408
  } catch (err) {
384
409
  return '';
@@ -490,14 +515,16 @@ async function main(argv) {
490
515
  headersAt: slot ? slot.headersAt : null,
491
516
  model: slot ? slot.model : null,
492
517
  modelName: slot ? slot.modelName : null,
493
- // Claude Code hands this line the effort outright, so the slot is
494
- // already current and nothing else need be read. It is only when the
495
- // slot has none - the very first update of a session, or a build that
496
- // does not send it - that the transcript is worth a look.
518
+ // Claude Code hands this line the effort outright, and that value is
519
+ // read first: the slot beside it was loaded from disk before this
520
+ // update was stored, so on a session's very first line it is empty and
521
+ // trusting it printed no effort at all until the second update. The
522
+ // slot still answers for a build that sends none, and only when it has
523
+ // none either is the transcript worth a look.
497
524
  effort:
498
- slot && slot.effort
499
- ? slot.effort
500
- : view.pickEffort(
525
+ (input && input.effort && typeof input.effort.level === 'string' && input.effort.level)
526
+ || (slot && slot.effort)
527
+ || view.pickEffort(
501
528
  null,
502
529
  usage.liveEffort(mine || (slot && slot.sessionId) || null),
503
530
  collected.settings ? collected.settings.effortLevel : null
@@ -559,6 +586,7 @@ module.exports = {
559
586
  CHAIN_TIMEOUT_MS,
560
587
  feedFile,
561
588
  readFeed,
589
+ missFrom,
562
590
  writeFeed,
563
591
  slotFrom,
564
592
  record,
@@ -48,7 +48,8 @@ function pluginDir() {
48
48
  // runs hook commands through `cmd /c` on Windows and `sh -c` elsewhere, so the
49
49
  // quoting has to survive both.
50
50
  function quote(file) {
51
- return '"' + String(file).replace(/\\/g, '/') + '"';
51
+ const normalized = String(file).replace(/\\/g, '/');
52
+ return normalized.includes(' ') ? '"' + normalized + '"' : normalized;
52
53
  }
53
54
 
54
55
  function scriptPath(name) {
@@ -393,6 +393,9 @@ function empty() {
393
393
  // means refusing fan-outs on a fresh window for a reason nobody gave. A
394
394
  // stored cap with no session here predates this rule and is ignored.
395
395
  ceilingSession: null,
396
+ // Applies to EVERY session, survives a reset, and is only cleared by
397
+ // asking. The session-scoped cap above lapses on purpose; this does not.
398
+ ceilingAlways: null,
396
399
  setAt: null,
397
400
  setBy: null,
398
401
  session: null,
@@ -437,6 +440,10 @@ function read() {
437
440
  // Anything outside 1-100 is not a ceiling, and enforcing a number that was
438
441
  // never a percentage would refuse work over a typo.
439
442
  base.ceilingSession = typeof parsed.ceilingSession === 'string' ? parsed.ceilingSession : null;
443
+ base.ceilingAlways =
444
+ Number.isFinite(parsed.ceilingAlways) && parsed.ceilingAlways > 0 && parsed.ceilingAlways <= 100
445
+ ? parsed.ceilingAlways
446
+ : null;
440
447
  base.ceilingPercent =
441
448
  Number.isFinite(parsed.ceilingPercent) && parsed.ceilingPercent > 0 && parsed.ceilingPercent <= 100
442
449
  ? parsed.ceilingPercent
@@ -1066,12 +1073,39 @@ function tierNow(options) {
1066
1073
  },
1067
1074
  running: {
1068
1075
  model: running,
1069
- effort: effort ? effort.effort : null,
1076
+ effort: ultracodeName(effort ? effort.effort : null, sessionId, opts.now),
1070
1077
  source: effort ? effort.source : null,
1071
1078
  },
1072
1079
  };
1073
1080
  }
1074
1081
 
1082
+ // Ultracode reads as "xhigh" everywhere, and that is technically true and
1083
+ // practically wrong.
1084
+ //
1085
+ // Ultracode RESOLVES to xhigh, so every effort reader - the settings file, the
1086
+ // live transcript, CLAUDE_EFFORT - honestly reports xhigh, and the display said
1087
+ // xhigh at somebody who had typed ultracode. The two are the same amount of
1088
+ // reasoning and a very different amount of everything else, so a person who
1089
+ // asked for one and is shown the other has no way to tell it took.
1090
+ //
1091
+ // The plugin already knows. brief.js sets an ultracode flag on the session's
1092
+ // activity mark the moment the word appears in a prompt, and it STICKS for the
1093
+ // session because ultracode is a session-level trigger rather than a per-prompt
1094
+ // one. Nothing read it back. This does.
1095
+ //
1096
+ // Only xhigh is renamed. If some future effort is higher, reporting it as
1097
+ // ultracode would be a downgrade dressed up as a label, so it is left alone.
1098
+ function ultracodeName(effort, sessionId, now) {
1099
+ if (effort !== 'xhigh' || !sessionId) return effort;
1100
+ try {
1101
+ const entry = require('./activity.js').read()[sessionId];
1102
+ return entry && entry.ultracode ? 'ultracode' : effort;
1103
+ } catch (err) {
1104
+ // The real effort is a better answer than no answer.
1105
+ return effort;
1106
+ }
1107
+ }
1108
+
1075
1109
  function sameFamily(a, b) {
1076
1110
  const left = modelRank(a);
1077
1111
  const right = modelRank(b);
@@ -1105,10 +1139,58 @@ function tierLine(tier, options) {
1105
1139
  if (opts.terse) {
1106
1140
  return differs ? runningText + source + ', yours ' + baseText : runningText + source;
1107
1141
  }
1142
+ // Reporting the tier is not the same as deciding about it.
1143
+ //
1144
+ // This line has always said what is running and left it there, and the
1145
+ // observed result is that the tier is simply never revisited: a session opens
1146
+ // on the dearest model at the dearest effort and stays there through work
1147
+ // that did not need either. Saying the number is not a prompt to act on it.
1148
+ //
1149
+ // So when the tier IS an expensive one, the line asks for a decision rather
1150
+ // than leaving a fact lying around. Only then: on a cheap tier there is
1151
+ // nothing to decide, and a sentence asking every turn would be exactly the
1152
+ // per-turn cost this plugin exists to avoid.
1153
+ //
1154
+ // Deliberately names no slash command. /model and /effort do not exist in
1155
+ // Codex, and telling Codex to run them is telling it to do nothing while
1156
+ // believing it acted - the mistake levers() already exists to prevent.
1157
+ const dear = topTier(run.model || base.model, run.effort || base.effort);
1158
+ // What the plugin can actually do about the tier, stated exactly.
1159
+ //
1160
+ // Nothing in a hook, a tool or the SDK can change a running session's model
1161
+ // or effort - that was researched to the primary sources on 2026-09-14 and
1162
+ // the answer is a flat no. /model and /effort are the user's, by design. So
1163
+ // the most honest thing this line can do is name the decision and, where it
1164
+ // is cheap, say so: on Fable 5.1 on a subscription an /effort change keeps
1165
+ // the prompt cache (prompt-caching doc, v2.1.260+), which makes stepping
1166
+ // effort down mid-session free. On every other model it rebuilds the cache,
1167
+ // and on a large context that can cost more than a few cheaper turns save -
1168
+ // so there the advice is to choose at the START of a session.
1169
+ const onFable = sameFamily(run.model || base.model, 'fable');
1170
+ const decide = dear
1171
+ ? ' Decide in one line whether the work in front of you needs ' + runningText + '.' +
1172
+ (onFable
1173
+ ? ' On Fable 5.1 an /effort change keeps the cache, so if the next stretch is mechanical, /effort low now costs nothing.'
1174
+ : ' Changing effort or model mid-session rebuilds the cache, so if it does not, say so and choose lower at the next session start rather than switching now.')
1175
+ : '';
1108
1176
  return differs
1109
1177
  ? 'Running ' + runningText + source + '; your baseline is ' + baseText + '. The gap is the ' +
1110
- 'interesting part: your baseline is yours and is not being changed.'
1111
- : 'Running ' + runningText + source + '.';
1178
+ 'interesting part: your baseline is yours and is not being changed.' + decide
1179
+ : 'Running ' + runningText + source + '.' + decide;
1180
+ }
1181
+
1182
+ // Whether this tier is dear enough to be worth a decision. Opus or above, or an
1183
+ // effort at or above xhigh; either alone is enough to be worth asking.
1184
+ //
1185
+ // MODEL_ORDER runs CHEAPEST first - haiku, sonnet, opus, mythos, fable - so the
1186
+ // dear end is a HIGH rank, not a low one. Getting that backwards asked haiku to
1187
+ // justify itself and let opus through silently, which is the exact inverse of
1188
+ // the point.
1189
+ function topTier(model, effort) {
1190
+ const rank = modelRank(model);
1191
+ const dearModel = rank !== null && rank >= MODEL_ORDER.indexOf('opus');
1192
+ const dearEffort = ['xhigh', 'max', 'ultra', 'ultracode'].includes(String(effort || '').toLowerCase());
1193
+ return dearModel || dearEffort;
1112
1194
  }
1113
1195
 
1114
1196
  // ---------------------------------------------------------------------------
@@ -1476,6 +1558,9 @@ function main(argv) {
1476
1558
  if (ceilingArg.error) return ceilingArg.error;
1477
1559
  const state = read();
1478
1560
  const previous = state.ceilingPercent;
1561
+ if (flag('--always')) {
1562
+ state.ceilingAlways = ceilingArg.clear ? null : ceilingArg.percent;
1563
+ }
1479
1564
  state.ceilingPercent = ceilingArg.clear ? null : ceilingArg.percent;
1480
1565
  state.ceilingSession = ceilingArg.clear ? null : sessionId;
1481
1566
  write(state);
@@ -1495,6 +1580,7 @@ function clearCeiling(now) {
1495
1580
  const previous = state.ceilingPercent;
1496
1581
  state.ceilingPercent = null;
1497
1582
  state.ceilingSession = null;
1583
+ state.ceilingAlways = null;
1498
1584
  write(state);
1499
1585
  logChange({ plane: 'mode', key: 'ceiling', from: previous, to: null, by: 'user', reason: null }, now);
1500
1586
  }
@@ -1503,6 +1589,13 @@ function clearCeiling(now) {
1503
1589
  // refusal, because a percentage on its own does not tell anyone what changes.
1504
1590
  function ceilingLine() {
1505
1591
  const state = read();
1592
+ if (Number.isFinite(state.ceilingAlways)) {
1593
+ return (
1594
+ 'Standing cap ' + state.ceilingAlways + '%, for this and every future session on this ' +
1595
+ 'machine, including after a limit resets. Past it, fan-out calls are refused at the hook; ' +
1596
+ 'nothing else is blocked. "mode --cap off --always" removes it.'
1597
+ );
1598
+ }
1506
1599
  if (state.ceilingPercent === null || state.ceilingPercent === undefined) {
1507
1600
  return 'Ceiling off. Nothing is refused; the plugin reports and does not intervene.';
1508
1601
  }
@@ -1641,6 +1734,8 @@ module.exports = {
1641
1734
  adviceDecline,
1642
1735
  adviceMute,
1643
1736
  tierNow,
1737
+ ultracodeName,
1738
+ topTier,
1644
1739
  tierLine,
1645
1740
  ledger,
1646
1741
  explain,
@@ -137,6 +137,11 @@ function pulseText(parts) {
137
137
 
138
138
  const head = '[usage-limits] ' + (parts.fanout ? 'Before this fan-out: ' : '') + bits.join(', ') + '.';
139
139
  if (parts.fanout) {
140
+ // Parallel Agent calls fire this once per call, so the advice sentence was
141
+ // being said six times in a row for one reading. The reading itself is
142
+ // never throttled; the advice is said once per session per ten minutes,
143
+ // through the same state file as every other throttle here.
144
+ if (parts.advice === false) return head;
140
145
  // Said before every Workflow or Agent call. The agents spend this same
141
146
  // window, nothing can speak again until they stop, and a main-loop turn
142
147
  // with a large context costs more than one whole fresh-context agent.
@@ -147,6 +152,10 @@ function pulseText(parts) {
147
152
  '. Fewer agents with a fresh context beat another turn of a long one.'
148
153
  );
149
154
  }
155
+ // The tight and gone sentences are advice too, and they were said on every
156
+ // due pulse - three times in ten minutes on 2026-09-20. Same slot and same
157
+ // ten minutes as the fan-out advice; the reading itself is never withheld.
158
+ if ((parts.pressure === 'gone' || parts.pressure === 'tight') && parts.advice === false) return head;
150
159
  if (parts.pressure === 'gone') {
151
160
  return head + ' The budget is gone. Stop adding work, save what exists and write the handoff.';
152
161
  }
@@ -249,8 +258,10 @@ async function run(now, hookInput) {
249
258
  // ask the model to agree with it.
250
259
  if (event === 'PreToolUse' && ceiling.isMultiplier(tool)) {
251
260
  try {
261
+ const worst = ceilingPercent(now, null, hookInput && hookInput.transcript_path);
252
262
  const at = ceiling.assess({
253
- percent: ceilingPercent(now),
263
+ percent: worst ? worst.percent : null,
264
+ label: worst ? worst.label : null,
254
265
  state: budget.state,
255
266
  env: process.env,
256
267
  // The cap only binds in the session that set it.
@@ -289,10 +300,12 @@ async function run(now, hookInput) {
289
300
  // The cheap path, and the one taken almost every time. A fan-out is never
290
301
  // throttled: it is said every time, because every time it is about to cost.
291
302
  if (!fanout && !due(all, throttleKey, now, every)) return '';
303
+ const adviceKey = (sessionId || '_') + '#fanout-advice';
304
+ const sayAdvice = fanout && due(all, adviceKey, now, 10 * 60 * 1000);
292
305
 
293
306
  // Claimed before the scan rather than after, so a slow scan cannot let a
294
307
  // second tool call start another one.
295
- writeState(trim(all, throttleKey, now));
308
+ writeState(sayAdvice ? trim(trim(all, throttleKey, now), adviceKey, now) : trim(all, throttleKey, now));
296
309
 
297
310
  // A reading as old as the interval is replaced with the one Claude Code
298
311
  // would take for /usage, so a turn that runs for an hour is measured
@@ -398,6 +411,11 @@ async function run(now, hookInput) {
398
411
  return recheck;
399
412
  }
400
413
 
414
+ // The pressure advice shares the fan-out advice's slot and its ten minutes.
415
+ // Unlike the fan-out's, it is stamped after the fact and only when a line
416
+ // actually carried it: a pulse with no reading to hang it on says nothing,
417
+ // and must not spend the slot on nothing.
418
+ const sayPressureAdvice = !fanout && (pressure === 'tight' || pressure === 'gone') && due(readState(), adviceKey, now, 10 * 60 * 1000);
401
419
  const spoken = pulseText({
402
420
  label: binding.label,
403
421
  percentUsed: binding.percentUsed,
@@ -410,50 +428,82 @@ async function run(now, hookInput) {
410
428
  sessions: active,
411
429
  pressure,
412
430
  fanout,
431
+ advice: fanout ? sayAdvice : sayPressureAdvice,
413
432
  });
433
+ if (spoken && sayPressureAdvice) writeState(trim(readState(), adviceKey, now));
414
434
  return recheck ? (spoken ? spoken + ' ' + recheck : recheck) : spoken;
415
435
  }
416
436
 
417
- // The cheapest percentage good enough to enforce a ceiling against.
437
+ // The cheapest reading good enough to enforce a ceiling against.
418
438
  //
419
439
  // The ceiling is checked before every fan-out, which is far too often to scan
420
- // transcripts for. Two sources are already paid for: the corrected reading the
421
- // last scan left behind, and the account snapshot, which is one small file
422
- // read. The highest of them wins, because a ceiling means "no window past
423
- // here" - taking the emptiest window would be a ceiling that never binds.
440
+ // transcripts for. snapshotWindows() is the no-scan view every other cheap
441
+ // reader uses: the account snapshot, the corrections the last scan left
442
+ // behind, and whether each window is one this agent can spend into. The
443
+ // fullest of those wins, because a ceiling means "no window past here".
424
444
  //
425
- // Returns null when neither source has anything, and a ceiling with no reading
426
- // behind it never refuses. Guessing high would block work over a number nobody
445
+ // This used to take the highest number on disk, whatever window it belonged
446
+ // to - including a weekly scoped to a model this session was not running, and
447
+ // the snapshot of a window that had already reset. See ceiling.worstWindow().
448
+ //
449
+ // Returns null when nothing is readable, and a ceiling with no reading behind
450
+ // it never refuses. Guessing high would block work over a number nobody
427
451
  // measured; guessing low would not be a ceiling at all.
428
- function ceilingPercent(now) {
429
- let worst = null;
430
- const consider = (value) => {
431
- if (!Number.isFinite(value)) return;
432
- if (worst === null || value > worst) worst = value;
433
- };
452
+ function ceilingPercent(now, collected, transcriptPath) {
434
453
  const codexHome = usage.isCodex() ? require('./codex.js').homeDir() : null;
435
454
  try {
436
- const entries = reading.read(codexHome);
437
- for (const key of Object.keys(entries)) {
438
- const entry = reading.correctedFor(key, now, null, codexHome);
439
- if (entry) consider(entry.percentUsed);
440
- }
455
+ const running = lastModel(transcriptPath);
456
+ return ceiling.worstWindow(
457
+ usage.snapshotWindows(collected || usage.collect(now), now, codexHome, running ? [running] : [])
458
+ );
441
459
  } catch (err) {
442
- // A missing or unreadable correction just means the snapshot decides.
460
+ // No snapshot is a reason not to enforce, not a reason to throw.
461
+ return null;
443
462
  }
463
+ }
464
+
465
+ // The model this session's last reply came from, off the tail of its transcript.
466
+ //
467
+ // The setting says what a session started on; /model changes what it runs
468
+ // without touching settings.json. A weekly scoped to the model actually
469
+ // running must count, so the pulse adds this to the setting rather than
470
+ // replacing it - suppressing a window takes both sides agreeing. One read of
471
+ // the last 64 KB, from the end, whole lines only. Null on anything unexpected.
472
+ function lastModel(transcriptPath) {
473
+ if (!transcriptPath) return null;
474
+ let fd = null;
444
475
  try {
445
- const snapshot = usage.collect(now);
446
- const utilization = snapshot && snapshot.utilization;
447
- if (utilization && typeof utilization === 'object') {
448
- for (const key of Object.keys(utilization)) {
449
- const window = utilization[key];
450
- if (window && typeof window === 'object') consider(Number(window.utilization));
476
+ fd = fs.openSync(transcriptPath, 'r');
477
+ const size = fs.fstatSync(fd).size;
478
+ const length = Math.min(size, 64 * 1024);
479
+ const buffer = Buffer.alloc(length);
480
+ fs.readSync(fd, buffer, 0, length, size - length);
481
+ const lines = buffer.toString('utf8').split('\n');
482
+ // The first line of a partial read is cut; only a read from the start is whole.
483
+ if (length < size) lines.shift();
484
+ for (let i = lines.length - 1; i >= 0; i -= 1) {
485
+ if (!lines[i].includes('"assistant"')) continue;
486
+ try {
487
+ const entry = JSON.parse(lines[i]);
488
+ const model = entry && entry.type === 'assistant' && entry.message && entry.message.model;
489
+ // "<synthetic>" marks a reply Claude Code wrote itself, not a model.
490
+ if (model && typeof model === 'string' && !model.startsWith('<')) return model;
491
+ } catch (err) {
492
+ // A half-written last line while the transcript is being appended.
451
493
  }
452
494
  }
495
+ return null;
453
496
  } catch (err) {
454
- // Same: no snapshot is a reason not to enforce, not a reason to throw.
497
+ return null;
498
+ } finally {
499
+ if (fd !== null) {
500
+ try {
501
+ fs.closeSync(fd);
502
+ } catch (err) {
503
+ // Already closed is fine.
504
+ }
505
+ }
455
506
  }
456
- return worst;
457
507
  }
458
508
 
459
509
  // PostToolUse does not take plain stdout as context the way UserPromptSubmit
@@ -550,5 +600,6 @@ module.exports = {
550
600
  envelope,
551
601
  refusal,
552
602
  ceilingPercent,
603
+ lastModel,
553
604
  run,
554
605
  };