claude-usage-limits 1.25.0 → 1.39.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -175,7 +175,7 @@ function bindingReset(now) {
175
175
  function sessionId(argv, env) {
176
176
  const at = argv.indexOf('--session-id');
177
177
  if (at !== -1 && argv[at + 1]) return argv[at + 1];
178
- return env.CLAUDE_SESSION_ID || env.CODEX_SESSION_ID || 'defer-' + Date.now().toString(36);
178
+ return env.CLAUDE_CODE_SESSION_ID || env.CLAUDE_SESSION_ID || env.CODEX_SESSION_ID || 'defer-' + Date.now().toString(36);
179
179
  }
180
180
 
181
181
  function argOf(argv, name) {
@@ -210,7 +210,7 @@ function plan(options) {
210
210
 
211
211
  function status(now) {
212
212
  const state = relay.read();
213
- const armed = state.armed;
213
+ const armed = relay.armedFor(state, process.env.CLAUDE_CODE_SESSION_ID) || state.armed;
214
214
  if (!armed) return 'Nothing is deferred.';
215
215
  const when = Number(armed.wakeAt);
216
216
  const deferred = armed.deferred === true;
@@ -230,9 +230,10 @@ function status(now) {
230
230
 
231
231
  function cancel() {
232
232
  const state = relay.read();
233
- if (!state.armed) return 'Nothing was deferred.';
234
- const label = formatClock(Number(state.armed.wakeAt));
235
- const result = relay.disarm('cancelled by hand', Date.now());
233
+ const own = relay.armedFor(state, process.env.CLAUDE_CODE_SESSION_ID) || state.armed;
234
+ if (!own) return 'Nothing was deferred.';
235
+ const label = formatClock(Number(own.wakeAt));
236
+ const result = relay.disarm('cancelled by hand', Date.now(), own.id);
236
237
  return result && result.ok === false
237
238
  ? 'Could not cancel: ' + result.error
238
239
  : 'Cancelled the run booked for ' + label + '.';
@@ -286,9 +287,10 @@ function main(argv, now) {
286
287
  // session can tell the two apart - they read the same record.
287
288
  try {
288
289
  const held = relay.read();
289
- if (held.armed) {
290
- held.armed.deferred = true;
291
- held.armed.continuation = Boolean(work);
290
+ const mine = (typeof sessionId !== 'undefined' && relay.armedFor(held, sessionId)) || held.armed;
291
+ if (mine) {
292
+ mine.deferred = true;
293
+ mine.continuation = Boolean(work);
292
294
  relay.write(held);
293
295
  }
294
296
  } catch (err) {
@@ -124,9 +124,28 @@ function slotFrom(input, previous, now) {
124
124
  cwd: typeof input.cwd === 'string' ? input.cwd : prior.cwd || null,
125
125
  version: typeof input.version === 'string' ? input.version : prior.version || null,
126
126
  fastMode: input.fast_mode === true,
127
+ cacheMiss: missFrom(input.prompt_cache, prior.cacheMiss),
127
128
  };
128
129
  }
129
130
 
131
+ // The last prompt-cache miss and its diagnosed causes, from the status line's
132
+ // prompt_cache object: last_miss_at is epoch seconds and last_miss_cause is
133
+ // null until the first miss and whenever no cause was found (Claude Code
134
+ // 2.1.260 and later; both per code.claude.com/docs/en/statusline). The moment
135
+ // is what lets the brief tell a new miss from the same one arriving on every
136
+ // refresh. A refresh without the object, or without a diagnosed cause, keeps
137
+ // what the last one said.
138
+ function missFrom(cache, prior) {
139
+ if (!cache || typeof cache !== 'object') return prior || null;
140
+ const at = typeof cache.last_miss_at === 'number' && Number.isFinite(cache.last_miss_at) ? cache.last_miss_at * 1000 : null;
141
+ const cause = cache.last_miss_cause;
142
+ const causes = cause && typeof cause === 'object' && Array.isArray(cause.causes)
143
+ ? cause.causes.filter((name) => typeof name === 'string')
144
+ : [];
145
+ if (at === null || !causes.length) return prior || null;
146
+ return { at, causes };
147
+ }
148
+
130
149
  function trim(all, keep) {
131
150
  const ordered = Object.keys(all).sort((a, b) => (all[b].at || 0) - (all[a].at || 0));
132
151
  const kept = {};
@@ -378,7 +397,13 @@ function runPrevious(command, raw, env, budgetMs) {
378
397
  windowsHide: true,
379
398
  stdio: ['pipe', 'pipe', 'ignore'],
380
399
  });
381
- if (result.error || result.status !== 0) return '';
400
+ // A command that never reads its stdin - echo, most one-line status lines -
401
+ // can exit before the JSON is written, and the write then fails with EPIPE.
402
+ // That says nothing about the command, which ran and printed. Measured on a
403
+ // Linux runner: the whole test took 85 ms, so no timeout was involved, and
404
+ // the line was dropped anyway. Only the command's own failure drops it.
405
+ if (result.status !== 0) return '';
406
+ if (result.error && result.error.code !== 'EPIPE') return '';
382
407
  return String(result.stdout || '').replace(/\s+$/, '');
383
408
  } catch (err) {
384
409
  return '';
@@ -490,14 +515,16 @@ async function main(argv) {
490
515
  headersAt: slot ? slot.headersAt : null,
491
516
  model: slot ? slot.model : null,
492
517
  modelName: slot ? slot.modelName : null,
493
- // Claude Code hands this line the effort outright, so the slot is
494
- // already current and nothing else need be read. It is only when the
495
- // slot has none - the very first update of a session, or a build that
496
- // does not send it - that the transcript is worth a look.
518
+ // Claude Code hands this line the effort outright, and that value is
519
+ // read first: the slot beside it was loaded from disk before this
520
+ // update was stored, so on a session's very first line it is empty and
521
+ // trusting it printed no effort at all until the second update. The
522
+ // slot still answers for a build that sends none, and only when it has
523
+ // none either is the transcript worth a look.
497
524
  effort:
498
- slot && slot.effort
499
- ? slot.effort
500
- : view.pickEffort(
525
+ (input && input.effort && typeof input.effort.level === 'string' && input.effort.level)
526
+ || (slot && slot.effort)
527
+ || view.pickEffort(
501
528
  null,
502
529
  usage.liveEffort(mine || (slot && slot.sessionId) || null),
503
530
  collected.settings ? collected.settings.effortLevel : null
@@ -559,6 +586,7 @@ module.exports = {
559
586
  CHAIN_TIMEOUT_MS,
560
587
  feedFile,
561
588
  readFeed,
589
+ missFrom,
562
590
  writeFeed,
563
591
  slotFrom,
564
592
  record,
@@ -48,7 +48,8 @@ function pluginDir() {
48
48
  // runs hook commands through `cmd /c` on Windows and `sh -c` elsewhere, so the
49
49
  // quoting has to survive both.
50
50
  function quote(file) {
51
- return '"' + String(file).replace(/\\/g, '/') + '"';
51
+ const normalized = String(file).replace(/\\/g, '/');
52
+ return normalized.includes(' ') ? '"' + normalized + '"' : normalized;
52
53
  }
53
54
 
54
55
  function scriptPath(name) {
@@ -192,7 +192,7 @@ function describe(settings, state) {
192
192
  function logSettingsChange(changes, direction, env) {
193
193
  if (!changes || !changes.length) return;
194
194
  const e = env || process.env;
195
- const by = e.CLAUDE_SESSION_ID || e.CODEX_SESSION_ID ? 'claude' : 'user';
195
+ const by = e.CLAUDE_CODE_SESSION_ID || e.CLAUDE_SESSION_ID || e.CODEX_SESSION_ID ? 'claude' : 'user';
196
196
  try {
197
197
  const mode = require('./mode.js');
198
198
  for (const change of changes) {
@@ -388,6 +388,14 @@ function empty() {
388
388
  // reported: past it, fan-out calls are refused at the hook. Null means no
389
389
  // ceiling, and a ceiling nobody set never refuses anything. See ceiling.js.
390
390
  ceilingPercent: null,
391
+ // WHOSE cap it is. A cap is said about the work in front of someone, so
392
+ // it binds only in the session that set it; carrying it into the next one
393
+ // means refusing fan-outs on a fresh window for a reason nobody gave. A
394
+ // stored cap with no session here predates this rule and is ignored.
395
+ ceilingSession: null,
396
+ // Applies to EVERY session, survives a reset, and is only cleared by
397
+ // asking. The session-scoped cap above lapses on purpose; this does not.
398
+ ceilingAlways: null,
391
399
  setAt: null,
392
400
  setBy: null,
393
401
  session: null,
@@ -431,6 +439,11 @@ function read() {
431
439
  base.guardPercent = Number.isFinite(parsed.guardPercent) ? parsed.guardPercent : null;
432
440
  // Anything outside 1-100 is not a ceiling, and enforcing a number that was
433
441
  // never a percentage would refuse work over a typo.
442
+ base.ceilingSession = typeof parsed.ceilingSession === 'string' ? parsed.ceilingSession : null;
443
+ base.ceilingAlways =
444
+ Number.isFinite(parsed.ceilingAlways) && parsed.ceilingAlways > 0 && parsed.ceilingAlways <= 100
445
+ ? parsed.ceilingAlways
446
+ : null;
434
447
  base.ceilingPercent =
435
448
  Number.isFinite(parsed.ceilingPercent) && parsed.ceilingPercent > 0 && parsed.ceilingPercent <= 100
436
449
  ? parsed.ceilingPercent
@@ -1060,12 +1073,39 @@ function tierNow(options) {
1060
1073
  },
1061
1074
  running: {
1062
1075
  model: running,
1063
- effort: effort ? effort.effort : null,
1076
+ effort: ultracodeName(effort ? effort.effort : null, sessionId, opts.now),
1064
1077
  source: effort ? effort.source : null,
1065
1078
  },
1066
1079
  };
1067
1080
  }
1068
1081
 
1082
+ // Ultracode reads as "xhigh" everywhere, and that is technically true and
1083
+ // practically wrong.
1084
+ //
1085
+ // Ultracode RESOLVES to xhigh, so every effort reader - the settings file, the
1086
+ // live transcript, CLAUDE_EFFORT - honestly reports xhigh, and the display said
1087
+ // xhigh at somebody who had typed ultracode. The two are the same amount of
1088
+ // reasoning and a very different amount of everything else, so a person who
1089
+ // asked for one and is shown the other has no way to tell it took.
1090
+ //
1091
+ // The plugin already knows. brief.js sets an ultracode flag on the session's
1092
+ // activity mark the moment the word appears in a prompt, and it STICKS for the
1093
+ // session because ultracode is a session-level trigger rather than a per-prompt
1094
+ // one. Nothing read it back. This does.
1095
+ //
1096
+ // Only xhigh is renamed. If some future effort is higher, reporting it as
1097
+ // ultracode would be a downgrade dressed up as a label, so it is left alone.
1098
+ function ultracodeName(effort, sessionId, now) {
1099
+ if (effort !== 'xhigh' || !sessionId) return effort;
1100
+ try {
1101
+ const entry = require('./activity.js').read()[sessionId];
1102
+ return entry && entry.ultracode ? 'ultracode' : effort;
1103
+ } catch (err) {
1104
+ // The real effort is a better answer than no answer.
1105
+ return effort;
1106
+ }
1107
+ }
1108
+
1069
1109
  function sameFamily(a, b) {
1070
1110
  const left = modelRank(a);
1071
1111
  const right = modelRank(b);
@@ -1099,10 +1139,58 @@ function tierLine(tier, options) {
1099
1139
  if (opts.terse) {
1100
1140
  return differs ? runningText + source + ', yours ' + baseText : runningText + source;
1101
1141
  }
1142
+ // Reporting the tier is not the same as deciding about it.
1143
+ //
1144
+ // This line has always said what is running and left it there, and the
1145
+ // observed result is that the tier is simply never revisited: a session opens
1146
+ // on the dearest model at the dearest effort and stays there through work
1147
+ // that did not need either. Saying the number is not a prompt to act on it.
1148
+ //
1149
+ // So when the tier IS an expensive one, the line asks for a decision rather
1150
+ // than leaving a fact lying around. Only then: on a cheap tier there is
1151
+ // nothing to decide, and a sentence asking every turn would be exactly the
1152
+ // per-turn cost this plugin exists to avoid.
1153
+ //
1154
+ // Deliberately names no slash command. /model and /effort do not exist in
1155
+ // Codex, and telling Codex to run them is telling it to do nothing while
1156
+ // believing it acted - the mistake levers() already exists to prevent.
1157
+ const dear = topTier(run.model || base.model, run.effort || base.effort);
1158
+ // What the plugin can actually do about the tier, stated exactly.
1159
+ //
1160
+ // Nothing in a hook, a tool or the SDK can change a running session's model
1161
+ // or effort - that was researched to the primary sources on 2026-09-14 and
1162
+ // the answer is a flat no. /model and /effort are the user's, by design. So
1163
+ // the most honest thing this line can do is name the decision and, where it
1164
+ // is cheap, say so: on Fable 5.1 on a subscription an /effort change keeps
1165
+ // the prompt cache (prompt-caching doc, v2.1.260+), which makes stepping
1166
+ // effort down mid-session free. On every other model it rebuilds the cache,
1167
+ // and on a large context that can cost more than a few cheaper turns save -
1168
+ // so there the advice is to choose at the START of a session.
1169
+ const onFable = sameFamily(run.model || base.model, 'fable');
1170
+ const decide = dear
1171
+ ? ' Decide in one line whether the work in front of you needs ' + runningText + '.' +
1172
+ (onFable
1173
+ ? ' On Fable 5.1 an /effort change keeps the cache, so if the next stretch is mechanical, /effort low now costs nothing.'
1174
+ : ' Changing effort or model mid-session rebuilds the cache, so if it does not, say so and choose lower at the next session start rather than switching now.')
1175
+ : '';
1102
1176
  return differs
1103
1177
  ? 'Running ' + runningText + source + '; your baseline is ' + baseText + '. The gap is the ' +
1104
- 'interesting part: your baseline is yours and is not being changed.'
1105
- : 'Running ' + runningText + source + '.';
1178
+ 'interesting part: your baseline is yours and is not being changed.' + decide
1179
+ : 'Running ' + runningText + source + '.' + decide;
1180
+ }
1181
+
1182
+ // Whether this tier is dear enough to be worth a decision. Opus or above, or an
1183
+ // effort at or above xhigh; either alone is enough to be worth asking.
1184
+ //
1185
+ // MODEL_ORDER runs CHEAPEST first - haiku, sonnet, opus, mythos, fable - so the
1186
+ // dear end is a HIGH rank, not a low one. Getting that backwards asked haiku to
1187
+ // justify itself and let opus through silently, which is the exact inverse of
1188
+ // the point.
1189
+ function topTier(model, effort) {
1190
+ const rank = modelRank(model);
1191
+ const dearModel = rank !== null && rank >= MODEL_ORDER.indexOf('opus');
1192
+ const dearEffort = ['xhigh', 'max', 'ultra', 'ultracode'].includes(String(effort || '').toLowerCase());
1193
+ return dearModel || dearEffort;
1106
1194
  }
1107
1195
 
1108
1196
  // ---------------------------------------------------------------------------
@@ -1240,7 +1328,10 @@ function setMode(name, opts, now) {
1240
1328
  state.setAt = at;
1241
1329
  state.setBy = 'user';
1242
1330
  if (opts && Number.isFinite(opts.guard)) state.guardPercent = opts.guard;
1243
- if (opts && Number.isFinite(opts.ceiling)) state.ceilingPercent = opts.ceiling;
1331
+ if (opts && Number.isFinite(opts.ceiling)) {
1332
+ state.ceilingPercent = opts.ceiling;
1333
+ state.ceilingSession = opts.sessionId || null;
1334
+ }
1244
1335
  write(state);
1245
1336
  logChange({ plane: 'mode', key: 'mode', from: before, to: name, by: 'user', reason: null }, at);
1246
1337
 
@@ -1288,7 +1379,10 @@ function main(argv) {
1288
1379
  const next = args[at + 1];
1289
1380
  return next && next.indexOf('--') !== 0 ? next : null;
1290
1381
  };
1291
- const sessionId = value('--session-id') || process.env.CLAUDE_SESSION_ID || null;
1382
+ // CLAUDE_CODE_SESSION_ID is the one Claude Code actually exports;
1383
+ // CLAUDE_SESSION_ID never existed, so this read null and a cap set from the
1384
+ // command line could record no owner and therefore never bind.
1385
+ const sessionId = value('--session-id') || process.env.CLAUDE_CODE_SESSION_ID || process.env.CLAUDE_SESSION_ID || null;
1292
1386
  const decided = resolve({ sessionId });
1293
1387
 
1294
1388
  if (flag('--list')) return list(decided);
@@ -1464,7 +1558,11 @@ function main(argv) {
1464
1558
  if (ceilingArg.error) return ceilingArg.error;
1465
1559
  const state = read();
1466
1560
  const previous = state.ceilingPercent;
1561
+ if (flag('--always')) {
1562
+ state.ceilingAlways = ceilingArg.clear ? null : ceilingArg.percent;
1563
+ }
1467
1564
  state.ceilingPercent = ceilingArg.clear ? null : ceilingArg.percent;
1565
+ state.ceilingSession = ceilingArg.clear ? null : sessionId;
1468
1566
  write(state);
1469
1567
  logChange(
1470
1568
  { plane: 'mode', key: 'ceiling', from: previous, to: state.ceilingPercent, by: 'user', reason: null },
@@ -1481,6 +1579,8 @@ function clearCeiling(now) {
1481
1579
  const state = read();
1482
1580
  const previous = state.ceilingPercent;
1483
1581
  state.ceilingPercent = null;
1582
+ state.ceilingSession = null;
1583
+ state.ceilingAlways = null;
1484
1584
  write(state);
1485
1585
  logChange({ plane: 'mode', key: 'ceiling', from: previous, to: null, by: 'user', reason: null }, now);
1486
1586
  }
@@ -1489,14 +1589,22 @@ function clearCeiling(now) {
1489
1589
  // refusal, because a percentage on its own does not tell anyone what changes.
1490
1590
  function ceilingLine() {
1491
1591
  const state = read();
1592
+ if (Number.isFinite(state.ceilingAlways)) {
1593
+ return (
1594
+ 'Standing cap ' + state.ceilingAlways + '%, for this and every future session on this ' +
1595
+ 'machine, including after a limit resets. Past it, fan-out calls are refused at the hook; ' +
1596
+ 'nothing else is blocked. "mode --cap off --always" removes it.'
1597
+ );
1598
+ }
1492
1599
  if (state.ceilingPercent === null || state.ceilingPercent === undefined) {
1493
1600
  return 'Ceiling off. Nothing is refused; the plugin reports and does not intervene.';
1494
1601
  }
1495
1602
  return (
1496
- 'Ceiling ' + state.ceilingPercent + '%. Past that, fan-out calls (Agent, Task, Workflow and ' +
1497
- 'their equivalents) are refused at the hook, in every session on this machine. Nothing else ' +
1603
+ 'Cap ' + state.ceilingPercent + '%' + (state.ceilingSession ? ', for THIS session only' : ' (stored against no session, so it is ignored - set it again to apply it here)') +
1604
+ '. Past that, fan-out calls (Agent, Task, Workflow and ' +
1605
+ 'their equivalents) are refused at the hook. Nothing else ' +
1498
1606
  'is blocked, so the work still finishes - sequentially, in one session, which is where the ' +
1499
- 'saving comes from. "mode --cap off" removes it.'
1607
+ 'saving comes from. It lapses when this session ends; "mode --cap off" removes it now.'
1500
1608
  );
1501
1609
  }
1502
1610
 
@@ -1626,6 +1734,8 @@ module.exports = {
1626
1734
  adviceDecline,
1627
1735
  adviceMute,
1628
1736
  tierNow,
1737
+ ultracodeName,
1738
+ topTier,
1629
1739
  tierLine,
1630
1740
  ledger,
1631
1741
  explain,
@@ -137,6 +137,11 @@ function pulseText(parts) {
137
137
 
138
138
  const head = '[usage-limits] ' + (parts.fanout ? 'Before this fan-out: ' : '') + bits.join(', ') + '.';
139
139
  if (parts.fanout) {
140
+ // Parallel Agent calls fire this once per call, so the advice sentence was
141
+ // being said six times in a row for one reading. The reading itself is
142
+ // never throttled; the advice is said once per session per ten minutes,
143
+ // through the same state file as every other throttle here.
144
+ if (parts.advice === false) return head;
140
145
  // Said before every Workflow or Agent call. The agents spend this same
141
146
  // window, nothing can speak again until they stop, and a main-loop turn
142
147
  // with a large context costs more than one whole fresh-context agent.
@@ -147,6 +152,10 @@ function pulseText(parts) {
147
152
  '. Fewer agents with a fresh context beat another turn of a long one.'
148
153
  );
149
154
  }
155
+ // The tight and gone sentences are advice too, and they were said on every
156
+ // due pulse - three times in ten minutes on 2026-09-20. Same slot and same
157
+ // ten minutes as the fan-out advice; the reading itself is never withheld.
158
+ if ((parts.pressure === 'gone' || parts.pressure === 'tight') && parts.advice === false) return head;
150
159
  if (parts.pressure === 'gone') {
151
160
  return head + ' The budget is gone. Stop adding work, save what exists and write the handoff.';
152
161
  }
@@ -249,10 +258,14 @@ async function run(now, hookInput) {
249
258
  // ask the model to agree with it.
250
259
  if (event === 'PreToolUse' && ceiling.isMultiplier(tool)) {
251
260
  try {
261
+ const worst = ceilingPercent(now, null, hookInput && hookInput.transcript_path);
252
262
  const at = ceiling.assess({
253
- percent: ceilingPercent(now),
263
+ percent: worst ? worst.percent : null,
264
+ label: worst ? worst.label : null,
254
265
  state: budget.state,
255
266
  env: process.env,
267
+ // The cap only binds in the session that set it.
268
+ sessionId,
256
269
  });
257
270
  const call = ceiling.verdict(at, tool);
258
271
  if (call.decision === 'deny') return { deny: true, reason: call.reason };
@@ -287,10 +300,12 @@ async function run(now, hookInput) {
287
300
  // The cheap path, and the one taken almost every time. A fan-out is never
288
301
  // throttled: it is said every time, because every time it is about to cost.
289
302
  if (!fanout && !due(all, throttleKey, now, every)) return '';
303
+ const adviceKey = (sessionId || '_') + '#fanout-advice';
304
+ const sayAdvice = fanout && due(all, adviceKey, now, 10 * 60 * 1000);
290
305
 
291
306
  // Claimed before the scan rather than after, so a slow scan cannot let a
292
307
  // second tool call start another one.
293
- writeState(trim(all, throttleKey, now));
308
+ writeState(sayAdvice ? trim(trim(all, throttleKey, now), adviceKey, now) : trim(all, throttleKey, now));
294
309
 
295
310
  // A reading as old as the interval is replaced with the one Claude Code
296
311
  // would take for /usage, so a turn that runs for an hour is measured
@@ -396,6 +411,11 @@ async function run(now, hookInput) {
396
411
  return recheck;
397
412
  }
398
413
 
414
+ // The pressure advice shares the fan-out advice's slot and its ten minutes.
415
+ // Unlike the fan-out's, it is stamped after the fact and only when a line
416
+ // actually carried it: a pulse with no reading to hang it on says nothing,
417
+ // and must not spend the slot on nothing.
418
+ const sayPressureAdvice = !fanout && (pressure === 'tight' || pressure === 'gone') && due(readState(), adviceKey, now, 10 * 60 * 1000);
399
419
  const spoken = pulseText({
400
420
  label: binding.label,
401
421
  percentUsed: binding.percentUsed,
@@ -408,50 +428,82 @@ async function run(now, hookInput) {
408
428
  sessions: active,
409
429
  pressure,
410
430
  fanout,
431
+ advice: fanout ? sayAdvice : sayPressureAdvice,
411
432
  });
433
+ if (spoken && sayPressureAdvice) writeState(trim(readState(), adviceKey, now));
412
434
  return recheck ? (spoken ? spoken + ' ' + recheck : recheck) : spoken;
413
435
  }
414
436
 
415
- // The cheapest percentage good enough to enforce a ceiling against.
437
+ // The cheapest reading good enough to enforce a ceiling against.
416
438
  //
417
439
  // The ceiling is checked before every fan-out, which is far too often to scan
418
- // transcripts for. Two sources are already paid for: the corrected reading the
419
- // last scan left behind, and the account snapshot, which is one small file
420
- // read. The highest of them wins, because a ceiling means "no window past
421
- // here" - taking the emptiest window would be a ceiling that never binds.
440
+ // transcripts for. snapshotWindows() is the no-scan view every other cheap
441
+ // reader uses: the account snapshot, the corrections the last scan left
442
+ // behind, and whether each window is one this agent can spend into. The
443
+ // fullest of those wins, because a ceiling means "no window past here".
422
444
  //
423
- // Returns null when neither source has anything, and a ceiling with no reading
424
- // behind it never refuses. Guessing high would block work over a number nobody
445
+ // This used to take the highest number on disk, whatever window it belonged
446
+ // to - including a weekly scoped to a model this session was not running, and
447
+ // the snapshot of a window that had already reset. See ceiling.worstWindow().
448
+ //
449
+ // Returns null when nothing is readable, and a ceiling with no reading behind
450
+ // it never refuses. Guessing high would block work over a number nobody
425
451
  // measured; guessing low would not be a ceiling at all.
426
- function ceilingPercent(now) {
427
- let worst = null;
428
- const consider = (value) => {
429
- if (!Number.isFinite(value)) return;
430
- if (worst === null || value > worst) worst = value;
431
- };
452
+ function ceilingPercent(now, collected, transcriptPath) {
432
453
  const codexHome = usage.isCodex() ? require('./codex.js').homeDir() : null;
433
454
  try {
434
- const entries = reading.read(codexHome);
435
- for (const key of Object.keys(entries)) {
436
- const entry = reading.correctedFor(key, now, null, codexHome);
437
- if (entry) consider(entry.percentUsed);
438
- }
455
+ const running = lastModel(transcriptPath);
456
+ return ceiling.worstWindow(
457
+ usage.snapshotWindows(collected || usage.collect(now), now, codexHome, running ? [running] : [])
458
+ );
439
459
  } catch (err) {
440
- // A missing or unreadable correction just means the snapshot decides.
460
+ // No snapshot is a reason not to enforce, not a reason to throw.
461
+ return null;
441
462
  }
463
+ }
464
+
465
+ // The model this session's last reply came from, off the tail of its transcript.
466
+ //
467
+ // The setting says what a session started on; /model changes what it runs
468
+ // without touching settings.json. A weekly scoped to the model actually
469
+ // running must count, so the pulse adds this to the setting rather than
470
+ // replacing it - suppressing a window takes both sides agreeing. One read of
471
+ // the last 64 KB, from the end, whole lines only. Null on anything unexpected.
472
+ function lastModel(transcriptPath) {
473
+ if (!transcriptPath) return null;
474
+ let fd = null;
442
475
  try {
443
- const snapshot = usage.collect(now);
444
- const utilization = snapshot && snapshot.utilization;
445
- if (utilization && typeof utilization === 'object') {
446
- for (const key of Object.keys(utilization)) {
447
- const window = utilization[key];
448
- if (window && typeof window === 'object') consider(Number(window.utilization));
476
+ fd = fs.openSync(transcriptPath, 'r');
477
+ const size = fs.fstatSync(fd).size;
478
+ const length = Math.min(size, 64 * 1024);
479
+ const buffer = Buffer.alloc(length);
480
+ fs.readSync(fd, buffer, 0, length, size - length);
481
+ const lines = buffer.toString('utf8').split('\n');
482
+ // The first line of a partial read is cut; only a read from the start is whole.
483
+ if (length < size) lines.shift();
484
+ for (let i = lines.length - 1; i >= 0; i -= 1) {
485
+ if (!lines[i].includes('"assistant"')) continue;
486
+ try {
487
+ const entry = JSON.parse(lines[i]);
488
+ const model = entry && entry.type === 'assistant' && entry.message && entry.message.model;
489
+ // "<synthetic>" marks a reply Claude Code wrote itself, not a model.
490
+ if (model && typeof model === 'string' && !model.startsWith('<')) return model;
491
+ } catch (err) {
492
+ // A half-written last line while the transcript is being appended.
449
493
  }
450
494
  }
495
+ return null;
451
496
  } catch (err) {
452
- // Same: no snapshot is a reason not to enforce, not a reason to throw.
497
+ return null;
498
+ } finally {
499
+ if (fd !== null) {
500
+ try {
501
+ fs.closeSync(fd);
502
+ } catch (err) {
503
+ // Already closed is fine.
504
+ }
505
+ }
453
506
  }
454
- return worst;
455
507
  }
456
508
 
457
509
  // PostToolUse does not take plain stdout as context the way UserPromptSubmit
@@ -548,5 +600,6 @@ module.exports = {
548
600
  envelope,
549
601
  refusal,
550
602
  ceilingPercent,
603
+ lastModel,
551
604
  run,
552
605
  };