claude-usage-limits 1.25.0 → 1.39.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/README.md +107 -2
- package/commands/relay.md +10 -5
- package/commands/usage-mode.md +3 -0
- package/package.json +4 -3
- package/skills/usage-limits/SKILL.md +39 -2
- package/skills/usage-limits/references/how-it-works.md +3 -2
- package/skills/usage-limits/scripts/agy-hook.js +11 -14
- package/skills/usage-limits/scripts/brief.js +281 -36
- package/skills/usage-limits/scripts/ceiling.js +73 -13
- package/skills/usage-limits/scripts/defer.js +10 -8
- package/skills/usage-limits/scripts/feed.js +36 -8
- package/skills/usage-limits/scripts/install-antigravity.js +2 -1
- package/skills/usage-limits/scripts/lowpower.js +1 -1
- package/skills/usage-limits/scripts/mode.js +118 -8
- package/skills/usage-limits/scripts/pulse.js +82 -29
- package/skills/usage-limits/scripts/relay.js +693 -65
- package/skills/usage-limits/scripts/stop.js +2 -2
- package/skills/usage-limits/scripts/tally.js +10 -1
- package/skills/usage-limits/scripts/usage.js +113 -14
- package/skills/usage-limits/scripts/wake.js +356 -31
|
@@ -175,7 +175,7 @@ function bindingReset(now) {
|
|
|
175
175
|
function sessionId(argv, env) {
|
|
176
176
|
const at = argv.indexOf('--session-id');
|
|
177
177
|
if (at !== -1 && argv[at + 1]) return argv[at + 1];
|
|
178
|
-
return env.CLAUDE_SESSION_ID || env.CODEX_SESSION_ID || 'defer-' + Date.now().toString(36);
|
|
178
|
+
return env.CLAUDE_CODE_SESSION_ID || env.CLAUDE_SESSION_ID || env.CODEX_SESSION_ID || 'defer-' + Date.now().toString(36);
|
|
179
179
|
}
|
|
180
180
|
|
|
181
181
|
function argOf(argv, name) {
|
|
@@ -210,7 +210,7 @@ function plan(options) {
|
|
|
210
210
|
|
|
211
211
|
function status(now) {
|
|
212
212
|
const state = relay.read();
|
|
213
|
-
const armed = state.armed;
|
|
213
|
+
const armed = relay.armedFor(state, process.env.CLAUDE_CODE_SESSION_ID) || state.armed;
|
|
214
214
|
if (!armed) return 'Nothing is deferred.';
|
|
215
215
|
const when = Number(armed.wakeAt);
|
|
216
216
|
const deferred = armed.deferred === true;
|
|
@@ -230,9 +230,10 @@ function status(now) {
|
|
|
230
230
|
|
|
231
231
|
function cancel() {
|
|
232
232
|
const state = relay.read();
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
const
|
|
233
|
+
const own = relay.armedFor(state, process.env.CLAUDE_CODE_SESSION_ID) || state.armed;
|
|
234
|
+
if (!own) return 'Nothing was deferred.';
|
|
235
|
+
const label = formatClock(Number(own.wakeAt));
|
|
236
|
+
const result = relay.disarm('cancelled by hand', Date.now(), own.id);
|
|
236
237
|
return result && result.ok === false
|
|
237
238
|
? 'Could not cancel: ' + result.error
|
|
238
239
|
: 'Cancelled the run booked for ' + label + '.';
|
|
@@ -286,9 +287,10 @@ function main(argv, now) {
|
|
|
286
287
|
// session can tell the two apart - they read the same record.
|
|
287
288
|
try {
|
|
288
289
|
const held = relay.read();
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
290
|
+
const mine = (typeof sessionId !== 'undefined' && relay.armedFor(held, sessionId)) || held.armed;
|
|
291
|
+
if (mine) {
|
|
292
|
+
mine.deferred = true;
|
|
293
|
+
mine.continuation = Boolean(work);
|
|
292
294
|
relay.write(held);
|
|
293
295
|
}
|
|
294
296
|
} catch (err) {
|
|
@@ -124,9 +124,28 @@ function slotFrom(input, previous, now) {
|
|
|
124
124
|
cwd: typeof input.cwd === 'string' ? input.cwd : prior.cwd || null,
|
|
125
125
|
version: typeof input.version === 'string' ? input.version : prior.version || null,
|
|
126
126
|
fastMode: input.fast_mode === true,
|
|
127
|
+
cacheMiss: missFrom(input.prompt_cache, prior.cacheMiss),
|
|
127
128
|
};
|
|
128
129
|
}
|
|
129
130
|
|
|
131
|
+
// The last prompt-cache miss and its diagnosed causes, from the status line's
|
|
132
|
+
// prompt_cache object: last_miss_at is epoch seconds and last_miss_cause is
|
|
133
|
+
// null until the first miss and whenever no cause was found (Claude Code
|
|
134
|
+
// 2.1.260 and later; both per code.claude.com/docs/en/statusline). The moment
|
|
135
|
+
// is what lets the brief tell a new miss from the same one arriving on every
|
|
136
|
+
// refresh. A refresh without the object, or without a diagnosed cause, keeps
|
|
137
|
+
// what the last one said.
|
|
138
|
+
function missFrom(cache, prior) {
|
|
139
|
+
if (!cache || typeof cache !== 'object') return prior || null;
|
|
140
|
+
const at = typeof cache.last_miss_at === 'number' && Number.isFinite(cache.last_miss_at) ? cache.last_miss_at * 1000 : null;
|
|
141
|
+
const cause = cache.last_miss_cause;
|
|
142
|
+
const causes = cause && typeof cause === 'object' && Array.isArray(cause.causes)
|
|
143
|
+
? cause.causes.filter((name) => typeof name === 'string')
|
|
144
|
+
: [];
|
|
145
|
+
if (at === null || !causes.length) return prior || null;
|
|
146
|
+
return { at, causes };
|
|
147
|
+
}
|
|
148
|
+
|
|
130
149
|
function trim(all, keep) {
|
|
131
150
|
const ordered = Object.keys(all).sort((a, b) => (all[b].at || 0) - (all[a].at || 0));
|
|
132
151
|
const kept = {};
|
|
@@ -378,7 +397,13 @@ function runPrevious(command, raw, env, budgetMs) {
|
|
|
378
397
|
windowsHide: true,
|
|
379
398
|
stdio: ['pipe', 'pipe', 'ignore'],
|
|
380
399
|
});
|
|
381
|
-
|
|
400
|
+
// A command that never reads its stdin - echo, most one-line status lines -
|
|
401
|
+
// can exit before the JSON is written, and the write then fails with EPIPE.
|
|
402
|
+
// That says nothing about the command, which ran and printed. Measured on a
|
|
403
|
+
// Linux runner: the whole test took 85 ms, so no timeout was involved, and
|
|
404
|
+
// the line was dropped anyway. Only the command's own failure drops it.
|
|
405
|
+
if (result.status !== 0) return '';
|
|
406
|
+
if (result.error && result.error.code !== 'EPIPE') return '';
|
|
382
407
|
return String(result.stdout || '').replace(/\s+$/, '');
|
|
383
408
|
} catch (err) {
|
|
384
409
|
return '';
|
|
@@ -490,14 +515,16 @@ async function main(argv) {
|
|
|
490
515
|
headersAt: slot ? slot.headersAt : null,
|
|
491
516
|
model: slot ? slot.model : null,
|
|
492
517
|
modelName: slot ? slot.modelName : null,
|
|
493
|
-
// Claude Code hands this line the effort outright,
|
|
494
|
-
//
|
|
495
|
-
//
|
|
496
|
-
//
|
|
518
|
+
// Claude Code hands this line the effort outright, and that value is
|
|
519
|
+
// read first: the slot beside it was loaded from disk before this
|
|
520
|
+
// update was stored, so on a session's very first line it is empty and
|
|
521
|
+
// trusting it printed no effort at all until the second update. The
|
|
522
|
+
// slot still answers for a build that sends none, and only when it has
|
|
523
|
+
// none either is the transcript worth a look.
|
|
497
524
|
effort:
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
525
|
+
(input && input.effort && typeof input.effort.level === 'string' && input.effort.level)
|
|
526
|
+
|| (slot && slot.effort)
|
|
527
|
+
|| view.pickEffort(
|
|
501
528
|
null,
|
|
502
529
|
usage.liveEffort(mine || (slot && slot.sessionId) || null),
|
|
503
530
|
collected.settings ? collected.settings.effortLevel : null
|
|
@@ -559,6 +586,7 @@ module.exports = {
|
|
|
559
586
|
CHAIN_TIMEOUT_MS,
|
|
560
587
|
feedFile,
|
|
561
588
|
readFeed,
|
|
589
|
+
missFrom,
|
|
562
590
|
writeFeed,
|
|
563
591
|
slotFrom,
|
|
564
592
|
record,
|
|
@@ -48,7 +48,8 @@ function pluginDir() {
|
|
|
48
48
|
// runs hook commands through `cmd /c` on Windows and `sh -c` elsewhere, so the
|
|
49
49
|
// quoting has to survive both.
|
|
50
50
|
function quote(file) {
|
|
51
|
-
|
|
51
|
+
const normalized = String(file).replace(/\\/g, '/');
|
|
52
|
+
return normalized.includes(' ') ? '"' + normalized + '"' : normalized;
|
|
52
53
|
}
|
|
53
54
|
|
|
54
55
|
function scriptPath(name) {
|
|
@@ -192,7 +192,7 @@ function describe(settings, state) {
|
|
|
192
192
|
function logSettingsChange(changes, direction, env) {
|
|
193
193
|
if (!changes || !changes.length) return;
|
|
194
194
|
const e = env || process.env;
|
|
195
|
-
const by = e.CLAUDE_SESSION_ID || e.CODEX_SESSION_ID ? 'claude' : 'user';
|
|
195
|
+
const by = e.CLAUDE_CODE_SESSION_ID || e.CLAUDE_SESSION_ID || e.CODEX_SESSION_ID ? 'claude' : 'user';
|
|
196
196
|
try {
|
|
197
197
|
const mode = require('./mode.js');
|
|
198
198
|
for (const change of changes) {
|
|
@@ -388,6 +388,14 @@ function empty() {
|
|
|
388
388
|
// reported: past it, fan-out calls are refused at the hook. Null means no
|
|
389
389
|
// ceiling, and a ceiling nobody set never refuses anything. See ceiling.js.
|
|
390
390
|
ceilingPercent: null,
|
|
391
|
+
// WHOSE cap it is. A cap is said about the work in front of someone, so
|
|
392
|
+
// it binds only in the session that set it; carrying it into the next one
|
|
393
|
+
// means refusing fan-outs on a fresh window for a reason nobody gave. A
|
|
394
|
+
// stored cap with no session here predates this rule and is ignored.
|
|
395
|
+
ceilingSession: null,
|
|
396
|
+
// Applies to EVERY session, survives a reset, and is only cleared by
|
|
397
|
+
// asking. The session-scoped cap above lapses on purpose; this does not.
|
|
398
|
+
ceilingAlways: null,
|
|
391
399
|
setAt: null,
|
|
392
400
|
setBy: null,
|
|
393
401
|
session: null,
|
|
@@ -431,6 +439,11 @@ function read() {
|
|
|
431
439
|
base.guardPercent = Number.isFinite(parsed.guardPercent) ? parsed.guardPercent : null;
|
|
432
440
|
// Anything outside 1-100 is not a ceiling, and enforcing a number that was
|
|
433
441
|
// never a percentage would refuse work over a typo.
|
|
442
|
+
base.ceilingSession = typeof parsed.ceilingSession === 'string' ? parsed.ceilingSession : null;
|
|
443
|
+
base.ceilingAlways =
|
|
444
|
+
Number.isFinite(parsed.ceilingAlways) && parsed.ceilingAlways > 0 && parsed.ceilingAlways <= 100
|
|
445
|
+
? parsed.ceilingAlways
|
|
446
|
+
: null;
|
|
434
447
|
base.ceilingPercent =
|
|
435
448
|
Number.isFinite(parsed.ceilingPercent) && parsed.ceilingPercent > 0 && parsed.ceilingPercent <= 100
|
|
436
449
|
? parsed.ceilingPercent
|
|
@@ -1060,12 +1073,39 @@ function tierNow(options) {
|
|
|
1060
1073
|
},
|
|
1061
1074
|
running: {
|
|
1062
1075
|
model: running,
|
|
1063
|
-
effort: effort ? effort.effort : null,
|
|
1076
|
+
effort: ultracodeName(effort ? effort.effort : null, sessionId, opts.now),
|
|
1064
1077
|
source: effort ? effort.source : null,
|
|
1065
1078
|
},
|
|
1066
1079
|
};
|
|
1067
1080
|
}
|
|
1068
1081
|
|
|
1082
|
+
// Ultracode reads as "xhigh" everywhere, and that is technically true and
|
|
1083
|
+
// practically wrong.
|
|
1084
|
+
//
|
|
1085
|
+
// Ultracode RESOLVES to xhigh, so every effort reader - the settings file, the
|
|
1086
|
+
// live transcript, CLAUDE_EFFORT - honestly reports xhigh, and the display said
|
|
1087
|
+
// xhigh at somebody who had typed ultracode. The two are the same amount of
|
|
1088
|
+
// reasoning and a very different amount of everything else, so a person who
|
|
1089
|
+
// asked for one and is shown the other has no way to tell it took.
|
|
1090
|
+
//
|
|
1091
|
+
// The plugin already knows. brief.js sets an ultracode flag on the session's
|
|
1092
|
+
// activity mark the moment the word appears in a prompt, and it STICKS for the
|
|
1093
|
+
// session because ultracode is a session-level trigger rather than a per-prompt
|
|
1094
|
+
// one. Nothing read it back. This does.
|
|
1095
|
+
//
|
|
1096
|
+
// Only xhigh is renamed. If some future effort is higher, reporting it as
|
|
1097
|
+
// ultracode would be a downgrade dressed up as a label, so it is left alone.
|
|
1098
|
+
function ultracodeName(effort, sessionId, now) {
|
|
1099
|
+
if (effort !== 'xhigh' || !sessionId) return effort;
|
|
1100
|
+
try {
|
|
1101
|
+
const entry = require('./activity.js').read()[sessionId];
|
|
1102
|
+
return entry && entry.ultracode ? 'ultracode' : effort;
|
|
1103
|
+
} catch (err) {
|
|
1104
|
+
// The real effort is a better answer than no answer.
|
|
1105
|
+
return effort;
|
|
1106
|
+
}
|
|
1107
|
+
}
|
|
1108
|
+
|
|
1069
1109
|
function sameFamily(a, b) {
|
|
1070
1110
|
const left = modelRank(a);
|
|
1071
1111
|
const right = modelRank(b);
|
|
@@ -1099,10 +1139,58 @@ function tierLine(tier, options) {
|
|
|
1099
1139
|
if (opts.terse) {
|
|
1100
1140
|
return differs ? runningText + source + ', yours ' + baseText : runningText + source;
|
|
1101
1141
|
}
|
|
1142
|
+
// Reporting the tier is not the same as deciding about it.
|
|
1143
|
+
//
|
|
1144
|
+
// This line has always said what is running and left it there, and the
|
|
1145
|
+
// observed result is that the tier is simply never revisited: a session opens
|
|
1146
|
+
// on the dearest model at the dearest effort and stays there through work
|
|
1147
|
+
// that did not need either. Saying the number is not a prompt to act on it.
|
|
1148
|
+
//
|
|
1149
|
+
// So when the tier IS an expensive one, the line asks for a decision rather
|
|
1150
|
+
// than leaving a fact lying around. Only then: on a cheap tier there is
|
|
1151
|
+
// nothing to decide, and a sentence asking every turn would be exactly the
|
|
1152
|
+
// per-turn cost this plugin exists to avoid.
|
|
1153
|
+
//
|
|
1154
|
+
// Deliberately names no slash command. /model and /effort do not exist in
|
|
1155
|
+
// Codex, and telling Codex to run them is telling it to do nothing while
|
|
1156
|
+
// believing it acted - the mistake levers() already exists to prevent.
|
|
1157
|
+
const dear = topTier(run.model || base.model, run.effort || base.effort);
|
|
1158
|
+
// What the plugin can actually do about the tier, stated exactly.
|
|
1159
|
+
//
|
|
1160
|
+
// Nothing in a hook, a tool or the SDK can change a running session's model
|
|
1161
|
+
// or effort - that was researched to the primary sources on 2026-09-14 and
|
|
1162
|
+
// the answer is a flat no. /model and /effort are the user's, by design. So
|
|
1163
|
+
// the most honest thing this line can do is name the decision and, where it
|
|
1164
|
+
// is cheap, say so: on Fable 5.1 on a subscription an /effort change keeps
|
|
1165
|
+
// the prompt cache (prompt-caching doc, v2.1.260+), which makes stepping
|
|
1166
|
+
// effort down mid-session free. On every other model it rebuilds the cache,
|
|
1167
|
+
// and on a large context that can cost more than a few cheaper turns save -
|
|
1168
|
+
// so there the advice is to choose at the START of a session.
|
|
1169
|
+
const onFable = sameFamily(run.model || base.model, 'fable');
|
|
1170
|
+
const decide = dear
|
|
1171
|
+
? ' Decide in one line whether the work in front of you needs ' + runningText + '.' +
|
|
1172
|
+
(onFable
|
|
1173
|
+
? ' On Fable 5.1 an /effort change keeps the cache, so if the next stretch is mechanical, /effort low now costs nothing.'
|
|
1174
|
+
: ' Changing effort or model mid-session rebuilds the cache, so if it does not, say so and choose lower at the next session start rather than switching now.')
|
|
1175
|
+
: '';
|
|
1102
1176
|
return differs
|
|
1103
1177
|
? 'Running ' + runningText + source + '; your baseline is ' + baseText + '. The gap is the ' +
|
|
1104
|
-
'interesting part: your baseline is yours and is not being changed.'
|
|
1105
|
-
: 'Running ' + runningText + source + '.';
|
|
1178
|
+
'interesting part: your baseline is yours and is not being changed.' + decide
|
|
1179
|
+
: 'Running ' + runningText + source + '.' + decide;
|
|
1180
|
+
}
|
|
1181
|
+
|
|
1182
|
+
// Whether this tier is dear enough to be worth a decision. Opus or above, or an
|
|
1183
|
+
// effort at or above xhigh; either alone is enough to be worth asking.
|
|
1184
|
+
//
|
|
1185
|
+
// MODEL_ORDER runs CHEAPEST first - haiku, sonnet, opus, mythos, fable - so the
|
|
1186
|
+
// dear end is a HIGH rank, not a low one. Getting that backwards asked haiku to
|
|
1187
|
+
// justify itself and let opus through silently, which is the exact inverse of
|
|
1188
|
+
// the point.
|
|
1189
|
+
function topTier(model, effort) {
|
|
1190
|
+
const rank = modelRank(model);
|
|
1191
|
+
const dearModel = rank !== null && rank >= MODEL_ORDER.indexOf('opus');
|
|
1192
|
+
const dearEffort = ['xhigh', 'max', 'ultra', 'ultracode'].includes(String(effort || '').toLowerCase());
|
|
1193
|
+
return dearModel || dearEffort;
|
|
1106
1194
|
}
|
|
1107
1195
|
|
|
1108
1196
|
// ---------------------------------------------------------------------------
|
|
@@ -1240,7 +1328,10 @@ function setMode(name, opts, now) {
|
|
|
1240
1328
|
state.setAt = at;
|
|
1241
1329
|
state.setBy = 'user';
|
|
1242
1330
|
if (opts && Number.isFinite(opts.guard)) state.guardPercent = opts.guard;
|
|
1243
|
-
if (opts && Number.isFinite(opts.ceiling))
|
|
1331
|
+
if (opts && Number.isFinite(opts.ceiling)) {
|
|
1332
|
+
state.ceilingPercent = opts.ceiling;
|
|
1333
|
+
state.ceilingSession = opts.sessionId || null;
|
|
1334
|
+
}
|
|
1244
1335
|
write(state);
|
|
1245
1336
|
logChange({ plane: 'mode', key: 'mode', from: before, to: name, by: 'user', reason: null }, at);
|
|
1246
1337
|
|
|
@@ -1288,7 +1379,10 @@ function main(argv) {
|
|
|
1288
1379
|
const next = args[at + 1];
|
|
1289
1380
|
return next && next.indexOf('--') !== 0 ? next : null;
|
|
1290
1381
|
};
|
|
1291
|
-
|
|
1382
|
+
// CLAUDE_CODE_SESSION_ID is the one Claude Code actually exports;
|
|
1383
|
+
// CLAUDE_SESSION_ID never existed, so this read null and a cap set from the
|
|
1384
|
+
// command line could record no owner and therefore never bind.
|
|
1385
|
+
const sessionId = value('--session-id') || process.env.CLAUDE_CODE_SESSION_ID || process.env.CLAUDE_SESSION_ID || null;
|
|
1292
1386
|
const decided = resolve({ sessionId });
|
|
1293
1387
|
|
|
1294
1388
|
if (flag('--list')) return list(decided);
|
|
@@ -1464,7 +1558,11 @@ function main(argv) {
|
|
|
1464
1558
|
if (ceilingArg.error) return ceilingArg.error;
|
|
1465
1559
|
const state = read();
|
|
1466
1560
|
const previous = state.ceilingPercent;
|
|
1561
|
+
if (flag('--always')) {
|
|
1562
|
+
state.ceilingAlways = ceilingArg.clear ? null : ceilingArg.percent;
|
|
1563
|
+
}
|
|
1467
1564
|
state.ceilingPercent = ceilingArg.clear ? null : ceilingArg.percent;
|
|
1565
|
+
state.ceilingSession = ceilingArg.clear ? null : sessionId;
|
|
1468
1566
|
write(state);
|
|
1469
1567
|
logChange(
|
|
1470
1568
|
{ plane: 'mode', key: 'ceiling', from: previous, to: state.ceilingPercent, by: 'user', reason: null },
|
|
@@ -1481,6 +1579,8 @@ function clearCeiling(now) {
|
|
|
1481
1579
|
const state = read();
|
|
1482
1580
|
const previous = state.ceilingPercent;
|
|
1483
1581
|
state.ceilingPercent = null;
|
|
1582
|
+
state.ceilingSession = null;
|
|
1583
|
+
state.ceilingAlways = null;
|
|
1484
1584
|
write(state);
|
|
1485
1585
|
logChange({ plane: 'mode', key: 'ceiling', from: previous, to: null, by: 'user', reason: null }, now);
|
|
1486
1586
|
}
|
|
@@ -1489,14 +1589,22 @@ function clearCeiling(now) {
|
|
|
1489
1589
|
// refusal, because a percentage on its own does not tell anyone what changes.
|
|
1490
1590
|
function ceilingLine() {
|
|
1491
1591
|
const state = read();
|
|
1592
|
+
if (Number.isFinite(state.ceilingAlways)) {
|
|
1593
|
+
return (
|
|
1594
|
+
'Standing cap ' + state.ceilingAlways + '%, for this and every future session on this ' +
|
|
1595
|
+
'machine, including after a limit resets. Past it, fan-out calls are refused at the hook; ' +
|
|
1596
|
+
'nothing else is blocked. "mode --cap off --always" removes it.'
|
|
1597
|
+
);
|
|
1598
|
+
}
|
|
1492
1599
|
if (state.ceilingPercent === null || state.ceilingPercent === undefined) {
|
|
1493
1600
|
return 'Ceiling off. Nothing is refused; the plugin reports and does not intervene.';
|
|
1494
1601
|
}
|
|
1495
1602
|
return (
|
|
1496
|
-
'
|
|
1497
|
-
'
|
|
1603
|
+
'Cap ' + state.ceilingPercent + '%' + (state.ceilingSession ? ', for THIS session only' : ' (stored against no session, so it is ignored - set it again to apply it here)') +
|
|
1604
|
+
'. Past that, fan-out calls (Agent, Task, Workflow and ' +
|
|
1605
|
+
'their equivalents) are refused at the hook. Nothing else ' +
|
|
1498
1606
|
'is blocked, so the work still finishes - sequentially, in one session, which is where the ' +
|
|
1499
|
-
'saving comes from. "mode --cap off" removes it.'
|
|
1607
|
+
'saving comes from. It lapses when this session ends; "mode --cap off" removes it now.'
|
|
1500
1608
|
);
|
|
1501
1609
|
}
|
|
1502
1610
|
|
|
@@ -1626,6 +1734,8 @@ module.exports = {
|
|
|
1626
1734
|
adviceDecline,
|
|
1627
1735
|
adviceMute,
|
|
1628
1736
|
tierNow,
|
|
1737
|
+
ultracodeName,
|
|
1738
|
+
topTier,
|
|
1629
1739
|
tierLine,
|
|
1630
1740
|
ledger,
|
|
1631
1741
|
explain,
|
|
@@ -137,6 +137,11 @@ function pulseText(parts) {
|
|
|
137
137
|
|
|
138
138
|
const head = '[usage-limits] ' + (parts.fanout ? 'Before this fan-out: ' : '') + bits.join(', ') + '.';
|
|
139
139
|
if (parts.fanout) {
|
|
140
|
+
// Parallel Agent calls fire this once per call, so the advice sentence was
|
|
141
|
+
// being said six times in a row for one reading. The reading itself is
|
|
142
|
+
// never throttled; the advice is said once per session per ten minutes,
|
|
143
|
+
// through the same state file as every other throttle here.
|
|
144
|
+
if (parts.advice === false) return head;
|
|
140
145
|
// Said before every Workflow or Agent call. The agents spend this same
|
|
141
146
|
// window, nothing can speak again until they stop, and a main-loop turn
|
|
142
147
|
// with a large context costs more than one whole fresh-context agent.
|
|
@@ -147,6 +152,10 @@ function pulseText(parts) {
|
|
|
147
152
|
'. Fewer agents with a fresh context beat another turn of a long one.'
|
|
148
153
|
);
|
|
149
154
|
}
|
|
155
|
+
// The tight and gone sentences are advice too, and they were said on every
|
|
156
|
+
// due pulse - three times in ten minutes on 2026-09-20. Same slot and same
|
|
157
|
+
// ten minutes as the fan-out advice; the reading itself is never withheld.
|
|
158
|
+
if ((parts.pressure === 'gone' || parts.pressure === 'tight') && parts.advice === false) return head;
|
|
150
159
|
if (parts.pressure === 'gone') {
|
|
151
160
|
return head + ' The budget is gone. Stop adding work, save what exists and write the handoff.';
|
|
152
161
|
}
|
|
@@ -249,10 +258,14 @@ async function run(now, hookInput) {
|
|
|
249
258
|
// ask the model to agree with it.
|
|
250
259
|
if (event === 'PreToolUse' && ceiling.isMultiplier(tool)) {
|
|
251
260
|
try {
|
|
261
|
+
const worst = ceilingPercent(now, null, hookInput && hookInput.transcript_path);
|
|
252
262
|
const at = ceiling.assess({
|
|
253
|
-
percent:
|
|
263
|
+
percent: worst ? worst.percent : null,
|
|
264
|
+
label: worst ? worst.label : null,
|
|
254
265
|
state: budget.state,
|
|
255
266
|
env: process.env,
|
|
267
|
+
// The cap only binds in the session that set it.
|
|
268
|
+
sessionId,
|
|
256
269
|
});
|
|
257
270
|
const call = ceiling.verdict(at, tool);
|
|
258
271
|
if (call.decision === 'deny') return { deny: true, reason: call.reason };
|
|
@@ -287,10 +300,12 @@ async function run(now, hookInput) {
|
|
|
287
300
|
// The cheap path, and the one taken almost every time. A fan-out is never
|
|
288
301
|
// throttled: it is said every time, because every time it is about to cost.
|
|
289
302
|
if (!fanout && !due(all, throttleKey, now, every)) return '';
|
|
303
|
+
const adviceKey = (sessionId || '_') + '#fanout-advice';
|
|
304
|
+
const sayAdvice = fanout && due(all, adviceKey, now, 10 * 60 * 1000);
|
|
290
305
|
|
|
291
306
|
// Claimed before the scan rather than after, so a slow scan cannot let a
|
|
292
307
|
// second tool call start another one.
|
|
293
|
-
writeState(trim(all, throttleKey, now));
|
|
308
|
+
writeState(sayAdvice ? trim(trim(all, throttleKey, now), adviceKey, now) : trim(all, throttleKey, now));
|
|
294
309
|
|
|
295
310
|
// A reading as old as the interval is replaced with the one Claude Code
|
|
296
311
|
// would take for /usage, so a turn that runs for an hour is measured
|
|
@@ -396,6 +411,11 @@ async function run(now, hookInput) {
|
|
|
396
411
|
return recheck;
|
|
397
412
|
}
|
|
398
413
|
|
|
414
|
+
// The pressure advice shares the fan-out advice's slot and its ten minutes.
|
|
415
|
+
// Unlike the fan-out's, it is stamped after the fact and only when a line
|
|
416
|
+
// actually carried it: a pulse with no reading to hang it on says nothing,
|
|
417
|
+
// and must not spend the slot on nothing.
|
|
418
|
+
const sayPressureAdvice = !fanout && (pressure === 'tight' || pressure === 'gone') && due(readState(), adviceKey, now, 10 * 60 * 1000);
|
|
399
419
|
const spoken = pulseText({
|
|
400
420
|
label: binding.label,
|
|
401
421
|
percentUsed: binding.percentUsed,
|
|
@@ -408,50 +428,82 @@ async function run(now, hookInput) {
|
|
|
408
428
|
sessions: active,
|
|
409
429
|
pressure,
|
|
410
430
|
fanout,
|
|
431
|
+
advice: fanout ? sayAdvice : sayPressureAdvice,
|
|
411
432
|
});
|
|
433
|
+
if (spoken && sayPressureAdvice) writeState(trim(readState(), adviceKey, now));
|
|
412
434
|
return recheck ? (spoken ? spoken + ' ' + recheck : recheck) : spoken;
|
|
413
435
|
}
|
|
414
436
|
|
|
415
|
-
// The cheapest
|
|
437
|
+
// The cheapest reading good enough to enforce a ceiling against.
|
|
416
438
|
//
|
|
417
439
|
// The ceiling is checked before every fan-out, which is far too often to scan
|
|
418
|
-
// transcripts for.
|
|
419
|
-
//
|
|
420
|
-
//
|
|
421
|
-
//
|
|
440
|
+
// transcripts for. snapshotWindows() is the no-scan view every other cheap
|
|
441
|
+
// reader uses: the account snapshot, the corrections the last scan left
|
|
442
|
+
// behind, and whether each window is one this agent can spend into. The
|
|
443
|
+
// fullest of those wins, because a ceiling means "no window past here".
|
|
422
444
|
//
|
|
423
|
-
//
|
|
424
|
-
//
|
|
445
|
+
// This used to take the highest number on disk, whatever window it belonged
|
|
446
|
+
// to - including a weekly scoped to a model this session was not running, and
|
|
447
|
+
// the snapshot of a window that had already reset. See ceiling.worstWindow().
|
|
448
|
+
//
|
|
449
|
+
// Returns null when nothing is readable, and a ceiling with no reading behind
|
|
450
|
+
// it never refuses. Guessing high would block work over a number nobody
|
|
425
451
|
// measured; guessing low would not be a ceiling at all.
|
|
426
|
-
function ceilingPercent(now) {
|
|
427
|
-
let worst = null;
|
|
428
|
-
const consider = (value) => {
|
|
429
|
-
if (!Number.isFinite(value)) return;
|
|
430
|
-
if (worst === null || value > worst) worst = value;
|
|
431
|
-
};
|
|
452
|
+
function ceilingPercent(now, collected, transcriptPath) {
|
|
432
453
|
const codexHome = usage.isCodex() ? require('./codex.js').homeDir() : null;
|
|
433
454
|
try {
|
|
434
|
-
const
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
}
|
|
455
|
+
const running = lastModel(transcriptPath);
|
|
456
|
+
return ceiling.worstWindow(
|
|
457
|
+
usage.snapshotWindows(collected || usage.collect(now), now, codexHome, running ? [running] : [])
|
|
458
|
+
);
|
|
439
459
|
} catch (err) {
|
|
440
|
-
//
|
|
460
|
+
// No snapshot is a reason not to enforce, not a reason to throw.
|
|
461
|
+
return null;
|
|
441
462
|
}
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
// The model this session's last reply came from, off the tail of its transcript.
|
|
466
|
+
//
|
|
467
|
+
// The setting says what a session started on; /model changes what it runs
|
|
468
|
+
// without touching settings.json. A weekly scoped to the model actually
|
|
469
|
+
// running must count, so the pulse adds this to the setting rather than
|
|
470
|
+
// replacing it - suppressing a window takes both sides agreeing. One read of
|
|
471
|
+
// the last 64 KB, from the end, whole lines only. Null on anything unexpected.
|
|
472
|
+
function lastModel(transcriptPath) {
|
|
473
|
+
if (!transcriptPath) return null;
|
|
474
|
+
let fd = null;
|
|
442
475
|
try {
|
|
443
|
-
|
|
444
|
-
const
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
476
|
+
fd = fs.openSync(transcriptPath, 'r');
|
|
477
|
+
const size = fs.fstatSync(fd).size;
|
|
478
|
+
const length = Math.min(size, 64 * 1024);
|
|
479
|
+
const buffer = Buffer.alloc(length);
|
|
480
|
+
fs.readSync(fd, buffer, 0, length, size - length);
|
|
481
|
+
const lines = buffer.toString('utf8').split('\n');
|
|
482
|
+
// The first line of a partial read is cut; only a read from the start is whole.
|
|
483
|
+
if (length < size) lines.shift();
|
|
484
|
+
for (let i = lines.length - 1; i >= 0; i -= 1) {
|
|
485
|
+
if (!lines[i].includes('"assistant"')) continue;
|
|
486
|
+
try {
|
|
487
|
+
const entry = JSON.parse(lines[i]);
|
|
488
|
+
const model = entry && entry.type === 'assistant' && entry.message && entry.message.model;
|
|
489
|
+
// "<synthetic>" marks a reply Claude Code wrote itself, not a model.
|
|
490
|
+
if (model && typeof model === 'string' && !model.startsWith('<')) return model;
|
|
491
|
+
} catch (err) {
|
|
492
|
+
// A half-written last line while the transcript is being appended.
|
|
449
493
|
}
|
|
450
494
|
}
|
|
495
|
+
return null;
|
|
451
496
|
} catch (err) {
|
|
452
|
-
|
|
497
|
+
return null;
|
|
498
|
+
} finally {
|
|
499
|
+
if (fd !== null) {
|
|
500
|
+
try {
|
|
501
|
+
fs.closeSync(fd);
|
|
502
|
+
} catch (err) {
|
|
503
|
+
// Already closed is fine.
|
|
504
|
+
}
|
|
505
|
+
}
|
|
453
506
|
}
|
|
454
|
-
return worst;
|
|
455
507
|
}
|
|
456
508
|
|
|
457
509
|
// PostToolUse does not take plain stdout as context the way UserPromptSubmit
|
|
@@ -548,5 +600,6 @@ module.exports = {
|
|
|
548
600
|
envelope,
|
|
549
601
|
refusal,
|
|
550
602
|
ceilingPercent,
|
|
603
|
+
lastModel,
|
|
551
604
|
run,
|
|
552
605
|
};
|