claude-multiacc 1.0.14 → 1.0.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -6
- package/bin/claude +215 -12
- package/bin/claude-accounts +306 -30
- package/install.sh +25 -4
- package/lib/report.py +81 -2
- package/package.json +1 -1
- package/tests/run-tests.sh +522 -8
package/bin/claude-accounts
CHANGED
|
@@ -267,6 +267,25 @@ sys.path = [sys.argv[3]] + [p for p in sys.path if p not in ('', '.')]
|
|
|
267
267
|
from audit import audit_account # noqa: E402 (shared with the shim's rule)
|
|
268
268
|
machine = sys.argv[4]
|
|
269
269
|
now = time.time()
|
|
270
|
+
# Must match the shim's window (bin/claude), or status would call data "fresh" that
|
|
271
|
+
# selection has already stopped ranking on. Same fallback as the shim for a garbled
|
|
272
|
+
# override: a bad env var must never be the thing that stops status from printing.
|
|
273
|
+
try:
|
|
274
|
+
STALE_AFTER = int(os.environ.get('CLAUDE_MULTIACC_STALE_AFTER') or 3600)
|
|
275
|
+
except ValueError:
|
|
276
|
+
STALE_AFTER = 3600
|
|
277
|
+
if STALE_AFTER <= 0:
|
|
278
|
+
STALE_AFTER = 3600
|
|
279
|
+
|
|
280
|
+
def telem_fetched_at(aid):
|
|
281
|
+
"""Epoch of aid's last SUCCESSFUL usage fetch, or 0. Never raises: this file is
|
|
282
|
+
hand-editable and syncs between machines, and one corrupt copy must not stop
|
|
283
|
+
status from printing the pool-wide verdict it exists to show."""
|
|
284
|
+
try:
|
|
285
|
+
v = json.load(open(os.path.join(root, aid, 'limits.json'))).get('fetched_at', 0)
|
|
286
|
+
except Exception:
|
|
287
|
+
return 0
|
|
288
|
+
return v if isinstance(v, (int, float)) and not isinstance(v, bool) and v > 0 else 0
|
|
270
289
|
|
|
271
290
|
def last_pick(aid):
|
|
272
291
|
path = os.path.join(root, 'selection.log')
|
|
@@ -288,12 +307,15 @@ print(f"server : {doc.get('server','-')} (root: {doc.get('server_root','-')}
|
|
|
288
307
|
print(f"threshold : {doc.get('threshold', 90)}% (any bucket at/above => account excluded)")
|
|
289
308
|
print()
|
|
290
309
|
needs_login = []
|
|
310
|
+
selectable_ids = []
|
|
291
311
|
for a in doc.get('accounts', []):
|
|
292
312
|
aid = a['id']
|
|
293
313
|
d = os.path.join(root, aid)
|
|
294
314
|
st = audit_account(root, a, machine=machine)
|
|
295
315
|
if st['state'] in ('expired', 'blocked', 'missing'):
|
|
296
316
|
needs_login.append(aid)
|
|
317
|
+
else:
|
|
318
|
+
selectable_ids.append(aid)
|
|
297
319
|
banner = {'ok': '', 'remote': ' [not logged in here — grant lives elsewhere]',
|
|
298
320
|
'missing': ' ** NO LOGIN — claude-accounts relogin %s **' % aid,
|
|
299
321
|
'expired': ' ** LOGIN EXPIRED — claude-accounts relogin %s **' % aid,
|
|
@@ -325,9 +347,33 @@ for a in doc.get('accounts', []):
|
|
|
325
347
|
if os.path.isfile(lpath):
|
|
326
348
|
try:
|
|
327
349
|
lim = json.load(open(lpath))
|
|
328
|
-
|
|
350
|
+
if not isinstance(lim, dict):
|
|
351
|
+
raise ValueError('not an object')
|
|
352
|
+
fetched = telem_fetched_at(aid)
|
|
353
|
+
age = int(now - fetched) if fetched else None
|
|
329
354
|
parts = [f"{b['name']}={b['percent']}%" for b in lim.get('buckets', [])]
|
|
330
|
-
|
|
355
|
+
if age is None:
|
|
356
|
+
shown = 'never fetched'
|
|
357
|
+
elif age >= 86400:
|
|
358
|
+
shown = f'{age // 86400}d {age % 86400 // 3600}h old'
|
|
359
|
+
elif age >= 3600:
|
|
360
|
+
shown = f'{age // 3600}h {age % 3600 // 60}m old'
|
|
361
|
+
else:
|
|
362
|
+
shown = f'{age}s old'
|
|
363
|
+
# Loud, because stale numbers do not look stale: they look like a healthy
|
|
364
|
+
# account sitting at 2%. Past the ranking window the shim ignores them
|
|
365
|
+
# entirely and picks at random, so the reading below is decoration.
|
|
366
|
+
stale = age is None or age > STALE_AFTER
|
|
367
|
+
flag = ' << STALE — NOT USED FOR RANKING' if stale else ''
|
|
368
|
+
print(f" limits : {' '.join(parts) or '(none)'} [{shown}, max {lim.get('max_percent')}%]{flag}")
|
|
369
|
+
err = lim.get('last_error')
|
|
370
|
+
if isinstance(err, str) and err:
|
|
371
|
+
when = lim.get('last_error_at')
|
|
372
|
+
ago = f' ({int(now - when)}s ago)' if isinstance(when, (int, float)) else ''
|
|
373
|
+
print(f" telemetry : last fetch FAILED{ago}: {err}")
|
|
374
|
+
retry = lim.get('retry_after', 0)
|
|
375
|
+
if isinstance(retry, (int, float)) and retry > now:
|
|
376
|
+
print(f" retrying in {int(retry - now)}s")
|
|
331
377
|
except Exception as e:
|
|
332
378
|
print(f" limits : unreadable ({e})")
|
|
333
379
|
else:
|
|
@@ -349,6 +395,40 @@ for a in doc.get('accounts', []):
|
|
|
349
395
|
if needs_login:
|
|
350
396
|
print(f"{len(needs_login)} account(s) are EXCLUDED from selection: {', '.join(needs_login)}")
|
|
351
397
|
print("What each one needs: claude-accounts expired")
|
|
398
|
+
# Pool-wide verdict. Per-account ages are easy to skim past; "the pool is picking at
|
|
399
|
+
# random" is not. This is the line that would have caught an eleven-day outage on the
|
|
400
|
+
# first day someone ran status.
|
|
401
|
+
# The verdict comes from lib/report.py so this text, `--json`, and the shim's own
|
|
402
|
+
# ranking can never drift apart — a status that disagrees with selection is worse than
|
|
403
|
+
# no status at all.
|
|
404
|
+
try:
|
|
405
|
+
from report import telemetry_state, build # noqa: E402
|
|
406
|
+
doc_rows = build(root, 'claude', machine, 'status')['accounts']
|
|
407
|
+
verdict = telemetry_state(doc_rows, now)
|
|
408
|
+
except Exception as e:
|
|
409
|
+
verdict = 'unknown'
|
|
410
|
+
verdict_err = str(e)[:200]
|
|
411
|
+
if verdict == 'unknown':
|
|
412
|
+
# Silence here would recreate the observability half of the incident: a blind pool
|
|
413
|
+
# that says nothing. Say the verdict could not be computed, and why.
|
|
414
|
+
print(f"RANKING STATE UNKNOWN: could not compute the pool-wide telemetry verdict "
|
|
415
|
+
f"({verdict_err}). Check the per-account ages above by hand.")
|
|
416
|
+
elif verdict == 'blind':
|
|
417
|
+
print("RANKING IS BLIND: no account has usage telemetry inside the "
|
|
418
|
+
f"{STALE_AFTER}s window, and the last readings are too old to mean anything, "
|
|
419
|
+
"so every account scores the same and `claude` picks at RANDOM — including "
|
|
420
|
+
"accounts that are nearly out of weekly headroom.")
|
|
421
|
+
print(" why : see the 'telemetry' lines above (a setup token cannot read the usage "
|
|
422
|
+
"endpoint — it has no user:profile scope; only an OAuth login on this machine can)")
|
|
423
|
+
print(" fix : claude-accounts limits --force # then, if it still fails:")
|
|
424
|
+
print(" claude-accounts login <acct-NN> # per account, on THIS machine")
|
|
425
|
+
elif verdict == 'degraded':
|
|
426
|
+
print("RANKING IS DEGRADED: no account has telemetry inside the "
|
|
427
|
+
f"{STALE_AFTER}s window, so `claude` is ranking on the last readings whose "
|
|
428
|
+
"weekly bucket has not reset yet. Better than random, but it cannot see usage "
|
|
429
|
+
"since those readings were taken.")
|
|
430
|
+
print(" fix : claude-accounts limits --force # then, if it still fails:")
|
|
431
|
+
print(" claude-accounts login <acct-NN> # per account, on THIS machine")
|
|
352
432
|
PYEOF
|
|
353
433
|
}
|
|
354
434
|
|
|
@@ -1016,7 +1096,7 @@ cmd_limits() {
|
|
|
1016
1096
|
[ "$threshold" -gt 90 ] && threshold=90
|
|
1017
1097
|
[ "$threshold" -lt 1 ] && threshold=90
|
|
1018
1098
|
"$PYBIN" - "$ACC_ROOT" "$threshold" "$quiet" "$USAGE_URL" "$force" <<'PYEOF' 2>>"$ACC_ROOT/limits.log"
|
|
1019
|
-
import json, os, sys, time, urllib.request
|
|
1099
|
+
import hashlib, json, os, sys, time, urllib.request
|
|
1020
1100
|
|
|
1021
1101
|
root, threshold, quiet, url, force = sys.argv[1], int(sys.argv[2]), sys.argv[3] == '1', sys.argv[4], sys.argv[5] == '1'
|
|
1022
1102
|
now = time.time()
|
|
@@ -1037,6 +1117,13 @@ CLIENT_ID = os.environ.get('CLAUDE_MULTIACC_CLIENT_ID',
|
|
|
1037
1117
|
REFRESH_MIN_EXPIRED = 300
|
|
1038
1118
|
REFRESH_FAIL_BACKOFF = 600 # transient (network/5xx/429): retry in 10 min
|
|
1039
1119
|
REFRESH_DENIED_BACKOFF = 21600 # 4xx = grant likely revoked: 6h; re-login needed anyway
|
|
1120
|
+
# A usage fetch the server says will never succeed (x-should-retry: false — e.g. the
|
|
1121
|
+
# 403 "does not meet scope requirement user:profile" that a setup token ALWAYS gets)
|
|
1122
|
+
# is not a hiccup to retry every pass. Retrying it is what manufactured the 429s that
|
|
1123
|
+
# then hid the real cause for days, so a definitive refusal parks for 6h and says
|
|
1124
|
+
# exactly which ceremony fixes it.
|
|
1125
|
+
USAGE_DENIED_BACKOFF = 21600
|
|
1126
|
+
USAGE_FAIL_BACKOFF = 900 # anything else non-2xx: 15 min, doubling to 30
|
|
1040
1127
|
|
|
1041
1128
|
def say(msg):
|
|
1042
1129
|
if not quiet:
|
|
@@ -1058,6 +1145,15 @@ def parse_iso(s):
|
|
|
1058
1145
|
# 4xx from the refresh endpoint), never for a transient network/5xx/429 hiccup.
|
|
1059
1146
|
def mark_expired(d, slug, detail=''):
|
|
1060
1147
|
mpath = os.path.join(d, '.expired')
|
|
1148
|
+
# An org-blocked marker is the strongest statement there is about an account and
|
|
1149
|
+
# only `verify` or a re-login may lift it (see clear_expired). Overwriting its
|
|
1150
|
+
# REASON with a weaker one is how it gets lifted by accident: the next successful
|
|
1151
|
+
# fetch sees a reason clear_expired is willing to drop, and the block vanishes.
|
|
1152
|
+
try:
|
|
1153
|
+
if 'reason=org-blocked' in open(mpath, errors='replace').read():
|
|
1154
|
+
return
|
|
1155
|
+
except OSError:
|
|
1156
|
+
pass
|
|
1061
1157
|
try:
|
|
1062
1158
|
with open(mpath + '.tmp', 'w') as f:
|
|
1063
1159
|
f.write(f'{int(now)}\n')
|
|
@@ -1089,6 +1185,39 @@ try:
|
|
|
1089
1185
|
except Exception as e:
|
|
1090
1186
|
sys.exit(f'cannot read manifest: {e}')
|
|
1091
1187
|
|
|
1188
|
+
def token_digest(path):
|
|
1189
|
+
"""Stable, non-secret identifier for a credential file's CONTENTS. Used to
|
|
1190
|
+
remember which setup token this endpoint refused, so a re-minted one still gets a
|
|
1191
|
+
try while the refused one is never spent again. Truncated: this only has to
|
|
1192
|
+
distinguish credentials, and a full hash of a secret is not something to write
|
|
1193
|
+
into a file that syncs between machines."""
|
|
1194
|
+
try:
|
|
1195
|
+
with open(path, 'rb') as f:
|
|
1196
|
+
return hashlib.sha256(f.read()).hexdigest()[:16]
|
|
1197
|
+
except OSError:
|
|
1198
|
+
return None
|
|
1199
|
+
|
|
1200
|
+
|
|
1201
|
+
def park_dead_grant(aid, d, slug, detail):
|
|
1202
|
+
"""Park an account whose OAuth grant is dead — UNLESS it also has a portable setup
|
|
1203
|
+
token, in which case the grant being dead proves nothing about the account: the
|
|
1204
|
+
token still authenticates every real call. Parking it would take a perfectly
|
|
1205
|
+
working account out of the pool over a credential the pool does not need for work,
|
|
1206
|
+
only for telemetry. (The shim's auth_dead() checks the .expired marker BEFORE
|
|
1207
|
+
server.token, so a marker written here really would remove it.)"""
|
|
1208
|
+
try:
|
|
1209
|
+
portable = os.path.getsize(os.path.join(d, 'server.token')) > 0
|
|
1210
|
+
except OSError:
|
|
1211
|
+
portable = False
|
|
1212
|
+
if portable:
|
|
1213
|
+
say(f'{aid}: OAuth grant is dead ({slug}) but its setup token still works — '
|
|
1214
|
+
f'staying in the pool. TELEMETRY is dead for it until it has an OAuth login '
|
|
1215
|
+
f'here: claude-accounts login {aid}')
|
|
1216
|
+
return
|
|
1217
|
+
mark_expired(d, slug, detail)
|
|
1218
|
+
say(f'{aid}: {detail} — re-login needed (claude-accounts relogin {aid})')
|
|
1219
|
+
|
|
1220
|
+
|
|
1092
1221
|
def refresh_oauth(aid, d, cpath):
|
|
1093
1222
|
"""Refresh a long-expired OAuth access token via the refresh-token grant and
|
|
1094
1223
|
persist the ROTATED credential atomically (0600). Returns the new bearer, or
|
|
@@ -1105,20 +1234,22 @@ def refresh_oauth(aid, d, cpath):
|
|
|
1105
1234
|
return None
|
|
1106
1235
|
except Exception:
|
|
1107
1236
|
return None
|
|
1108
|
-
|
|
1109
|
-
|
|
1237
|
+
# NB: no "must have an accessToken" gate. A credential whose access token was
|
|
1238
|
+
# cleared but whose REFRESH token is alive is exactly the shape a grant is supposed
|
|
1239
|
+
# to recover from; requiring the dead half to be present meant such an account could
|
|
1240
|
+
# never come back, and (with a setup token beside it) went dark for telemetry
|
|
1241
|
+
# forever. The expiresAt gate below is what protects a live session's credential.
|
|
1110
1242
|
if not o.get('refreshToken'):
|
|
1111
|
-
|
|
1112
|
-
|
|
1243
|
+
park_dead_grant(aid, d, 'no-refresh-token',
|
|
1244
|
+
'credential has no refresh token and its access token expired')
|
|
1113
1245
|
return None
|
|
1114
1246
|
if o.get('expiresAt', 0) / 1000.0 > now - REFRESH_MIN_EXPIRED:
|
|
1115
1247
|
return None # not expired long enough to prove no live session owns it
|
|
1116
1248
|
if o.get('refreshTokenExpiresAt', 0) / 1000.0 <= now:
|
|
1117
1249
|
# Nothing can revive this account: park it so the shim stops selecting it
|
|
1118
1250
|
# (every run under it would fail with "OAuth session expired").
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
say(f'{aid}: refresh token expired — re-login needed (claude-accounts relogin {aid})')
|
|
1251
|
+
park_dead_grant(aid, d, 'refresh-token-expired',
|
|
1252
|
+
'the refresh token itself expired; only a re-login can fix it')
|
|
1122
1253
|
return None
|
|
1123
1254
|
if not force:
|
|
1124
1255
|
try:
|
|
@@ -1138,6 +1269,7 @@ def refresh_oauth(aid, d, cpath):
|
|
|
1138
1269
|
pass
|
|
1139
1270
|
say(f'{aid}: oauth refresh failed ({why}); backing off {wait}s; limits left as-is')
|
|
1140
1271
|
|
|
1272
|
+
started_with = o['refreshToken']
|
|
1141
1273
|
body = json.dumps({'grant_type': 'refresh_token',
|
|
1142
1274
|
'refresh_token': o['refreshToken'],
|
|
1143
1275
|
'client_id': CLIENT_ID}).encode()
|
|
@@ -1166,11 +1298,11 @@ def refresh_oauth(aid, d, cpath):
|
|
|
1166
1298
|
except Exception:
|
|
1167
1299
|
pass
|
|
1168
1300
|
if 'invalid_grant' in body:
|
|
1169
|
-
|
|
1170
|
-
|
|
1301
|
+
park_dead_grant(aid, d, f'refresh-denied-http-{e.code}',
|
|
1302
|
+
'the refresh grant was refused as invalid_grant (revoked or rotated away)')
|
|
1171
1303
|
elif denials >= 3:
|
|
1172
|
-
|
|
1173
|
-
|
|
1304
|
+
park_dead_grant(aid, d, f'refresh-denied-http-{e.code}',
|
|
1305
|
+
f'the refresh grant was refused {denials} times in a row')
|
|
1174
1306
|
back_off(REFRESH_DENIED_BACKOFF,
|
|
1175
1307
|
f'HTTP {e.code} — refresh token may be revoked; re-login needed',
|
|
1176
1308
|
denials=denials)
|
|
@@ -1198,10 +1330,26 @@ def refresh_oauth(aid, d, cpath):
|
|
|
1198
1330
|
if data.get('refresh_token_expires_in'):
|
|
1199
1331
|
o['refreshTokenExpiresAt'] = int((now + float(data['refresh_token_expires_in'])) * 1000)
|
|
1200
1332
|
doc['claudeAiOauth'] = o
|
|
1333
|
+
# CHECK-AND-SET. The grant rotates, and a live claude session refreshes the same
|
|
1334
|
+
# file. REFRESH_MIN_EXPIRED makes that unlikely, not impossible — and losing the
|
|
1335
|
+
# race by overwriting means the session's newer credential is destroyed. If the
|
|
1336
|
+
# on-disk refresh token is no longer the one this grant was issued against, the
|
|
1337
|
+
# other writer won: keep its result, discard ours.
|
|
1338
|
+
try:
|
|
1339
|
+
with open(cpath) as f:
|
|
1340
|
+
disk = json.load(f).get('claudeAiOauth', {})
|
|
1341
|
+
if isinstance(disk, dict) and disk.get('refreshToken') != started_with:
|
|
1342
|
+
say(f'{aid}: credential was refreshed by something else mid-flight — '
|
|
1343
|
+
f'keeping the newer one on disk')
|
|
1344
|
+
return None
|
|
1345
|
+
except Exception:
|
|
1346
|
+
pass
|
|
1201
1347
|
try:
|
|
1202
1348
|
fd = os.open(cpath + '.tmp', os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
|
|
1203
1349
|
with os.fdopen(fd, 'w') as f:
|
|
1204
1350
|
json.dump(doc, f)
|
|
1351
|
+
f.flush()
|
|
1352
|
+
os.fsync(f.fileno())
|
|
1205
1353
|
os.replace(cpath + '.tmp', cpath)
|
|
1206
1354
|
except Exception as e:
|
|
1207
1355
|
say(f'{aid}: token refreshed but credentials NOT persisted ({e}) — re-login may be needed')
|
|
@@ -1233,18 +1381,43 @@ for acct in manifest.get('accounts', []):
|
|
|
1233
1381
|
prev = json.load(open(lpath))
|
|
1234
1382
|
except Exception:
|
|
1235
1383
|
prev = {}
|
|
1384
|
+
# Valid JSON is not the same as a usable document: `[]`, `null` and `"broken"` all
|
|
1385
|
+
# parse, and every prev.get() below would then raise OUTSIDE the try — killing the
|
|
1386
|
+
# loop and starving every account after this one of telemetry. One corrupt file
|
|
1387
|
+
# must cost exactly one account.
|
|
1388
|
+
if not isinstance(prev, dict):
|
|
1389
|
+
prev = {}
|
|
1390
|
+
|
|
1391
|
+
def num(key, default=0):
|
|
1392
|
+
"""A field of the wrong type is the same as an absent one. limits.json is
|
|
1393
|
+
hand-editable, syncs between machines, and is written by more than one
|
|
1394
|
+
version of this tool at once."""
|
|
1395
|
+
v = prev.get(key, default)
|
|
1396
|
+
return v if isinstance(v, (int, float)) and not isinstance(v, bool) else default
|
|
1397
|
+
|
|
1236
1398
|
if not force:
|
|
1237
|
-
age = now -
|
|
1399
|
+
age = now - num('fetched_at')
|
|
1238
1400
|
if age < MIN_FETCH_INTERVAL:
|
|
1239
1401
|
continue
|
|
1240
|
-
retry_at =
|
|
1402
|
+
retry_at = num('retry_after')
|
|
1241
1403
|
if retry_at > now:
|
|
1242
|
-
|
|
1404
|
+
# Name the ACTUAL error. Reporting every park as a 429 is what let an
|
|
1405
|
+
# unauthorized account read as merely rate-limited for eleven days.
|
|
1406
|
+
why = prev.get('last_error')
|
|
1407
|
+
why = why if isinstance(why, str) and why else 'a failed fetch'
|
|
1408
|
+
say(f'{aid}: backing off after {why} ({int(retry_at - now)}s left); limits left as-is')
|
|
1243
1409
|
continue
|
|
1244
1410
|
|
|
1411
|
+
# BEARER ORDER MATTERS, and it is not the obvious one. A setup token authenticates
|
|
1412
|
+
# inference forever but is minted WITHOUT the user:profile scope this endpoint
|
|
1413
|
+
# requires, so it can only ever produce a 403 here. Trying it before the OAuth
|
|
1414
|
+
# refresh grant — which is what this did — killed telemetry for accounts whose
|
|
1415
|
+
# refresh token was still perfectly good, purely because a server.token sat beside
|
|
1416
|
+
# it. OAuth first, refresh second, token only as a genuine last resort.
|
|
1245
1417
|
bearer = None
|
|
1246
1418
|
source = None
|
|
1247
1419
|
cpath = os.path.join(d, '.credentials.json')
|
|
1420
|
+
tpath = os.path.join(d, 'server.token')
|
|
1248
1421
|
if os.path.isfile(cpath):
|
|
1249
1422
|
try:
|
|
1250
1423
|
c = json.load(open(cpath)).get('claudeAiOauth', {})
|
|
@@ -1252,11 +1425,6 @@ for acct in manifest.get('accounts', []):
|
|
|
1252
1425
|
bearer, source = c['accessToken'], 'oauth'
|
|
1253
1426
|
except Exception:
|
|
1254
1427
|
pass
|
|
1255
|
-
tpath = os.path.join(d, 'server.token')
|
|
1256
|
-
if not bearer and os.path.isfile(tpath):
|
|
1257
|
-
t = open(tpath).read().strip()
|
|
1258
|
-
if t:
|
|
1259
|
-
bearer, source = t, 'token'
|
|
1260
1428
|
if not bearer and os.path.isfile(cpath):
|
|
1261
1429
|
# Hard fail-open guard: NOTHING a single account's refresh does may abort
|
|
1262
1430
|
# the loop — every account after it would silently starve of telemetry.
|
|
@@ -1267,6 +1435,25 @@ for acct in manifest.get('accounts', []):
|
|
|
1267
1435
|
tok = None
|
|
1268
1436
|
if tok:
|
|
1269
1437
|
bearer, source = tok, 'oauth'
|
|
1438
|
+
if not bearer and os.path.isfile(tpath):
|
|
1439
|
+
# Once this endpoint has refused THIS token file for lacking a scope, asking
|
|
1440
|
+
# again is guaranteed to fail and only spends the account's hourly budget —
|
|
1441
|
+
# which is how a permanent authorization problem disguised itself as a rate
|
|
1442
|
+
# limit. Keyed on the file's mtime, so re-minting the token retries it.
|
|
1443
|
+
# Keyed on a non-secret digest of the token itself, not its mtime: `sync`
|
|
1444
|
+
# pushes tokens with rsync -a (mtimes preserved) and two different tokens can
|
|
1445
|
+
# land on the same whole second, either of which would skip a credential that
|
|
1446
|
+
# was never actually refused.
|
|
1447
|
+
denied_digest = prev.get('token_scope_denied')
|
|
1448
|
+
tdigest = token_digest(tpath)
|
|
1449
|
+
if not force and tdigest and denied_digest == tdigest:
|
|
1450
|
+
say(f'{aid}: setup token cannot read usage (no user:profile scope) and there '
|
|
1451
|
+
f'is no OAuth login here — telemetry stays dark until: '
|
|
1452
|
+
f'claude-accounts login {aid}')
|
|
1453
|
+
continue
|
|
1454
|
+
t = open(tpath).read().strip()
|
|
1455
|
+
if t:
|
|
1456
|
+
bearer, source = t, 'token'
|
|
1270
1457
|
if not bearer:
|
|
1271
1458
|
# Fail OPEN: no usable bearer => leave existing state; never block work on telemetry.
|
|
1272
1459
|
say(f'{aid}: no fresh bearer (expired oauth and/or no token); limits left as-is')
|
|
@@ -1282,6 +1469,24 @@ for acct in manifest.get('accounts', []):
|
|
|
1282
1469
|
resp = urllib.request.urlopen(req, timeout=15)
|
|
1283
1470
|
data = json.loads(resp.read().decode())
|
|
1284
1471
|
except urllib.error.HTTPError as e:
|
|
1472
|
+
body = ''
|
|
1473
|
+
try:
|
|
1474
|
+
body = e.read().decode('utf-8', 'replace')[:400]
|
|
1475
|
+
except Exception:
|
|
1476
|
+
pass
|
|
1477
|
+
|
|
1478
|
+
def park(wait, note):
|
|
1479
|
+
"""Record a failed fetch WITHOUT inventing freshness: fetched_at is left
|
|
1480
|
+
exactly as it was, so a parked account still reads as stale everywhere."""
|
|
1481
|
+
prev['retry_after'] = int(now + wait)
|
|
1482
|
+
prev['backoff'] = wait
|
|
1483
|
+
prev['last_error'] = note
|
|
1484
|
+
prev['last_error_at'] = int(now)
|
|
1485
|
+
tmp = lpath + '.tmp'
|
|
1486
|
+
with open(tmp, 'w') as f:
|
|
1487
|
+
json.dump(prev, f, indent=1)
|
|
1488
|
+
os.replace(tmp, lpath)
|
|
1489
|
+
|
|
1285
1490
|
if e.code == 429:
|
|
1286
1491
|
# Respect Retry-After; otherwise exponential backoff capped at 30 min.
|
|
1287
1492
|
try:
|
|
@@ -1289,19 +1494,69 @@ for acct in manifest.get('accounts', []):
|
|
|
1289
1494
|
except (TypeError, ValueError):
|
|
1290
1495
|
wait = 0
|
|
1291
1496
|
if wait <= 0:
|
|
1292
|
-
wait = min(1800, max(120, int(
|
|
1497
|
+
wait = min(1800, max(120, int(num('backoff', 60)) * 2))
|
|
1498
|
+
park(wait, f'HTTP 429 (rate limited, source={source})')
|
|
1499
|
+
say(f'{aid}: rate limited (429); backing off {wait}s; failing open')
|
|
1500
|
+
else:
|
|
1501
|
+
# EVERY non-2xx now backs off. Before this only a 429 did, so a permanent
|
|
1502
|
+
# refusal was re-issued on every scheduled pass from every machine in the
|
|
1503
|
+
# fleet — and those retries are what earned the 429s that made an
|
|
1504
|
+
# unauthorized account look merely rate-limited. Telemetry then sat frozen
|
|
1505
|
+
# while the shim ranked the whole pool NEUTRAL, i.e. picked at random.
|
|
1506
|
+
denied = (str(e.headers.get('x-should-retry') or '').lower() == 'false'
|
|
1507
|
+
or 'scope requirement' in body)
|
|
1508
|
+
# A setup token can NEVER grow the user:profile scope, so that refusal is
|
|
1509
|
+
# permanent and parks for hours. Any other "do not retry" may well be a
|
|
1510
|
+
# policy or endpoint issue someone fixes in minutes — park it for an hour,
|
|
1511
|
+
# not six, so recovery does not wait on a human running --force.
|
|
1512
|
+
scope_denied = denied and 'scope requirement' in body
|
|
1513
|
+
if scope_denied:
|
|
1514
|
+
wait = USAGE_DENIED_BACKOFF
|
|
1515
|
+
elif denied:
|
|
1516
|
+
wait = USAGE_DENIED_BACKOFF // 6
|
|
1517
|
+
else:
|
|
1518
|
+
wait = min(1800, max(USAGE_FAIL_BACKOFF,
|
|
1519
|
+
int(num('backoff', USAGE_FAIL_BACKOFF // 2)) * 2))
|
|
1520
|
+
if scope_denied and source == 'token':
|
|
1521
|
+
# Remember WHICH token was refused, so this credential is never spent
|
|
1522
|
+
# on this endpoint again — but a freshly minted one still gets a try.
|
|
1523
|
+
tdigest = token_digest(tpath)
|
|
1524
|
+
if tdigest:
|
|
1525
|
+
prev['token_scope_denied'] = tdigest
|
|
1526
|
+
park(wait, f'HTTP {e.code} (source={source})'
|
|
1527
|
+
+ (' — permanent, server said do not retry' if denied else ''))
|
|
1528
|
+
if scope_denied and source == 'token':
|
|
1529
|
+
# The exact shape of this outage: a setup token is minted WITHOUT the
|
|
1530
|
+
# user:profile scope the usage endpoint requires, so an account whose
|
|
1531
|
+
# OAuth grant lapsed keeps working for inference and goes permanently
|
|
1532
|
+
# dark for telemetry. Only a sign-in ON THIS MACHINE restores it.
|
|
1533
|
+
say(f'{aid}: usage endpoint refuses the setup token (HTTP {e.code} — a '
|
|
1534
|
+
f'setup token has no user:profile scope). Telemetry is DEAD for this '
|
|
1535
|
+
f'account until it has an OAuth login here: claude-accounts login {aid}. '
|
|
1536
|
+
f'Backing off {wait}s.')
|
|
1537
|
+
elif denied:
|
|
1538
|
+
say(f'{aid}: usage fetch refused for good (HTTP {e.code}, source={source}); '
|
|
1539
|
+
f'backing off {wait}s — re-login needed: claude-accounts login {aid}')
|
|
1540
|
+
else:
|
|
1541
|
+
say(f'{aid}: usage fetch failed (HTTP {e.code}); backing off {wait}s; failing open')
|
|
1542
|
+
continue
|
|
1543
|
+
except Exception as e:
|
|
1544
|
+
# Network/parse trouble is transient by nature, but it still must not be retried
|
|
1545
|
+
# every 5 minutes forever — that is how a fleet talks itself into a 429.
|
|
1546
|
+
wait = min(1800, max(USAGE_FAIL_BACKOFF,
|
|
1547
|
+
int(num('backoff', USAGE_FAIL_BACKOFF // 2)) * 2))
|
|
1548
|
+
try:
|
|
1293
1549
|
prev['retry_after'] = int(now + wait)
|
|
1294
1550
|
prev['backoff'] = wait
|
|
1551
|
+
prev['last_error'] = str(e)[:200]
|
|
1552
|
+
prev['last_error_at'] = int(now)
|
|
1295
1553
|
tmp = lpath + '.tmp'
|
|
1296
1554
|
with open(tmp, 'w') as f:
|
|
1297
1555
|
json.dump(prev, f, indent=1)
|
|
1298
1556
|
os.replace(tmp, lpath)
|
|
1299
|
-
|
|
1300
|
-
|
|
1301
|
-
|
|
1302
|
-
continue
|
|
1303
|
-
except Exception as e:
|
|
1304
|
-
say(f'{aid}: usage fetch failed ({e}); failing open')
|
|
1557
|
+
except Exception:
|
|
1558
|
+
pass
|
|
1559
|
+
say(f'{aid}: usage fetch failed ({e}); backing off {wait}s; failing open')
|
|
1305
1560
|
continue
|
|
1306
1561
|
|
|
1307
1562
|
def pct_of(v):
|
|
@@ -1387,8 +1642,21 @@ for acct in manifest.get('accounts', []):
|
|
|
1387
1642
|
session = [b['percent'] for b in buckets if b['group'] == 'session']
|
|
1388
1643
|
weeklyp = max(weekly) if weekly else maxp
|
|
1389
1644
|
sessionp = max(session) if session else 0
|
|
1645
|
+
# How long weekly_percent keeps meaning something. A weekly bucket only ever RISES
|
|
1646
|
+
# until its reset, so before that moment a stale percent is still a valid lower
|
|
1647
|
+
# bound and the shim can rank on it when nothing fresher exists; after it, the
|
|
1648
|
+
# number describes a week that is over and says nothing at all. Recording the
|
|
1649
|
+
# horizon here keeps the shim from having to parse buckets[] on every invocation.
|
|
1650
|
+
# It must come from the bucket weekly_percent actually CAME FROM: a low monthly
|
|
1651
|
+
# bucket resetting in an hour says nothing about an 80% weekly one that resets in
|
|
1652
|
+
# five days, and taking the minimum over all of them would throw the 80% away.
|
|
1653
|
+
wresets = [int(b['resets_epoch']) for b in buckets
|
|
1654
|
+
if b['group'] != 'session' and b['percent'] == weeklyp
|
|
1655
|
+
and isinstance(b.get('resets_epoch'), int)]
|
|
1390
1656
|
out = {'fetched_at': int(now), 'source': source, 'max_percent': maxp,
|
|
1391
|
-
'weekly_percent': weeklyp, 'session_percent': sessionp,
|
|
1657
|
+
'weekly_percent': weeklyp, 'session_percent': sessionp,
|
|
1658
|
+
'weekly_resets_epoch': min(wresets) if wresets else 0,
|
|
1659
|
+
'buckets': buckets}
|
|
1392
1660
|
tmp = lpath + '.tmp'
|
|
1393
1661
|
with open(tmp, 'w') as f:
|
|
1394
1662
|
json.dump(out, f, indent=1)
|
|
@@ -1483,6 +1751,14 @@ def mark_expired(d, slug, detail=''):
|
|
|
1483
1751
|
"""Park an account the shim must stop selecting. Verify is the strongest signal
|
|
1484
1752
|
there is — a real inference call that came back 'not authenticated'."""
|
|
1485
1753
|
mpath = os.path.join(d, '.expired')
|
|
1754
|
+
# Same rule as the limits path: an org block outranks every other reason and only
|
|
1755
|
+
# a PASSING verify or a re-login may lift it. Rewriting its reason with a weaker
|
|
1756
|
+
# one is how it gets lifted by accident later.
|
|
1757
|
+
try:
|
|
1758
|
+
if 'reason=org-blocked' in open(mpath, errors='replace').read():
|
|
1759
|
+
return
|
|
1760
|
+
except OSError:
|
|
1761
|
+
pass
|
|
1486
1762
|
try:
|
|
1487
1763
|
with open(mpath + '.tmp', 'w') as f:
|
|
1488
1764
|
f.write(f'{int(time.time())}\n')
|
package/install.sh
CHANGED
|
@@ -71,6 +71,27 @@ case "$ACC_ROOT$CODEX_ACC_ROOT" in
|
|
|
71
71
|
"*) echo "pool root contains a character that cannot be scheduled safely: $ACC_ROOT / $CODEX_ACC_ROOT" >&2; exit 1 ;;
|
|
72
72
|
esac
|
|
73
73
|
|
|
74
|
+
# A pool root under a temp directory is by definition gone tomorrow, but the launchd
|
|
75
|
+
# agent or crontab line naming it is not: it keeps firing forever against a path that
|
|
76
|
+
# no longer exists, one leaked set per run. That is how a Mac collected 57 orphaned
|
|
77
|
+
# agent sets and a server 4 stray cron blocks. EITHER root disqualifies the whole set,
|
|
78
|
+
# because one install writes agents for both providers: the leaked sets above named a
|
|
79
|
+
# temp CLAUDE root and the REAL codex pool, so they went right on polling a live usage
|
|
80
|
+
# endpoint every five minutes. Install the binaries for such a root; never schedule it.
|
|
81
|
+
ephemeral_root() { # $1 = a pool root
|
|
82
|
+
case "$1" in
|
|
83
|
+
/tmp/*|/private/tmp/*|/var/tmp/*|/private/var/tmp/*|/var/folders/*|/private/var/folders/*)
|
|
84
|
+
return 0 ;;
|
|
85
|
+
*) return 1 ;;
|
|
86
|
+
esac
|
|
87
|
+
}
|
|
88
|
+
if [ "$NO_SCHEDULE" != "1" ] && { ephemeral_root "$ACC_ROOT" || ephemeral_root "$CODEX_ACC_ROOT"; }; then
|
|
89
|
+
NO_SCHEDULE=1
|
|
90
|
+
echo " note: a pool root is under a temp directory (claude: $ACC_ROOT, codex: $CODEX_ACC_ROOT)" >&2
|
|
91
|
+
echo " — installing WITHOUT schedulers, which would outlive it." >&2
|
|
92
|
+
echo " Pass --no-schedule to silence this." >&2
|
|
93
|
+
fi
|
|
94
|
+
|
|
74
95
|
LABEL="com.claude-multiacc${INSTANCE:+.$INSTANCE}"
|
|
75
96
|
MARK_BEGIN="# >>> claude-multiacc >>>"
|
|
76
97
|
MARK_END="# <<< claude-multiacc <<<"
|
|
@@ -150,7 +171,7 @@ mac_schedule_install() {
|
|
|
150
171
|
<string>limits</string>
|
|
151
172
|
<string>--quiet</string>
|
|
152
173
|
</array>
|
|
153
|
-
$(plist_env_block) <key>StartInterval</key><integer>
|
|
174
|
+
$(plist_env_block) <key>StartInterval</key><integer>900</integer>
|
|
154
175
|
<key>RunAtLoad</key><true/>
|
|
155
176
|
<key>StandardOutPath</key><string>/dev/null</string>
|
|
156
177
|
<key>StandardErrorPath</key><string>/dev/null</string>
|
|
@@ -239,7 +260,7 @@ EOF
|
|
|
239
260
|
launchctl load -w "$PLIST_CODEX_LIMITS" 2>/dev/null || true
|
|
240
261
|
launchctl load -w "$PLIST_CODEX_HEALTH" 2>/dev/null || true
|
|
241
262
|
[ "${CLAUDE_MULTIACC_AUTOUPDATE:-1}" = "1" ] && launchctl load -w "$PLIST_UPDATE" 2>/dev/null || true
|
|
242
|
-
echo " launchd: limits refresh
|
|
263
|
+
echo " launchd: limits refresh (claude 15m / codex 5m) + weekly health checks + daily auto-update"
|
|
243
264
|
}
|
|
244
265
|
|
|
245
266
|
mac_schedule_remove() {
|
|
@@ -262,7 +283,7 @@ linux_schedule_install() {
|
|
|
262
283
|
env="$(cron_env_prefix)"
|
|
263
284
|
{ crontab -l 2>/dev/null | cron_strip_ours; } > "$tmp"
|
|
264
285
|
{
|
|
265
|
-
echo "*/
|
|
286
|
+
echo "*/15 * * * * $env$REPO_DIR/bin/claude-accounts limits --quiet >/dev/null 2>&1 $CRON_TAG"
|
|
266
287
|
echo "*/5 * * * * $env$REPO_DIR/bin/codex-accounts limits --quiet >/dev/null 2>&1 $CRON_TAG"
|
|
267
288
|
echo "17 9 * * 1 $env$REPO_DIR/bin/claude-accounts health >/dev/null 2>&1 $CRON_TAG"
|
|
268
289
|
echo "37 9 * * 1 $env$REPO_DIR/bin/codex-accounts health >/dev/null 2>&1 $CRON_TAG"
|
|
@@ -271,7 +292,7 @@ linux_schedule_install() {
|
|
|
271
292
|
} >> "$tmp"
|
|
272
293
|
crontab "$tmp"
|
|
273
294
|
rm -f "$tmp"
|
|
274
|
-
echo " cron: limits refresh
|
|
295
|
+
echo " cron: limits refresh (claude 15m / codex 5m) + weekly health checks + daily auto-update"
|
|
275
296
|
}
|
|
276
297
|
|
|
277
298
|
linux_schedule_remove() {
|