claude-multiacc 1.0.14 → 1.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -267,6 +267,25 @@ sys.path = [sys.argv[3]] + [p for p in sys.path if p not in ('', '.')]
267
267
  from audit import audit_account # noqa: E402 (shared with the shim's rule)
268
268
  machine = sys.argv[4]
269
269
  now = time.time()
270
+ # Must match the shim's window (bin/claude), or status would call data "fresh" that
271
+ # selection has already stopped ranking on. Same fallback as the shim for a garbled
272
+ # override: a bad env var must never be the thing that stops status from printing.
273
+ try:
274
+ STALE_AFTER = int(os.environ.get('CLAUDE_MULTIACC_STALE_AFTER') or 3600)
275
+ except ValueError:
276
+ STALE_AFTER = 3600
277
+ if STALE_AFTER <= 0:
278
+ STALE_AFTER = 3600
279
+
280
+ def telem_fetched_at(aid):
281
+ """Epoch of aid's last SUCCESSFUL usage fetch, or 0. Never raises: this file is
282
+ hand-editable and syncs between machines, and one corrupt copy must not stop
283
+ status from printing the pool-wide verdict it exists to show."""
284
+ try:
285
+ v = json.load(open(os.path.join(root, aid, 'limits.json'))).get('fetched_at', 0)
286
+ except Exception:
287
+ return 0
288
+ return v if isinstance(v, (int, float)) and not isinstance(v, bool) and v > 0 else 0
270
289
 
271
290
  def last_pick(aid):
272
291
  path = os.path.join(root, 'selection.log')
@@ -288,12 +307,15 @@ print(f"server : {doc.get('server','-')} (root: {doc.get('server_root','-')}
288
307
  print(f"threshold : {doc.get('threshold', 90)}% (any bucket at/above => account excluded)")
289
308
  print()
290
309
  needs_login = []
310
+ selectable_ids = []
291
311
  for a in doc.get('accounts', []):
292
312
  aid = a['id']
293
313
  d = os.path.join(root, aid)
294
314
  st = audit_account(root, a, machine=machine)
295
315
  if st['state'] in ('expired', 'blocked', 'missing'):
296
316
  needs_login.append(aid)
317
+ else:
318
+ selectable_ids.append(aid)
297
319
  banner = {'ok': '', 'remote': ' [not logged in here — grant lives elsewhere]',
298
320
  'missing': ' ** NO LOGIN — claude-accounts relogin %s **' % aid,
299
321
  'expired': ' ** LOGIN EXPIRED — claude-accounts relogin %s **' % aid,
@@ -325,9 +347,33 @@ for a in doc.get('accounts', []):
325
347
  if os.path.isfile(lpath):
326
348
  try:
327
349
  lim = json.load(open(lpath))
328
- age = int(now - lim.get('fetched_at', 0))
350
+ if not isinstance(lim, dict):
351
+ raise ValueError('not an object')
352
+ fetched = telem_fetched_at(aid)
353
+ age = int(now - fetched) if fetched else None
329
354
  parts = [f"{b['name']}={b['percent']}%" for b in lim.get('buckets', [])]
330
- print(f" limits : {' '.join(parts) or '(none)'} [{age}s old, max {lim.get('max_percent')}%]")
355
+ if age is None:
356
+ shown = 'never fetched'
357
+ elif age >= 86400:
358
+ shown = f'{age // 86400}d {age % 86400 // 3600}h old'
359
+ elif age >= 3600:
360
+ shown = f'{age // 3600}h {age % 3600 // 60}m old'
361
+ else:
362
+ shown = f'{age}s old'
363
+ # Loud, because stale numbers do not look stale: they look like a healthy
364
+ # account sitting at 2%. Past the ranking window the shim ignores them
365
+ # entirely and picks at random, so the reading below is decoration.
366
+ stale = age is None or age > STALE_AFTER
367
+ flag = ' << STALE — NOT USED FOR RANKING' if stale else ''
368
+ print(f" limits : {' '.join(parts) or '(none)'} [{shown}, max {lim.get('max_percent')}%]{flag}")
369
+ err = lim.get('last_error')
370
+ if isinstance(err, str) and err:
371
+ when = lim.get('last_error_at')
372
+ ago = f' ({int(now - when)}s ago)' if isinstance(when, (int, float)) else ''
373
+ print(f" telemetry : last fetch FAILED{ago}: {err}")
374
+ retry = lim.get('retry_after', 0)
375
+ if isinstance(retry, (int, float)) and retry > now:
376
+ print(f" retrying in {int(retry - now)}s")
331
377
  except Exception as e:
332
378
  print(f" limits : unreadable ({e})")
333
379
  else:
@@ -349,6 +395,40 @@ for a in doc.get('accounts', []):
349
395
  if needs_login:
350
396
  print(f"{len(needs_login)} account(s) are EXCLUDED from selection: {', '.join(needs_login)}")
351
397
  print("What each one needs: claude-accounts expired")
398
+ # Pool-wide verdict. Per-account ages are easy to skim past; "the pool is picking at
399
+ # random" is not. This is the line that would have caught an eleven-day outage on the
400
+ # first day someone ran status.
401
+ # The verdict comes from lib/report.py so this text, `--json`, and the shim's own
402
+ # ranking can never drift apart — a status that disagrees with selection is worse than
403
+ # no status at all.
404
+ try:
405
+ from report import telemetry_state, build # noqa: E402
406
+ doc_rows = build(root, 'claude', machine, 'status')['accounts']
407
+ verdict = telemetry_state(doc_rows, now)
408
+ except Exception as e:
409
+ verdict = 'unknown'
410
+ verdict_err = str(e)[:200]
411
+ if verdict == 'unknown':
412
+ # Silence here would recreate the observability half of the incident: a blind pool
413
+ # that says nothing. Say the verdict could not be computed, and why.
414
+ print(f"RANKING STATE UNKNOWN: could not compute the pool-wide telemetry verdict "
415
+ f"({verdict_err}). Check the per-account ages above by hand.")
416
+ elif verdict == 'blind':
417
+ print("RANKING IS BLIND: no account has usage telemetry inside the "
418
+ f"{STALE_AFTER}s window, and the last readings are too old to mean anything, "
419
+ "so every account scores the same and `claude` picks at RANDOM — including "
420
+ "accounts that are nearly out of weekly headroom.")
421
+ print(" why : see the 'telemetry' lines above (a setup token cannot read the usage "
422
+ "endpoint — it has no user:profile scope; only an OAuth login on this machine can)")
423
+ print(" fix : claude-accounts limits --force # then, if it still fails:")
424
+ print(" claude-accounts login <acct-NN> # per account, on THIS machine")
425
+ elif verdict == 'degraded':
426
+ print("RANKING IS DEGRADED: no account has telemetry inside the "
427
+ f"{STALE_AFTER}s window, so `claude` is ranking on the last readings whose "
428
+ "weekly bucket has not reset yet. Better than random, but it cannot see usage "
429
+ "since those readings were taken.")
430
+ print(" fix : claude-accounts limits --force # then, if it still fails:")
431
+ print(" claude-accounts login <acct-NN> # per account, on THIS machine")
352
432
  PYEOF
353
433
  }
354
434
 
@@ -1016,7 +1096,7 @@ cmd_limits() {
1016
1096
  [ "$threshold" -gt 90 ] && threshold=90
1017
1097
  [ "$threshold" -lt 1 ] && threshold=90
1018
1098
  "$PYBIN" - "$ACC_ROOT" "$threshold" "$quiet" "$USAGE_URL" "$force" <<'PYEOF' 2>>"$ACC_ROOT/limits.log"
1019
- import json, os, sys, time, urllib.request
1099
+ import hashlib, json, os, sys, time, urllib.request
1020
1100
 
1021
1101
  root, threshold, quiet, url, force = sys.argv[1], int(sys.argv[2]), sys.argv[3] == '1', sys.argv[4], sys.argv[5] == '1'
1022
1102
  now = time.time()
@@ -1037,6 +1117,13 @@ CLIENT_ID = os.environ.get('CLAUDE_MULTIACC_CLIENT_ID',
1037
1117
  REFRESH_MIN_EXPIRED = 300
1038
1118
  REFRESH_FAIL_BACKOFF = 600 # transient (network/5xx/429): retry in 10 min
1039
1119
  REFRESH_DENIED_BACKOFF = 21600 # 4xx = grant likely revoked: 6h; re-login needed anyway
1120
+ # A usage fetch the server says will never succeed (x-should-retry: false — e.g. the
1121
+ # 403 "does not meet scope requirement user:profile" that a setup token ALWAYS gets)
1122
+ # is not a hiccup to retry every pass. Retrying it is what manufactured the 429s that
1123
+ # then hid the real cause for days, so a definitive refusal parks for 6h and says
1124
+ # exactly which ceremony fixes it.
1125
+ USAGE_DENIED_BACKOFF = 21600
1126
+ USAGE_FAIL_BACKOFF = 900 # anything else non-2xx: 15 min, doubling to 30
1040
1127
 
1041
1128
  def say(msg):
1042
1129
  if not quiet:
@@ -1058,6 +1145,15 @@ def parse_iso(s):
1058
1145
  # 4xx from the refresh endpoint), never for a transient network/5xx/429 hiccup.
1059
1146
  def mark_expired(d, slug, detail=''):
1060
1147
  mpath = os.path.join(d, '.expired')
1148
+ # An org-blocked marker is the strongest statement there is about an account and
1149
+ # only `verify` or a re-login may lift it (see clear_expired). Overwriting its
1150
+ # REASON with a weaker one is how it gets lifted by accident: the next successful
1151
+ # fetch sees a reason clear_expired is willing to drop, and the block vanishes.
1152
+ try:
1153
+ if 'reason=org-blocked' in open(mpath, errors='replace').read():
1154
+ return
1155
+ except OSError:
1156
+ pass
1061
1157
  try:
1062
1158
  with open(mpath + '.tmp', 'w') as f:
1063
1159
  f.write(f'{int(now)}\n')
@@ -1089,6 +1185,39 @@ try:
1089
1185
  except Exception as e:
1090
1186
  sys.exit(f'cannot read manifest: {e}')
1091
1187
 
1188
+ def token_digest(path):
1189
+ """Stable, non-secret identifier for a credential file's CONTENTS. Used to
1190
+ remember which setup token this endpoint refused, so a re-minted one still gets a
1191
+ try while the refused one is never spent again. Truncated: this only has to
1192
+ distinguish credentials, and a full hash of a secret is not something to write
1193
+ into a file that syncs between machines."""
1194
+ try:
1195
+ with open(path, 'rb') as f:
1196
+ return hashlib.sha256(f.read()).hexdigest()[:16]
1197
+ except OSError:
1198
+ return None
1199
+
1200
+
1201
+ def park_dead_grant(aid, d, slug, detail):
1202
+ """Park an account whose OAuth grant is dead — UNLESS it also has a portable setup
1203
+ token, in which case the grant being dead proves nothing about the account: the
1204
+ token still authenticates every real call. Parking it would take a perfectly
1205
+ working account out of the pool over a credential the pool does not need for work,
1206
+ only for telemetry. (The shim's auth_dead() checks the .expired marker BEFORE
1207
+ server.token, so a marker written here really would remove it.)"""
1208
+ try:
1209
+ portable = os.path.getsize(os.path.join(d, 'server.token')) > 0
1210
+ except OSError:
1211
+ portable = False
1212
+ if portable:
1213
+ say(f'{aid}: OAuth grant is dead ({slug}) but its setup token still works — '
1214
+ f'staying in the pool. TELEMETRY is dead for it until it has an OAuth login '
1215
+ f'here: claude-accounts login {aid}')
1216
+ return
1217
+ mark_expired(d, slug, detail)
1218
+ say(f'{aid}: {detail} — re-login needed (claude-accounts relogin {aid})')
1219
+
1220
+
1092
1221
  def refresh_oauth(aid, d, cpath):
1093
1222
  """Refresh a long-expired OAuth access token via the refresh-token grant and
1094
1223
  persist the ROTATED credential atomically (0600). Returns the new bearer, or
@@ -1105,20 +1234,22 @@ def refresh_oauth(aid, d, cpath):
1105
1234
  return None
1106
1235
  except Exception:
1107
1236
  return None
1108
- if not o.get('accessToken'):
1109
- return None
1237
+ # NB: no "must have an accessToken" gate. A credential whose access token was
1238
+ # cleared but whose REFRESH token is alive is exactly the shape a grant is supposed
1239
+ # to recover from; requiring the dead half to be present meant such an account could
1240
+ # never come back, and (with a setup token beside it) went dark for telemetry
1241
+ # forever. The expiresAt gate below is what protects a live session's credential.
1110
1242
  if not o.get('refreshToken'):
1111
- mark_expired(d, 'no-refresh-token',
1112
- 'credential has no refresh token and its access token expired')
1243
+ park_dead_grant(aid, d, 'no-refresh-token',
1244
+ 'credential has no refresh token and its access token expired')
1113
1245
  return None
1114
1246
  if o.get('expiresAt', 0) / 1000.0 > now - REFRESH_MIN_EXPIRED:
1115
1247
  return None # not expired long enough to prove no live session owns it
1116
1248
  if o.get('refreshTokenExpiresAt', 0) / 1000.0 <= now:
1117
1249
  # Nothing can revive this account: park it so the shim stops selecting it
1118
1250
  # (every run under it would fail with "OAuth session expired").
1119
- mark_expired(d, 'refresh-token-expired',
1120
- 'the refresh token itself expired; only a re-login can fix it')
1121
- say(f'{aid}: refresh token expired — re-login needed (claude-accounts relogin {aid})')
1251
+ park_dead_grant(aid, d, 'refresh-token-expired',
1252
+ 'the refresh token itself expired; only a re-login can fix it')
1122
1253
  return None
1123
1254
  if not force:
1124
1255
  try:
@@ -1138,6 +1269,7 @@ def refresh_oauth(aid, d, cpath):
1138
1269
  pass
1139
1270
  say(f'{aid}: oauth refresh failed ({why}); backing off {wait}s; limits left as-is')
1140
1271
 
1272
+ started_with = o['refreshToken']
1141
1273
  body = json.dumps({'grant_type': 'refresh_token',
1142
1274
  'refresh_token': o['refreshToken'],
1143
1275
  'client_id': CLIENT_ID}).encode()
@@ -1166,11 +1298,11 @@ def refresh_oauth(aid, d, cpath):
1166
1298
  except Exception:
1167
1299
  pass
1168
1300
  if 'invalid_grant' in body:
1169
- mark_expired(d, f'refresh-denied-http-{e.code}',
1170
- 'the refresh grant was refused as invalid_grant (revoked or rotated away)')
1301
+ park_dead_grant(aid, d, f'refresh-denied-http-{e.code}',
1302
+ 'the refresh grant was refused as invalid_grant (revoked or rotated away)')
1171
1303
  elif denials >= 3:
1172
- mark_expired(d, f'refresh-denied-http-{e.code}',
1173
- f'the refresh grant was refused {denials} times in a row')
1304
+ park_dead_grant(aid, d, f'refresh-denied-http-{e.code}',
1305
+ f'the refresh grant was refused {denials} times in a row')
1174
1306
  back_off(REFRESH_DENIED_BACKOFF,
1175
1307
  f'HTTP {e.code} — refresh token may be revoked; re-login needed',
1176
1308
  denials=denials)
@@ -1198,10 +1330,26 @@ def refresh_oauth(aid, d, cpath):
1198
1330
  if data.get('refresh_token_expires_in'):
1199
1331
  o['refreshTokenExpiresAt'] = int((now + float(data['refresh_token_expires_in'])) * 1000)
1200
1332
  doc['claudeAiOauth'] = o
1333
+ # CHECK-AND-SET. The grant rotates, and a live claude session refreshes the same
1334
+ # file. REFRESH_MIN_EXPIRED makes that unlikely, not impossible — and losing the
1335
+ # race by overwriting means the session's newer credential is destroyed. If the
1336
+ # on-disk refresh token is no longer the one this grant was issued against, the
1337
+ # other writer won: keep its result, discard ours.
1338
+ try:
1339
+ with open(cpath) as f:
1340
+ disk = json.load(f).get('claudeAiOauth', {})
1341
+ if isinstance(disk, dict) and disk.get('refreshToken') != started_with:
1342
+ say(f'{aid}: credential was refreshed by something else mid-flight — '
1343
+ f'keeping the newer one on disk')
1344
+ return None
1345
+ except Exception:
1346
+ pass
1201
1347
  try:
1202
1348
  fd = os.open(cpath + '.tmp', os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
1203
1349
  with os.fdopen(fd, 'w') as f:
1204
1350
  json.dump(doc, f)
1351
+ f.flush()
1352
+ os.fsync(f.fileno())
1205
1353
  os.replace(cpath + '.tmp', cpath)
1206
1354
  except Exception as e:
1207
1355
  say(f'{aid}: token refreshed but credentials NOT persisted ({e}) — re-login may be needed')
@@ -1233,18 +1381,43 @@ for acct in manifest.get('accounts', []):
1233
1381
  prev = json.load(open(lpath))
1234
1382
  except Exception:
1235
1383
  prev = {}
1384
+ # Valid JSON is not the same as a usable document: `[]`, `null` and `"broken"` all
1385
+ # parse, and every prev.get() below would then raise OUTSIDE the try — killing the
1386
+ # loop and starving every account after this one of telemetry. One corrupt file
1387
+ # must cost exactly one account.
1388
+ if not isinstance(prev, dict):
1389
+ prev = {}
1390
+
1391
+ def num(key, default=0):
1392
+ """A field of the wrong type is the same as an absent one. limits.json is
1393
+ hand-editable, syncs between machines, and is written by more than one
1394
+ version of this tool at once."""
1395
+ v = prev.get(key, default)
1396
+ return v if isinstance(v, (int, float)) and not isinstance(v, bool) else default
1397
+
1236
1398
  if not force:
1237
- age = now - prev.get('fetched_at', 0)
1399
+ age = now - num('fetched_at')
1238
1400
  if age < MIN_FETCH_INTERVAL:
1239
1401
  continue
1240
- retry_at = prev.get('retry_after', 0)
1402
+ retry_at = num('retry_after')
1241
1403
  if retry_at > now:
1242
- say(f'{aid}: backing off after 429 ({int(retry_at - now)}s left); limits left as-is')
1404
+ # Name the ACTUAL error. Reporting every park as a 429 is what let an
1405
+ # unauthorized account read as merely rate-limited for eleven days.
1406
+ why = prev.get('last_error')
1407
+ why = why if isinstance(why, str) and why else 'a failed fetch'
1408
+ say(f'{aid}: backing off after {why} ({int(retry_at - now)}s left); limits left as-is')
1243
1409
  continue
1244
1410
 
1411
+ # BEARER ORDER MATTERS, and it is not the obvious one. A setup token authenticates
1412
+ # inference forever but is minted WITHOUT the user:profile scope this endpoint
1413
+ # requires, so it can only ever produce a 403 here. Trying it before the OAuth
1414
+ # refresh grant — which is what this did — killed telemetry for accounts whose
1415
+ # refresh token was still perfectly good, purely because a server.token sat beside
1416
+ # it. OAuth first, refresh second, token only as a genuine last resort.
1245
1417
  bearer = None
1246
1418
  source = None
1247
1419
  cpath = os.path.join(d, '.credentials.json')
1420
+ tpath = os.path.join(d, 'server.token')
1248
1421
  if os.path.isfile(cpath):
1249
1422
  try:
1250
1423
  c = json.load(open(cpath)).get('claudeAiOauth', {})
@@ -1252,11 +1425,6 @@ for acct in manifest.get('accounts', []):
1252
1425
  bearer, source = c['accessToken'], 'oauth'
1253
1426
  except Exception:
1254
1427
  pass
1255
- tpath = os.path.join(d, 'server.token')
1256
- if not bearer and os.path.isfile(tpath):
1257
- t = open(tpath).read().strip()
1258
- if t:
1259
- bearer, source = t, 'token'
1260
1428
  if not bearer and os.path.isfile(cpath):
1261
1429
  # Hard fail-open guard: NOTHING a single account's refresh does may abort
1262
1430
  # the loop — every account after it would silently starve of telemetry.
@@ -1267,6 +1435,25 @@ for acct in manifest.get('accounts', []):
1267
1435
  tok = None
1268
1436
  if tok:
1269
1437
  bearer, source = tok, 'oauth'
1438
+ if not bearer and os.path.isfile(tpath):
1439
+ # Once this endpoint has refused THIS token file for lacking a scope, asking
1440
+ # again is guaranteed to fail and only spends the account's hourly budget —
1441
+ # which is how a permanent authorization problem disguised itself as a rate
1442
+ # limit. Keyed on the file's mtime, so re-minting the token retries it.
1443
+ # Keyed on a non-secret digest of the token itself, not its mtime: `sync`
1444
+ # pushes tokens with rsync -a (mtimes preserved) and two different tokens can
1445
+ # land on the same whole second, either of which would skip a credential that
1446
+ # was never actually refused.
1447
+ denied_digest = prev.get('token_scope_denied')
1448
+ tdigest = token_digest(tpath)
1449
+ if not force and tdigest and denied_digest == tdigest:
1450
+ say(f'{aid}: setup token cannot read usage (no user:profile scope) and there '
1451
+ f'is no OAuth login here — telemetry stays dark until: '
1452
+ f'claude-accounts login {aid}')
1453
+ continue
1454
+ t = open(tpath).read().strip()
1455
+ if t:
1456
+ bearer, source = t, 'token'
1270
1457
  if not bearer:
1271
1458
  # Fail OPEN: no usable bearer => leave existing state; never block work on telemetry.
1272
1459
  say(f'{aid}: no fresh bearer (expired oauth and/or no token); limits left as-is')
@@ -1282,6 +1469,24 @@ for acct in manifest.get('accounts', []):
1282
1469
  resp = urllib.request.urlopen(req, timeout=15)
1283
1470
  data = json.loads(resp.read().decode())
1284
1471
  except urllib.error.HTTPError as e:
1472
+ body = ''
1473
+ try:
1474
+ body = e.read().decode('utf-8', 'replace')[:400]
1475
+ except Exception:
1476
+ pass
1477
+
1478
+ def park(wait, note):
1479
+ """Record a failed fetch WITHOUT inventing freshness: fetched_at is left
1480
+ exactly as it was, so a parked account still reads as stale everywhere."""
1481
+ prev['retry_after'] = int(now + wait)
1482
+ prev['backoff'] = wait
1483
+ prev['last_error'] = note
1484
+ prev['last_error_at'] = int(now)
1485
+ tmp = lpath + '.tmp'
1486
+ with open(tmp, 'w') as f:
1487
+ json.dump(prev, f, indent=1)
1488
+ os.replace(tmp, lpath)
1489
+
1285
1490
  if e.code == 429:
1286
1491
  # Respect Retry-After; otherwise exponential backoff capped at 30 min.
1287
1492
  try:
@@ -1289,19 +1494,69 @@ for acct in manifest.get('accounts', []):
1289
1494
  except (TypeError, ValueError):
1290
1495
  wait = 0
1291
1496
  if wait <= 0:
1292
- wait = min(1800, max(120, int(prev.get('backoff', 60)) * 2))
1497
+ wait = min(1800, max(120, int(num('backoff', 60)) * 2))
1498
+ park(wait, f'HTTP 429 (rate limited, source={source})')
1499
+ say(f'{aid}: rate limited (429); backing off {wait}s; failing open')
1500
+ else:
1501
+ # EVERY non-2xx now backs off. Before this only a 429 did, so a permanent
1502
+ # refusal was re-issued on every scheduled pass from every machine in the
1503
+ # fleet — and those retries are what earned the 429s that made an
1504
+ # unauthorized account look merely rate-limited. Telemetry then sat frozen
1505
+ # while the shim ranked the whole pool NEUTRAL, i.e. picked at random.
1506
+ denied = (str(e.headers.get('x-should-retry') or '').lower() == 'false'
1507
+ or 'scope requirement' in body)
1508
+ # A setup token can NEVER grow the user:profile scope, so that refusal is
1509
+ # permanent and parks for hours. Any other "do not retry" may well be a
1510
+ # policy or endpoint issue someone fixes in minutes — park it for an hour,
1511
+ # not six, so recovery does not wait on a human running --force.
1512
+ scope_denied = denied and 'scope requirement' in body
1513
+ if scope_denied:
1514
+ wait = USAGE_DENIED_BACKOFF
1515
+ elif denied:
1516
+ wait = USAGE_DENIED_BACKOFF // 6
1517
+ else:
1518
+ wait = min(1800, max(USAGE_FAIL_BACKOFF,
1519
+ int(num('backoff', USAGE_FAIL_BACKOFF // 2)) * 2))
1520
+ if scope_denied and source == 'token':
1521
+ # Remember WHICH token was refused, so this credential is never spent
1522
+ # on this endpoint again — but a freshly minted one still gets a try.
1523
+ tdigest = token_digest(tpath)
1524
+ if tdigest:
1525
+ prev['token_scope_denied'] = tdigest
1526
+ park(wait, f'HTTP {e.code} (source={source})'
1527
+ + (' — permanent, server said do not retry' if denied else ''))
1528
+ if scope_denied and source == 'token':
1529
+ # The exact shape of this outage: a setup token is minted WITHOUT the
1530
+ # user:profile scope the usage endpoint requires, so an account whose
1531
+ # OAuth grant lapsed keeps working for inference and goes permanently
1532
+ # dark for telemetry. Only a sign-in ON THIS MACHINE restores it.
1533
+ say(f'{aid}: usage endpoint refuses the setup token (HTTP {e.code} — a '
1534
+ f'setup token has no user:profile scope). Telemetry is DEAD for this '
1535
+ f'account until it has an OAuth login here: claude-accounts login {aid}. '
1536
+ f'Backing off {wait}s.')
1537
+ elif denied:
1538
+ say(f'{aid}: usage fetch refused for good (HTTP {e.code}, source={source}); '
1539
+ f'backing off {wait}s — re-login needed: claude-accounts login {aid}')
1540
+ else:
1541
+ say(f'{aid}: usage fetch failed (HTTP {e.code}); backing off {wait}s; failing open')
1542
+ continue
1543
+ except Exception as e:
1544
+ # Network/parse trouble is transient by nature, but it still must not be retried
1545
+ # every 5 minutes forever — that is how a fleet talks itself into a 429.
1546
+ wait = min(1800, max(USAGE_FAIL_BACKOFF,
1547
+ int(num('backoff', USAGE_FAIL_BACKOFF // 2)) * 2))
1548
+ try:
1293
1549
  prev['retry_after'] = int(now + wait)
1294
1550
  prev['backoff'] = wait
1551
+ prev['last_error'] = str(e)[:200]
1552
+ prev['last_error_at'] = int(now)
1295
1553
  tmp = lpath + '.tmp'
1296
1554
  with open(tmp, 'w') as f:
1297
1555
  json.dump(prev, f, indent=1)
1298
1556
  os.replace(tmp, lpath)
1299
- say(f'{aid}: rate limited (429); backing off {wait}s; failing open')
1300
- else:
1301
- say(f'{aid}: usage fetch failed (HTTP {e.code}); failing open')
1302
- continue
1303
- except Exception as e:
1304
- say(f'{aid}: usage fetch failed ({e}); failing open')
1557
+ except Exception:
1558
+ pass
1559
+ say(f'{aid}: usage fetch failed ({e}); backing off {wait}s; failing open')
1305
1560
  continue
1306
1561
 
1307
1562
  def pct_of(v):
@@ -1387,8 +1642,21 @@ for acct in manifest.get('accounts', []):
1387
1642
  session = [b['percent'] for b in buckets if b['group'] == 'session']
1388
1643
  weeklyp = max(weekly) if weekly else maxp
1389
1644
  sessionp = max(session) if session else 0
1645
+ # How long weekly_percent keeps meaning something. A weekly bucket only ever RISES
1646
+ # until its reset, so before that moment a stale percent is still a valid lower
1647
+ # bound and the shim can rank on it when nothing fresher exists; after it, the
1648
+ # number describes a week that is over and says nothing at all. Recording the
1649
+ # horizon here keeps the shim from having to parse buckets[] on every invocation.
1650
+ # It must come from the bucket weekly_percent actually CAME FROM: a low monthly
1651
+ # bucket resetting in an hour says nothing about an 80% weekly one that resets in
1652
+ # five days, and taking the minimum over all of them would throw the 80% away.
1653
+ wresets = [int(b['resets_epoch']) for b in buckets
1654
+ if b['group'] != 'session' and b['percent'] == weeklyp
1655
+ and isinstance(b.get('resets_epoch'), int)]
1390
1656
  out = {'fetched_at': int(now), 'source': source, 'max_percent': maxp,
1391
- 'weekly_percent': weeklyp, 'session_percent': sessionp, 'buckets': buckets}
1657
+ 'weekly_percent': weeklyp, 'session_percent': sessionp,
1658
+ 'weekly_resets_epoch': min(wresets) if wresets else 0,
1659
+ 'buckets': buckets}
1392
1660
  tmp = lpath + '.tmp'
1393
1661
  with open(tmp, 'w') as f:
1394
1662
  json.dump(out, f, indent=1)
@@ -1483,6 +1751,14 @@ def mark_expired(d, slug, detail=''):
1483
1751
  """Park an account the shim must stop selecting. Verify is the strongest signal
1484
1752
  there is — a real inference call that came back 'not authenticated'."""
1485
1753
  mpath = os.path.join(d, '.expired')
1754
+ # Same rule as the limits path: an org block outranks every other reason and only
1755
+ # a PASSING verify or a re-login may lift it. Rewriting its reason with a weaker
1756
+ # one is how it gets lifted by accident later.
1757
+ try:
1758
+ if 'reason=org-blocked' in open(mpath, errors='replace').read():
1759
+ return
1760
+ except OSError:
1761
+ pass
1486
1762
  try:
1487
1763
  with open(mpath + '.tmp', 'w') as f:
1488
1764
  f.write(f'{int(time.time())}\n')
package/install.sh CHANGED
@@ -71,6 +71,27 @@ case "$ACC_ROOT$CODEX_ACC_ROOT" in
71
71
  "*) echo "pool root contains a character that cannot be scheduled safely: $ACC_ROOT / $CODEX_ACC_ROOT" >&2; exit 1 ;;
72
72
  esac
73
73
 
74
+ # A pool root under a temp directory is by definition gone tomorrow, but the launchd
75
+ # agent or crontab line naming it is not: it keeps firing forever against a path that
76
+ # no longer exists, one leaked set per run. That is how a Mac collected 57 orphaned
77
+ # agent sets and a server 4 stray cron blocks. EITHER root disqualifies the whole set,
78
+ # because one install writes agents for both providers: the leaked sets above named a
79
+ # temp CLAUDE root and the REAL codex pool, so they went right on polling a live usage
80
+ # endpoint every five minutes. Install the binaries for such a root; never schedule it.
81
+ ephemeral_root() { # $1 = a pool root
82
+ case "$1" in
83
+ /tmp/*|/private/tmp/*|/var/tmp/*|/private/var/tmp/*|/var/folders/*|/private/var/folders/*)
84
+ return 0 ;;
85
+ *) return 1 ;;
86
+ esac
87
+ }
88
+ if [ "$NO_SCHEDULE" != "1" ] && { ephemeral_root "$ACC_ROOT" || ephemeral_root "$CODEX_ACC_ROOT"; }; then
89
+ NO_SCHEDULE=1
90
+ echo " note: a pool root is under a temp directory (claude: $ACC_ROOT, codex: $CODEX_ACC_ROOT)" >&2
91
+ echo " — installing WITHOUT schedulers, which would outlive it." >&2
92
+ echo " Pass --no-schedule to silence this." >&2
93
+ fi
94
+
74
95
  LABEL="com.claude-multiacc${INSTANCE:+.$INSTANCE}"
75
96
  MARK_BEGIN="# >>> claude-multiacc >>>"
76
97
  MARK_END="# <<< claude-multiacc <<<"
@@ -150,7 +171,7 @@ mac_schedule_install() {
150
171
  <string>limits</string>
151
172
  <string>--quiet</string>
152
173
  </array>
153
- $(plist_env_block) <key>StartInterval</key><integer>300</integer>
174
+ $(plist_env_block) <key>StartInterval</key><integer>900</integer>
154
175
  <key>RunAtLoad</key><true/>
155
176
  <key>StandardOutPath</key><string>/dev/null</string>
156
177
  <key>StandardErrorPath</key><string>/dev/null</string>
@@ -239,7 +260,7 @@ EOF
239
260
  launchctl load -w "$PLIST_CODEX_LIMITS" 2>/dev/null || true
240
261
  launchctl load -w "$PLIST_CODEX_HEALTH" 2>/dev/null || true
241
262
  [ "${CLAUDE_MULTIACC_AUTOUPDATE:-1}" = "1" ] && launchctl load -w "$PLIST_UPDATE" 2>/dev/null || true
242
- echo " launchd: limits refresh every 5m (claude + codex) + weekly health checks + daily auto-update"
263
+ echo " launchd: limits refresh (claude 15m / codex 5m) + weekly health checks + daily auto-update"
243
264
  }
244
265
 
245
266
  mac_schedule_remove() {
@@ -262,7 +283,7 @@ linux_schedule_install() {
262
283
  env="$(cron_env_prefix)"
263
284
  { crontab -l 2>/dev/null | cron_strip_ours; } > "$tmp"
264
285
  {
265
- echo "*/5 * * * * $env$REPO_DIR/bin/claude-accounts limits --quiet >/dev/null 2>&1 $CRON_TAG"
286
+ echo "*/15 * * * * $env$REPO_DIR/bin/claude-accounts limits --quiet >/dev/null 2>&1 $CRON_TAG"
266
287
  echo "*/5 * * * * $env$REPO_DIR/bin/codex-accounts limits --quiet >/dev/null 2>&1 $CRON_TAG"
267
288
  echo "17 9 * * 1 $env$REPO_DIR/bin/claude-accounts health >/dev/null 2>&1 $CRON_TAG"
268
289
  echo "37 9 * * 1 $env$REPO_DIR/bin/codex-accounts health >/dev/null 2>&1 $CRON_TAG"
@@ -271,7 +292,7 @@ linux_schedule_install() {
271
292
  } >> "$tmp"
272
293
  crontab "$tmp"
273
294
  rm -f "$tmp"
274
- echo " cron: limits refresh every 5m (claude + codex) + weekly health checks + daily auto-update"
295
+ echo " cron: limits refresh (claude 15m / codex 5m) + weekly health checks + daily auto-update"
275
296
  }
276
297
 
277
298
  linux_schedule_remove() {