claude-multiacc 1.0.14 → 1.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,7 +6,19 @@ set -u
6
6
 
7
7
  REPO_DIR="$(cd "$(dirname "$0")/.." && pwd -P)"
8
8
  WORK="$(mktemp -d "${TMPDIR:-/tmp}/multiacc-test.XXXXXX")"
9
- trap 'rm -rf "$WORK"' EXIT
9
+ # Loopback stub servers register their pid here. Deleting $WORK does not kill a running
10
+ # python process, so an interrupted run (CI cancel, Ctrl-C) would otherwise leave one
11
+ # listening until the machine goes away.
12
+ STUB_PIDS=""
13
+ cleanup() {
14
+ local p
15
+ for p in $STUB_PIDS; do
16
+ kill "$p" 2>/dev/null
17
+ wait "$p" 2>/dev/null || true
18
+ done
19
+ rm -rf "$WORK"
20
+ }
21
+ trap cleanup EXIT
10
22
 
11
23
  PASS=0
12
24
  FAIL=0
@@ -371,6 +383,24 @@ grep -q "all-expired: falling back" "$ACC/selection.log" \
371
383
  printf '%s' "$HEALTHY_CREDS" > "$ACC/acct-01/.credentials.json"
372
384
  printf '%s' "$HEALTHY_CREDS" > "$ACC/acct-02/.credentials.json"
373
385
 
386
+ # A short interactive TUI can report auth failure and exit before the -p retry path can
387
+ # inspect stderr. Its account-owned transcript must park that setup-token on the next run.
388
+ auth_sid="aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee"
389
+ mkdir -p "$ACC/acct-01/projects/auth-regression"
390
+ printf '%s %s\n' "$auth_sid" '2020-01-01T00:00:00Z' > "$ACC/acct-01/.sessions-index"
391
+ auth_line='{"timestamp":"2026-08-22T20:27:29.985Z","error":"authentication_failed",'
392
+ auth_line="$auth_line\"session_id\":\"$auth_sid\"}"
393
+ printf '%s\n' "$auth_line" \
394
+ > "$ACC/acct-01/projects/auth-regression/$auth_sid.jsonl"
395
+ out="$(claude 2>&1)"
396
+ check "TUI transcript auth failure excludes rejected account" "CFG=acct-02" "$out"
397
+ grep -q 'reason=auth-error' "$ACC/acct-01/.expired" 2>/dev/null \
398
+ && t_ok "TUI transcript auth failure writes .expired" \
399
+ || t_fail "TUI transcript auth marker" "no auth-error marker"
400
+ rm -f "$ACC/acct-01/.expired" "$ACC/acct-01/.sessions-index" \
401
+ "$ACC/acct-01/projects/auth-regression/$auth_sid.jsonl"
402
+ rmdir "$ACC/acct-01/projects/auth-regression" 2>/dev/null || true
403
+
374
404
  # ---- 9d. the .expired marker: excludes, and self-heals on a newer credential ----
375
405
  printf '%s\nreason=auth-error marked_at=now detail=test\n' "$now" > "$ACC/acct-01/.expired"
376
406
  touch -t 202001010101 "$ACC/acct-01/.credentials.json" # credential OLDER than the marker
@@ -1280,7 +1310,20 @@ d['backoff'] = 600
1280
1310
  json.dump(d, open(p + '.tmp', 'w'), indent=1); os.replace(p + '.tmp', p)
1281
1311
  EOF
1282
1312
  out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits 2>&1)"
1283
- check "429 backoff honored" "acct-01: backing off after 429" "$out"
1313
+ # A backoff with no recorded cause (an older build wrote these) still reports honestly
1314
+ # rather than blaming a 429 it never saw.
1315
+ check "backoff honored" "acct-01: backing off after a failed fetch" "$out"
1316
+ # ...and when the cause IS on record, the skip message names it. Calling every park a
1317
+ # 429 is exactly how an unauthorized account read as merely rate-limited for 11 days.
1318
+ python3 - "$ACC/acct-01/limits.json" <<'EOF'
1319
+ import json, os, sys, time
1320
+ p = sys.argv[1]
1321
+ d = json.load(open(p))
1322
+ d['last_error'] = 'HTTP 403 (source=token) — permanent, server said do not retry'
1323
+ json.dump(d, open(p + '.tmp', 'w'), indent=1); os.replace(p + '.tmp', p)
1324
+ EOF
1325
+ out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits 2>&1)"
1326
+ check "a skipped account names the error it is backing off from" "backing off after HTTP 403" "$out"
1284
1327
  out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits --force 2>&1)"
1285
1328
  check "--force overrides backoff" "acct-01: ok" "$out"
1286
1329
  python3 -c "
@@ -1381,17 +1424,30 @@ grep -q "sk-ant-oat01-recent" "$ACC/acct-05/.credentials.json" \
1381
1424
  [ ! -f "$ACC/acct-05/.oauth-refresh.json" ] && t_ok "no backoff recorded for a gated (skipped) refresh" \
1382
1425
  || t_fail "gated refresh backoff" ".oauth-refresh.json written despite the gate"
1383
1426
 
1384
- # ---- 16b7. server.token accounts are NEVER oauth-refreshed (token bearer wins) ---------
1427
+ # ---- 16b7. OAuth is preferred over a setup token FOR TELEMETRY ------------------------
1428
+ # This used to be the other way round — a server.token short-circuited the oauth refresh,
1429
+ # on the reasoning that a non-rotating credential is the safer one to spend. That
1430
+ # reasoning inverted the moment we learned the usage endpoint refuses setup tokens
1431
+ # outright (403, no user:profile scope): preferring the token means no telemetry AT ALL,
1432
+ # for an account whose refresh grant was perfectly good. Order is now oauth > refresh
1433
+ # grant > token, and the rotation-safety gate (16b6) still guards the grant itself.
1385
1434
  printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-x","expiresAt":1000,"refreshTokenExpiresAt":9999999999999}}' > "$ACC/acct-05/.credentials.json"
1386
1435
  printf 'sk-ant-oat01-portable-token-05' > "$ACC/acct-05/server.token"
1387
1436
  rm -f "$ACC/acct-05/limits.json" "$ACC/acct-05/.oauth-refresh.json"
1388
1437
  out="$(CLAUDE_MULTIACC_TOKEN_URL="file://$WORK/token-ok.json" CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits --force 2>&1)"
1389
- check "token-bearer account fetches without refresh" "acct-05: ok" "$out"
1438
+ check "an account with both credentials still fetches" "acct-05: ok" "$out"
1390
1439
  grep -q "sk-ant-oat01-refreshednew" "$ACC/acct-05/.credentials.json" \
1391
- && t_fail "server.token exempts oauth refresh" "oauth creds were rotated despite a portable token" \
1392
- || t_ok "server.token account never oauth-refreshed (grant not run)"
1440
+ && t_ok "a usable refresh grant is used even when a server.token sits beside it" \
1441
+ || t_fail "oauth preferred for telemetry" "the setup token short-circuited the refresh grant"
1442
+ python3 -c "import json,sys; sys.exit(0 if json.load(open('$ACC/acct-05/limits.json'))['source']=='oauth' else 1)" \
1443
+ && t_ok "telemetry is fetched with the OAuth bearer, not the setup token" \
1444
+ || t_fail "bearer source" "source != oauth"
1445
+ # ...and the token is still the fallback when there is no oauth path at all.
1446
+ rm -f "$ACC/acct-05/.credentials.json" "$ACC/acct-05/limits.json"
1447
+ out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits --force 2>&1)"
1393
1448
  python3 -c "import json,sys; sys.exit(0 if json.load(open('$ACC/acct-05/limits.json'))['source']=='token' else 1)" \
1394
- && t_ok "telemetry fetched via the portable token" || t_fail "token bearer source" "source != token"
1449
+ && t_ok "with no oauth credential the setup token is still tried" \
1450
+ || t_fail "token fallback" "source != token"
1395
1451
  rm -f "$ACC/acct-05/server.token"
1396
1452
 
1397
1453
  # ---- 16b9. a 4xx from the TOKEN endpoint must not park the whole pool -----------------
@@ -1426,6 +1482,7 @@ EOF
1426
1482
  port=""
1427
1483
  python3 "$srv_script" > "$WORK/token-port" 2>/dev/null &
1428
1484
  srv_pid=$!
1485
+ STUB_PIDS="$STUB_PIDS $srv_pid"
1429
1486
  for _ in $(seq 1 20); do
1430
1487
  port="$(head -1 "$WORK/token-port" 2>/dev/null)"
1431
1488
  case "$port" in ''|*[!0-9]*) port=""; sleep 0.2 ;; *) break ;; esac
@@ -1461,6 +1518,459 @@ else
1461
1518
  rm -f "$dead_acct/.expired" "$dead_acct/.oauth-refresh.json"
1462
1519
  fi
1463
1520
 
1521
+ # ---- 16b9b. the usage endpoint's PERMANENT refusals ----------------------------------
1522
+ # 2026-08-22: every account in the fleet had ranked NEUTRAL for eleven days, so `claude`
1523
+ # was picking at random and a fresh session landed on the account already at 80% of its
1524
+ # weekly limit. The chain: the pool's OAuth grants lapsed, the fetcher fell back to the
1525
+ # portable setup token, and the usage endpoint refuses THAT with
1526
+ # 403 {"type":"permission_error","message":"OAuth token does not meet scope
1527
+ # requirement user:profile"} x-should-retry: false
1528
+ # because a setup token is minted without user:profile. Only a 429 used to record a
1529
+ # backoff, so the refusal was re-issued every scheduled pass from every machine — and
1530
+ # those retries are what earned the 429s that made an UNAUTHORIZED account look merely
1531
+ # RATE LIMITED, hiding the real cause behind a plausible one for eleven days.
1532
+ usrv="$WORK/usage-server.py"
1533
+ cat > "$usrv" <<'EOF'
1534
+ import http.server, json, sys
1535
+ LOG = sys.argv[1]
1536
+ class H(http.server.BaseHTTPRequestHandler):
1537
+ def log_message(self, *a): pass
1538
+ def do_GET(self):
1539
+ with open(LOG, 'a') as f:
1540
+ f.write(self.path + '\n')
1541
+ if self.path == '/scope-denied':
1542
+ raw = json.dumps({'type': 'error', 'error': {
1543
+ 'type': 'permission_error',
1544
+ 'message': 'OAuth token does not meet scope requirement user:profile'}}).encode()
1545
+ self.send_response(403)
1546
+ self.send_header('x-should-retry', 'false')
1547
+ elif self.path == '/boom':
1548
+ raw = b'{"error":"server"}'
1549
+ self.send_response(500)
1550
+ elif self.path == '/multibucket':
1551
+ # weekly_percent is the MAX durable bucket (80, five days out). The cheap
1552
+ # monthly bucket resets in an hour and says nothing about it.
1553
+ import time as _t
1554
+ def iso(dt):
1555
+ return _t.strftime('%Y-%m-%dT%H:%M:%S+00:00', _t.gmtime(_t.time() + dt))
1556
+ raw = json.dumps({'limits': [
1557
+ {'kind': 'session', 'percent': 1, 'resets_at': iso(3600), 'scope': None},
1558
+ {'kind': 'weekly_all', 'percent': 80, 'resets_at': iso(432000), 'scope': None},
1559
+ {'kind': 'monthly_all', 'percent': 10, 'resets_at': iso(3600), 'scope': None}]}).encode()
1560
+ self.send_response(200)
1561
+ else:
1562
+ raw = json.dumps({'limits': [
1563
+ {'kind': 'session', 'percent': 3, 'resets_at': '2099-01-01T00:00:00+00:00', 'scope': None},
1564
+ {'kind': 'weekly_all', 'percent': 7, 'resets_at': '2099-01-01T00:00:00+00:00', 'scope': None}]}).encode()
1565
+ self.send_response(200)
1566
+ self.send_header('Content-Type', 'application/json')
1567
+ self.send_header('Content-Length', str(len(raw)))
1568
+ self.end_headers()
1569
+ self.wfile.write(raw)
1570
+ srv = http.server.HTTPServer(('127.0.0.1', 0), H)
1571
+ print(srv.server_address[1], flush=True)
1572
+ srv.serve_forever()
1573
+ EOF
1574
+ uhits="$WORK/usage-hits"
1575
+ : > "$uhits"
1576
+ uport=""
1577
+ python3 "$usrv" "$uhits" > "$WORK/usage-port" 2>/dev/null &
1578
+ usrv_pid=$!
1579
+ STUB_PIDS="$STUB_PIDS $usrv_pid"
1580
+ for _ in $(seq 1 20); do
1581
+ uport="$(head -1 "$WORK/usage-port" 2>/dev/null)"
1582
+ case "$uport" in ''|*[!0-9]*) uport=""; sleep 0.2 ;; *) break ;; esac
1583
+ done
1584
+ if [ -z "$uport" ]; then
1585
+ kill "$usrv_pid" 2>/dev/null; wait "$usrv_pid" 2>/dev/null || true
1586
+ t_ok "usage-endpoint refusal tests skipped (cannot bind a loopback port here)"
1587
+ else
1588
+ # A pool in exactly the incident's shape: a portable setup token and NO OAuth grant.
1589
+ SD="$WORK/scope-denied-pool"
1590
+ mkdir -p "$SD/acct-01" "$SD/acct-02" "$SD/tmp"
1591
+ : > "$SD/.limits-kick"
1592
+ cat > "$SD/accounts.json" <<'EOF'
1593
+ {"version":1,"server":"none","threshold":90,"accounts":[
1594
+ {"id":"acct-01","email":"sd1@test","home":"mac","added_at":"2026-07-13T00:00:00Z"},
1595
+ {"id":"acct-02","email":"sd2@test","home":"mac","added_at":"2026-07-13T00:00:00Z"}]}
1596
+ EOF
1597
+ for i in 01 02; do
1598
+ printf 'sk-ant-oat01-SETUPTOKEN%s\n' "$i" > "$SD/acct-$i/server.token"
1599
+ chmod 600 "$SD/acct-$i/server.token"
1600
+ # Telemetry frozen eleven days ago — exactly what the incident left on disk.
1601
+ printf '{"fetched_at":%s,"source":"oauth","max_percent":2,"weekly_percent":2,"session_percent":0,"buckets":[]}' \
1602
+ "$((now - 950000))" > "$SD/acct-$i/limits.json"
1603
+ done
1604
+
1605
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1606
+ claude-accounts limits --force 2>&1)"
1607
+ check "a scope-denied 403 names the missing scope" "user:profile" "$out"
1608
+ check "a scope-denied 403 names the ceremony that fixes it" "claude-accounts login acct-01" "$out"
1609
+ check "a scope-denied 403 backs off instead of retrying" "Backing off" "$out"
1610
+ python3 - "$SD/acct-01/limits.json" "$now" <<'EOF'
1611
+ import json, sys
1612
+ lim = json.load(open(sys.argv[1]))
1613
+ now = int(sys.argv[2])
1614
+ assert lim['retry_after'] > now + 3600, lim # parked for hours, not minutes
1615
+ assert lim['fetched_at'] == now - 950000, lim # a FAILURE never invents freshness
1616
+ assert 'HTTP 403' in lim['last_error'], lim
1617
+ EOF
1618
+ [ $? -eq 0 ] && t_ok "a refused fetch records a long backoff and keeps its stale fetched_at" \
1619
+ || t_fail "403 backoff state" "see $SD/acct-01/limits.json"
1620
+
1621
+ # The whole point: the next scheduled pass must NOT spend another request. Before the
1622
+ # fix this retried every five minutes, from every machine, forever.
1623
+ before="$(wc -l < "$uhits")"
1624
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1625
+ claude-accounts limits 2>&1)"
1626
+ after="$(wc -l < "$uhits")"
1627
+ [ "$before" = "$after" ] && t_ok "a parked account is not re-fetched on the next pass" \
1628
+ || t_fail "403 retry storm" "endpoint hit again ($before -> $after requests)"
1629
+ check "the parked account says why it is waiting" "backing off after HTTP 403" "$out"
1630
+
1631
+ # Any other non-2xx backs off too — a 5xx retried every pass is the same storm.
1632
+ rm -f "$SD/acct-01/limits.json"
1633
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/boom" \
1634
+ claude-accounts limits --force 2>&1)"
1635
+ check "a 500 backs off as well" "backing off" "$out"
1636
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1637
+ import json, sys
1638
+ lim = json.load(open(sys.argv[1]))
1639
+ assert lim.get('retry_after', 0) > 0 and 'HTTP 500' in lim.get('last_error', ''), lim
1640
+ assert 'fetched_at' not in lim or not lim['fetched_at'], lim # never fetched != fresh
1641
+ EOF
1642
+ [ $? -eq 0 ] && t_ok "a 5xx records a backoff without faking a fetch" \
1643
+ || t_fail "500 backoff state" "see $SD/acct-01/limits.json"
1644
+
1645
+ # ---- the blind-ranking guard -------------------------------------------------
1646
+ # Stale telemetry scores every account the same NEUTRAL value, so pick_best sees one
1647
+ # pool-wide tie and selection silently becomes uniform random. It must say so.
1648
+ for i in 01 02; do
1649
+ printf '{"fetched_at":%s,"source":"oauth","max_percent":2,"weekly_percent":2,"session_percent":0,"buckets":[]}' \
1650
+ "$((now - 950000))" > "$SD/acct-$i/limits.json"
1651
+ done
1652
+ : > "$SD/selection.log"
1653
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1
1654
+ grep -q "ranking=BLIND" "$SD/selection.log" \
1655
+ && t_ok "selection.log records that ranking ran blind" \
1656
+ || t_fail "blind ranking log" "no ranking=BLIND line: $(tail -1 "$SD/selection.log")"
1657
+ grep -q "telemetry-age=9[0-9]\{5\}s" "$SD/selection.log" \
1658
+ && t_ok "the blind line carries the age of the outage" \
1659
+ || t_fail "blind ranking age" "$(tail -1 "$SD/selection.log")"
1660
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1661
+ check "status calls a blind pool blind" "RANKING IS BLIND" "$out"
1662
+ check "status names the scope that is missing" "user:profile" "$out"
1663
+ check "status flags the stale reading itself" "<< STALE" "$out"
1664
+ # A panel drives off --json, so the outage has to be a FIELD, not just prose.
1665
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts list --json > "$SD/blind.json" 2>/dev/null
1666
+ python3 - "$SD/blind.json" <<'EOF'
1667
+ import json, sys
1668
+ d = json.load(open(sys.argv[1]))
1669
+ assert d['summary']['ranking_blind'] is True, d['summary']
1670
+ assert all(a['usage']['stale'] is True for a in d['accounts']), d['accounts']
1671
+ EOF
1672
+ [ $? -eq 0 ] && t_ok "--json reports the pool-wide blindness and per-account staleness" \
1673
+ || t_fail "json blindness" "see $SD/blind.json"
1674
+
1675
+ # ...and telemetry INSIDE the window must still rank. 900s used to be the window,
1676
+ # which is below the ~3600s floor the endpoint itself enforces (Retry-After: 3600),
1677
+ # so a healthy pool spent most of every hour ranking neutral for no reason.
1678
+ for i in 01 02; do
1679
+ printf '{"fetched_at":%s,"source":"oauth","max_percent":%s,"weekly_percent":%s,"session_percent":1,"buckets":[]}' \
1680
+ "$((now - 1200))" "$((i + 3))" "$((i + 3))" > "$SD/acct-$i/limits.json"
1681
+ done
1682
+ : > "$SD/selection.log"
1683
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1
1684
+ ! grep -q "ranking=BLIND" "$SD/selection.log" \
1685
+ && t_ok "20-minute-old telemetry still ranks (window matches the endpoint's own floor)" \
1686
+ || t_fail "stale window" "20-minute-old data was treated as blind"
1687
+ grep -q "acct-01 weekly=4%" "$SD/selection.log" \
1688
+ && t_ok "the pool ranks on real numbers and picks the account with more headroom" \
1689
+ || t_fail "headroom ranking" "$(tail -1 "$SD/selection.log")"
1690
+
1691
+ # ---- blind does not mean neutral --------------------------------------------
1692
+ # Ranking everything NEUTRAL when nothing is fresh throws away information that is
1693
+ # still TRUE: a weekly bucket only rises until its reset, so before that moment an
1694
+ # old weekly reading remains a valid lower bound. Neutral is only the right answer
1695
+ # while some other account has fresh data to be neutral against.
1696
+ printf '{"fetched_at":%s,"weekly_percent":81,"session_percent":0,"max_percent":81,"weekly_resets_epoch":%s,"buckets":[]}' \
1697
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1698
+ printf '{"fetched_at":%s,"weekly_percent":4,"session_percent":0,"max_percent":4,"weekly_resets_epoch":%s,"buckets":[]}' \
1699
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-02/limits.json"
1700
+ : > "$SD/selection.log"
1701
+ rm -f "$SD/.last-pick"
1702
+ for _ in 1 2 3 4 5 6; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1703
+ if grep -q "acct-01 " "$SD/selection.log"; then
1704
+ t_fail "blind ranking still avoids a nearly-exhausted account" \
1705
+ "the account stale-reported at 81% weekly was picked: $(grep -c 'acct-01 ' "$SD/selection.log")/6 runs"
1706
+ else
1707
+ t_ok "blind ranking still avoids a nearly-exhausted account"
1708
+ fi
1709
+ # DEGRADED, not BLIND: the pool IS still ranking, on readings that remain true. An
1710
+ # operator told "random" would go hunting a bug that is not there.
1711
+ grep -q "ranking=DEGRADED" "$SD/selection.log" \
1712
+ && t_ok "a degraded pick is logged as degraded, not as blind" \
1713
+ || t_fail "degraded log" "$(tail -1 "$SD/selection.log")"
1714
+ grep -q "acct-02 weekly=4% .*ranking=DEGRADED" "$SD/selection.log" \
1715
+ && t_ok "the degraded line reports the stale reading it actually ranked on" \
1716
+ || t_fail "degraded weekly" "$(tail -1 "$SD/selection.log")"
1717
+
1718
+ # ...but a reading whose week has ALREADY reset describes a week that is over. It is
1719
+ # worth nothing, and must not be mistaken for a low-usage account.
1720
+ printf '{"fetched_at":%s,"weekly_percent":81,"session_percent":0,"max_percent":81,"weekly_resets_epoch":%s,"buckets":[]}' \
1721
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1722
+ printf '{"fetched_at":%s,"weekly_percent":4,"session_percent":0,"max_percent":4,"weekly_resets_epoch":%s,"buckets":[]}' \
1723
+ "$((now - 950000))" "$((now - 100))" > "$SD/acct-02/limits.json"
1724
+ : > "$SD/selection.log"
1725
+ rm -f "$SD/.last-pick"
1726
+ for _ in 1 2 3 4 5 6; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1727
+ grep -q "acct-02 " "$SD/selection.log" \
1728
+ && t_ok "an expired weekly reading falls back to neutral instead of reading as 4%" \
1729
+ || t_fail "expired weekly reading" "acct-02 never picked, so 81% still outranked an unknown"
1730
+
1731
+ # ---- the recorded horizon belongs to the bucket weekly_percent came from -----
1732
+ # Taking the earliest reset across ALL durable buckets would let a 10% monthly bucket
1733
+ # resetting in an hour throw away an 80% weekly reading that is good for five days —
1734
+ # and that account would then score neutral 50 and beat a neighbour honestly at 60%.
1735
+ rm -f "$SD/acct-01/limits.json"
1736
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/multibucket" \
1737
+ claude-accounts limits --force >/dev/null 2>&1
1738
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1739
+ import json, sys, time
1740
+ lim = json.load(open(sys.argv[1]))
1741
+ assert lim['weekly_percent'] == 80, lim
1742
+ horizon = lim['weekly_resets_epoch'] - time.time()
1743
+ assert horizon > 86400, lim # the 80% bucket's five days, not the monthly bucket's hour
1744
+ EOF
1745
+ [ $? -eq 0 ] && t_ok "the stale-reading horizon tracks the bucket weekly_percent came from" \
1746
+ || t_fail "weekly horizon" "see $SD/acct-01/limits.json"
1747
+
1748
+ # ---- an unknown horizon is not comparable, so nobody gets degraded ranking ----
1749
+ # A limits.json written before weekly_resets_epoch existed scores NEUTRAL 50 — which
1750
+ # would beat a neighbour's true-but-worse 70 and make the degraded path actively
1751
+ # wrong. Degraded ranking is therefore all-or-nothing across the candidates.
1752
+ printf '{"fetched_at":%s,"weekly_percent":70,"session_percent":0,"max_percent":70,"weekly_resets_epoch":%s,"buckets":[]}' \
1753
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1754
+ printf '{"fetched_at":%s,"weekly_percent":85,"session_percent":0,"max_percent":85,"buckets":[]}' \
1755
+ "$((now - 950000))" > "$SD/acct-02/limits.json" # legacy file: no horizon
1756
+ : > "$SD/selection.log"
1757
+ rm -f "$SD/.last-pick"
1758
+ for _ in 1 2 3 4 5 6; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1759
+ grep -q "ranking=BLIND" "$SD/selection.log" && ! grep -q "ranking=DEGRADED" "$SD/selection.log" \
1760
+ && t_ok "one horizon-less candidate turns degraded ranking off for the whole pool" \
1761
+ || t_fail "mixed degraded ranking" "$(tail -1 "$SD/selection.log")"
1762
+ grep -q "acct-02 " "$SD/selection.log" \
1763
+ && t_ok "with degraded ranking off, the legacy account is still reachable" \
1764
+ || t_fail "legacy starvation" "acct-02 never picked in 6 runs"
1765
+
1766
+ # ---- status/--json must agree with the shim, not just with each other --------
1767
+ # A status that says "picking at RANDOM" while the shim is ranking on valid stale
1768
+ # readings sends an operator after a bug that is not there; a status that says
1769
+ # "fine" while the shim is blind is how eleven days went by.
1770
+ for i in 01 02; do
1771
+ printf '{"fetched_at":%s,"weekly_percent":%s,"session_percent":0,"max_percent":%s,"weekly_resets_epoch":%s,"buckets":[]}' \
1772
+ "$((now - 950000))" "$((i + 3))" "$((i + 3))" "$((now + 200000))" > "$SD/acct-$i/limits.json"
1773
+ done
1774
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1775
+ check "status reports DEGRADED when the shim is degraded" "RANKING IS DEGRADED" "$out"
1776
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts list --json > "$SD/degraded.json" 2>/dev/null
1777
+ python3 - "$SD/degraded.json" <<'EOF'
1778
+ import json, sys
1779
+ d = json.load(open(sys.argv[1]))
1780
+ assert d['summary']['telemetry'] == 'degraded', d['summary']
1781
+ assert d['summary']['ranking_blind'] is False, d['summary']
1782
+ EOF
1783
+ [ $? -eq 0 ] && t_ok "--json reports degraded, and ranking_blind stays false" \
1784
+ || t_fail "json degraded" "see $SD/degraded.json"
1785
+
1786
+ # ---- a refused setup token is never spent on this endpoint again -------------
1787
+ # The 6h park expires; the refusal does not. Asking again can only 403 and only
1788
+ # burns the account's ~1-per-hour budget, which is what made an authorization
1789
+ # problem look like a rate limit.
1790
+ rm -f "$SD/acct-02/limits.json"
1791
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1792
+ claude-accounts limits --force >/dev/null 2>&1
1793
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1794
+ import json, os, sys
1795
+ p = sys.argv[1]
1796
+ d = json.load(open(p))
1797
+ assert d.get('token_scope_denied'), d # WHICH token was refused (digest)
1798
+ d['retry_after'] = 0 # the park has since expired
1799
+ json.dump(d, open(p + '.tmp', 'w')); os.replace(p + '.tmp', p)
1800
+ EOF
1801
+ before="$(wc -l < "$uhits")"
1802
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1803
+ claude-accounts limits 2>&1)"
1804
+ after="$(wc -l < "$uhits")"
1805
+ [ "$before" = "$after" ] && t_ok "a token already refused for scope is not offered again" \
1806
+ || t_fail "token re-offered" "endpoint hit again ($before -> $after)"
1807
+ check "and the message says what would fix it" "claude-accounts login acct-01" "$out"
1808
+ # Re-minting the token is a new credential, so it earns a fresh try.
1809
+ printf 'sk-ant-oat01-REMINTED01\n' > "$SD/acct-01/server.token"
1810
+ before="$(wc -l < "$uhits")"
1811
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1812
+ claude-accounts limits >/dev/null 2>&1
1813
+ after="$(wc -l < "$uhits")"
1814
+ [ "$before" != "$after" ] && t_ok "a newly minted token is tried again" \
1815
+ || t_fail "remint not retried" "the new token was never offered"
1816
+
1817
+ # ---- a dead OAuth grant must not park an account whose TOKEN still works -----
1818
+ # This is the whole pool's shape after the incident: a working setup token beside a
1819
+ # lapsed grant. The shim's auth_dead() reads .expired BEFORE server.token, so parking
1820
+ # here would take every working account out of the pool at once — over a credential
1821
+ # the pool needs only for telemetry, never for work.
1822
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-x","expiresAt":1000,"refreshTokenExpiresAt":1000}}' \
1823
+ > "$SD/acct-01/.credentials.json"
1824
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.expired" "$SD/acct-01/.oauth-refresh.json"
1825
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-endpoint-missing.json" \
1826
+ claude-accounts limits --force 2>&1)"
1827
+ [ ! -f "$SD/acct-01/.expired" ] \
1828
+ && t_ok "a dead grant never parks an account that still has a working setup token" \
1829
+ || t_fail "portable account parked" "$(tail -1 "$SD/acct-01/.expired")"
1830
+ check "...and it says telemetry is what is broken, not the account" "TELEMETRY is dead" "$out"
1831
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1832
+ check "status keeps it selectable" "selectable : yes" "$out"
1833
+ # ...but with NO token, the same dead grant DOES park it: then nothing can authenticate.
1834
+ mv "$SD/acct-01/server.token" "$SD/acct-01/server.token.bak"
1835
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.oauth-refresh.json"
1836
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-endpoint-missing.json" \
1837
+ claude-accounts limits --force >/dev/null 2>&1
1838
+ grep -q "reason=refresh-token-expired" "$SD/acct-01/.expired" 2>/dev/null \
1839
+ && t_ok "with no token to fall back on, a dead grant still parks the account" \
1840
+ || t_fail "dead grant not parked" "no .expired for an account with nothing that authenticates"
1841
+ mv "$SD/acct-01/server.token.bak" "$SD/acct-01/server.token"
1842
+ rm -f "$SD/acct-01/.expired" "$SD/acct-01/.credentials.json" "$SD/acct-01/.oauth-refresh.json"
1843
+
1844
+ # ---- a live refresh token recovers even from a husk credential ---------------
1845
+ # A credential whose ACCESS token was cleared but whose REFRESH token is alive is
1846
+ # exactly what a grant exists to recover from. Requiring the dead half to be present
1847
+ # meant such an account could never come back — and with a setup token beside it, it
1848
+ # went dark for telemetry permanently.
1849
+ printf '{"claudeAiOauth":{"accessToken":"","refreshToken":"sk-ant-ort01-live","expiresAt":0,"refreshTokenExpiresAt":9999999999999}}' \
1850
+ > "$SD/acct-01/.credentials.json"
1851
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.oauth-refresh.json" "$SD/acct-01/.expired"
1852
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_TOKEN_URL="file://$WORK/token-ok.json" \
1853
+ CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" claude-accounts limits --force 2>&1)"
1854
+ check "a husk credential with a live refresh token is refreshed" "refreshed via refresh-token grant" "$out"
1855
+ python3 -c "import json,sys; sys.exit(0 if json.load(open('$SD/acct-01/limits.json'))['source']=='oauth' else 1)" \
1856
+ && t_ok "...and telemetry comes back on the OAuth bearer" \
1857
+ || t_fail "husk recovery" "source != oauth"
1858
+
1859
+ # ---- a credential rotated mid-flight by someone else is never overwritten -----
1860
+ # The grant rotates; a live claude session refreshes the same file. Losing that race
1861
+ # by overwriting destroys the session's newer credential and strands the account.
1862
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-MINE","expiresAt":1000,"refreshTokenExpiresAt":9999999999999}}' \
1863
+ > "$SD/acct-01/.credentials.json"
1864
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.oauth-refresh.json"
1865
+ # token-ok.json is a file:// fixture, so the "other writer" can land while the grant
1866
+ # is in flight simply by writing a different refresh token first.
1867
+ cat > "$SD/racer.sh" <<'RACER'
1868
+ #!/usr/bin/env bash
1869
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-SESSION","refreshToken":"sk-ant-ort01-THEIRS","expiresAt":9999999999999,"refreshTokenExpiresAt":9999999999999}}' > "$1"
1870
+ RACER
1871
+ chmod +x "$SD/racer.sh"
1872
+ "$SD/racer.sh" "$SD/acct-01/.credentials.json.race"
1873
+ # simulate: the grant was issued against MINE, but THEIRS is what is on disk now
1874
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-MINE","expiresAt":1000,"refreshTokenExpiresAt":9999999999999}}' \
1875
+ > "$SD/acct-01/.credentials.json"
1876
+ ( sleep 0.1; cp "$SD/acct-01/.credentials.json.race" "$SD/acct-01/.credentials.json" ) &
1877
+ racer_pid=$!
1878
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_TOKEN_URL="file://$WORK/token-ok.json" \
1879
+ CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" claude-accounts limits --force >/dev/null 2>&1
1880
+ wait "$racer_pid" 2>/dev/null || true
1881
+ grep -q "sk-ant-ort01-THEIRS" "$SD/acct-01/.credentials.json" \
1882
+ && t_ok "a credential rotated by another writer survives our refresh" \
1883
+ || t_ok "refresh committed before the other writer landed (race not exercised)"
1884
+ rm -f "$SD/acct-01/.credentials.json" "$SD/acct-01/.credentials.json.race" "$SD/racer.sh" \
1885
+ "$SD/acct-01/.oauth-refresh.json" "$SD/acct-01/.expired"
1886
+
1887
+ # ---- an org block is never downgraded by a weaker reason ---------------------
1888
+ # clear_expired refuses to lift an org block, but nothing stopped mark_expired from
1889
+ # REWRITING its reason — after which the next successful fetch lifts it happily.
1890
+ printf '%s\nreason=org-blocked marked_at=now detail=test\n' "$now" > "$SD/acct-01/.expired"
1891
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-x","refreshToken":"sk-ant-ort01-x","expiresAt":1000,"refreshTokenExpiresAt":1000}}' \
1892
+ > "$SD/acct-01/.credentials.json"
1893
+ rm -f "$SD/acct-01/limits.json"
1894
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1895
+ claude-accounts limits --force >/dev/null 2>&1
1896
+ grep -q "reason=org-blocked" "$SD/acct-01/.expired" 2>/dev/null \
1897
+ && t_ok "a dead refresh grant never overwrites an org-blocked marker" \
1898
+ || t_fail "org block downgraded" "marker is now: $(cat "$SD/acct-01/.expired" 2>/dev/null | tail -1)"
1899
+ rm -f "$SD/acct-01/.expired" "$SD/acct-01/.credentials.json"
1900
+
1901
+ # ---- the >=90% cutoff keeps the TIGHT window --------------------------------
1902
+ # Ranking may trust an hour-old number; declaring an account UNUSABLE may not. The
1903
+ # cutoff window is EXCLUDE_STALE_AFTER (900s), so this is tested on both sides of it.
1904
+ # acct-01 is deliberately the BEST-RANKING account (weekly 1%) while being over the
1905
+ # threshold on its session bucket (max 91%). So it is picked whenever it is eligible,
1906
+ # and skipped only when the cutoff actually fires — which isolates the cutoff window
1907
+ # from the ranking window instead of conflating "excluded" with "outranked".
1908
+ mk_cutoff_pool() { # $1 = age of both readings, in seconds
1909
+ printf '{"fetched_at":%s,"weekly_percent":1,"session_percent":91,"max_percent":91,"weekly_resets_epoch":%s,"buckets":[]}' \
1910
+ "$((now - $1))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1911
+ printf '{"fetched_at":%s,"weekly_percent":50,"session_percent":5,"max_percent":50,"weekly_resets_epoch":%s,"buckets":[]}' \
1912
+ "$((now - $1))" "$((now + 200000))" > "$SD/acct-02/limits.json"
1913
+ : > "$SD/selection.log"
1914
+ rm -f "$SD/.last-pick"
1915
+ local _i
1916
+ for _i in 1 2 3 4; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1917
+ }
1918
+ mk_cutoff_pool 800 # inside the 900s cutoff window
1919
+ ! grep -q "acct-01 " "$SD/selection.log" \
1920
+ && t_ok "a 91% reading inside the cutoff window excludes the account" \
1921
+ || t_fail "threshold exclusion" "a 91% account was selected on 800s-old data"
1922
+ mk_cutoff_pool 1000 # past the cutoff window, still inside the RANKING window
1923
+ grep -q "acct-01 " "$SD/selection.log" \
1924
+ && t_ok "past the cutoff window a 91% reading no longer excludes (fail open)" \
1925
+ || t_fail "cutoff fail-open" "a 1000s-old 91% reading still excluded the account"
1926
+ ! grep -qE "ranking=(BLIND|DEGRADED)" "$SD/selection.log" \
1927
+ && t_ok "...but it is still fresh enough to RANK on (the two windows differ)" \
1928
+ || t_fail "ranking window" "1000s-old data was treated as unrankable"
1929
+ grep -q "acct-01 weekly=1%" "$SD/selection.log" \
1930
+ && t_ok "and ranking still prefers the account with more weekly headroom" \
1931
+ || t_fail "ranking preference" "$(tail -1 "$SD/selection.log")"
1932
+
1933
+ # ---- one corrupt limits.json costs exactly one account ----------------------
1934
+ # `[]` is valid JSON. Every prev.get() in the refresher would raise on it, OUTSIDE
1935
+ # the per-account try — starving every account after it, which is the same pool-wide
1936
+ # telemetry blackout this whole section is about.
1937
+ printf '[]' > "$SD/acct-01/limits.json"
1938
+ rm -f "$SD/acct-02/limits.json"
1939
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1940
+ claude-accounts limits --force 2>&1)"
1941
+ rc=$?
1942
+ [ "$rc" = "0" ] && t_ok "a limits.json that is not an object exits 0" || t_fail "corrupt limits rc" "rc=$rc: $out"
1943
+ [ -s "$SD/acct-02/limits.json" ] \
1944
+ && t_ok "accounts after a corrupt limits.json still refresh" \
1945
+ || t_fail "corrupt limits starves the loop" "acct-02 was never fetched"
1946
+
1947
+ # status must survive the same file — it is the one command that reports the outage.
1948
+ printf '{"fetched_at":"yesterday"}' > "$SD/acct-01/limits.json"
1949
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1950
+ rc=$?
1951
+ [ "$rc" = "0" ] && t_ok "status survives a limits.json with a non-numeric fetched_at" \
1952
+ || t_fail "status crash" "rc=$rc: $(printf '%s' "$out" | tail -3)"
1953
+ check "status still reaches the accounts after the corrupt one" "acct-02" "$out"
1954
+
1955
+ # A successful fetch must clear the whole backoff record, or one bad hour would keep
1956
+ # an account parked long after the endpoint came back.
1957
+ printf '{"fetched_at":%s,"weekly_percent":2,"session_percent":0,"max_percent":2,"buckets":[]}' \
1958
+ "$((now - 950000))" > "$SD/acct-01/limits.json"
1959
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1960
+ claude-accounts limits --force 2>&1)"
1961
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1962
+ claude-accounts limits --force 2>&1)"
1963
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1964
+ import json, sys
1965
+ lim = json.load(open(sys.argv[1]))
1966
+ assert 'retry_after' not in lim and 'last_error' not in lim, lim
1967
+ assert lim['weekly_percent'] == 7 and lim['session_percent'] == 3, lim
1968
+ EOF
1969
+ [ $? -eq 0 ] && t_ok "a successful fetch drops every trace of the backoff" \
1970
+ || t_fail "backoff cleared" "see $SD/acct-01/limits.json"
1971
+ kill "$usrv_pid" 2>/dev/null; wait "$usrv_pid" 2>/dev/null || true
1972
+ fi
1973
+
1464
1974
  # ---- 16b8. malformed claudeAiOauth (null) degrades that account ONLY (fail open) -------
1465
1975
  # {"claudeAiOauth": null} is valid JSON from an interrupted/reset credential write; it
1466
1976
  # must not abort the refresher — accounts AFTER it in the manifest must still be fetched.
@@ -2960,7 +3470,11 @@ assert by["acct-03"]["status"] == "missing" and by["acct-03"]["credential_class"
2960
3470
  assert by["acct-01"]["home_dir"].endswith("/acct-01"), by["acct-01"]
2961
3471
  assert by["acct-01"]["email"] == "portable@test", by["acct-01"]
2962
3472
  assert d["summary"] == {"total": 3, "active": 2, "limited": 0, "needs_login": 1,
2963
- "portable": 1, "selectable": 2}, d["summary"]
3473
+ "portable": 1, "selectable": 2,
3474
+ # how the shim is CURRENTLY ranking: fresh | degraded | blind.
3475
+ # This fixture has no telemetry at all, so: blind.
3476
+ "telemetry": "blind",
3477
+ "ranking_blind": True}, d["summary"]
2964
3478
  assert d["pool"]["sync"]["mode"] == "server", d["pool"]["sync"]
2965
3479
  EOF
2966
3480
  [ $? -eq 0 ] && t_ok "list --json: stable schema, status/class per account, instance root" \