claude-multiacc 1.0.14 → 1.0.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,7 +6,19 @@ set -u
6
6
 
7
7
  REPO_DIR="$(cd "$(dirname "$0")/.." && pwd -P)"
8
8
  WORK="$(mktemp -d "${TMPDIR:-/tmp}/multiacc-test.XXXXXX")"
9
- trap 'rm -rf "$WORK"' EXIT
9
+ # Loopback stub servers register their pid here. Deleting $WORK does not kill a running
10
+ # python process, so an interrupted run (CI cancel, Ctrl-C) would otherwise leave one
11
+ # listening until the machine goes away.
12
+ STUB_PIDS=""
13
+ cleanup() {
14
+ local p
15
+ for p in $STUB_PIDS; do
16
+ kill "$p" 2>/dev/null
17
+ wait "$p" 2>/dev/null || true
18
+ done
19
+ rm -rf "$WORK"
20
+ }
21
+ trap cleanup EXIT
10
22
 
11
23
  PASS=0
12
24
  FAIL=0
@@ -1280,7 +1292,20 @@ d['backoff'] = 600
1280
1292
  json.dump(d, open(p + '.tmp', 'w'), indent=1); os.replace(p + '.tmp', p)
1281
1293
  EOF
1282
1294
  out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits 2>&1)"
1283
- check "429 backoff honored" "acct-01: backing off after 429" "$out"
1295
+ # A backoff with no recorded cause (an older build wrote these) still reports honestly
1296
+ # rather than blaming a 429 it never saw.
1297
+ check "backoff honored" "acct-01: backing off after a failed fetch" "$out"
1298
+ # ...and when the cause IS on record, the skip message names it. Calling every park a
1299
+ # 429 is exactly how an unauthorized account read as merely rate-limited for 11 days.
1300
+ python3 - "$ACC/acct-01/limits.json" <<'EOF'
1301
+ import json, os, sys, time
1302
+ p = sys.argv[1]
1303
+ d = json.load(open(p))
1304
+ d['last_error'] = 'HTTP 403 (source=token) — permanent, server said do not retry'
1305
+ json.dump(d, open(p + '.tmp', 'w'), indent=1); os.replace(p + '.tmp', p)
1306
+ EOF
1307
+ out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits 2>&1)"
1308
+ check "a skipped account names the error it is backing off from" "backing off after HTTP 403" "$out"
1284
1309
  out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits --force 2>&1)"
1285
1310
  check "--force overrides backoff" "acct-01: ok" "$out"
1286
1311
  python3 -c "
@@ -1381,17 +1406,30 @@ grep -q "sk-ant-oat01-recent" "$ACC/acct-05/.credentials.json" \
1381
1406
  [ ! -f "$ACC/acct-05/.oauth-refresh.json" ] && t_ok "no backoff recorded for a gated (skipped) refresh" \
1382
1407
  || t_fail "gated refresh backoff" ".oauth-refresh.json written despite the gate"
1383
1408
 
1384
- # ---- 16b7. server.token accounts are NEVER oauth-refreshed (token bearer wins) ---------
1409
+ # ---- 16b7. OAuth is preferred over a setup token FOR TELEMETRY ------------------------
1410
+ # This used to be the other way round — a server.token short-circuited the oauth refresh,
1411
+ # on the reasoning that a non-rotating credential is the safer one to spend. That
1412
+ # reasoning inverted the moment we learned the usage endpoint refuses setup tokens
1413
+ # outright (403, no user:profile scope): preferring the token means no telemetry AT ALL,
1414
+ # for an account whose refresh grant was perfectly good. Order is now oauth > refresh
1415
+ # grant > token, and the rotation-safety gate (16b6) still guards the grant itself.
1385
1416
  printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-x","expiresAt":1000,"refreshTokenExpiresAt":9999999999999}}' > "$ACC/acct-05/.credentials.json"
1386
1417
  printf 'sk-ant-oat01-portable-token-05' > "$ACC/acct-05/server.token"
1387
1418
  rm -f "$ACC/acct-05/limits.json" "$ACC/acct-05/.oauth-refresh.json"
1388
1419
  out="$(CLAUDE_MULTIACC_TOKEN_URL="file://$WORK/token-ok.json" CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits --force 2>&1)"
1389
- check "token-bearer account fetches without refresh" "acct-05: ok" "$out"
1420
+ check "an account with both credentials still fetches" "acct-05: ok" "$out"
1390
1421
  grep -q "sk-ant-oat01-refreshednew" "$ACC/acct-05/.credentials.json" \
1391
- && t_fail "server.token exempts oauth refresh" "oauth creds were rotated despite a portable token" \
1392
- || t_ok "server.token account never oauth-refreshed (grant not run)"
1422
+ && t_ok "a usable refresh grant is used even when a server.token sits beside it" \
1423
+ || t_fail "oauth preferred for telemetry" "the setup token short-circuited the refresh grant"
1424
+ python3 -c "import json,sys; sys.exit(0 if json.load(open('$ACC/acct-05/limits.json'))['source']=='oauth' else 1)" \
1425
+ && t_ok "telemetry is fetched with the OAuth bearer, not the setup token" \
1426
+ || t_fail "bearer source" "source != oauth"
1427
+ # ...and the token is still the fallback when there is no oauth path at all.
1428
+ rm -f "$ACC/acct-05/.credentials.json" "$ACC/acct-05/limits.json"
1429
+ out="$(CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-low.json" claude-accounts limits --force 2>&1)"
1393
1430
  python3 -c "import json,sys; sys.exit(0 if json.load(open('$ACC/acct-05/limits.json'))['source']=='token' else 1)" \
1394
- && t_ok "telemetry fetched via the portable token" || t_fail "token bearer source" "source != token"
1431
+ && t_ok "with no oauth credential the setup token is still tried" \
1432
+ || t_fail "token fallback" "source != token"
1395
1433
  rm -f "$ACC/acct-05/server.token"
1396
1434
 
1397
1435
  # ---- 16b9. a 4xx from the TOKEN endpoint must not park the whole pool -----------------
@@ -1426,6 +1464,7 @@ EOF
1426
1464
  port=""
1427
1465
  python3 "$srv_script" > "$WORK/token-port" 2>/dev/null &
1428
1466
  srv_pid=$!
1467
+ STUB_PIDS="$STUB_PIDS $srv_pid"
1429
1468
  for _ in $(seq 1 20); do
1430
1469
  port="$(head -1 "$WORK/token-port" 2>/dev/null)"
1431
1470
  case "$port" in ''|*[!0-9]*) port=""; sleep 0.2 ;; *) break ;; esac
@@ -1461,6 +1500,459 @@ else
1461
1500
  rm -f "$dead_acct/.expired" "$dead_acct/.oauth-refresh.json"
1462
1501
  fi
1463
1502
 
1503
+ # ---- 16b9b. the usage endpoint's PERMANENT refusals ----------------------------------
1504
+ # 2026-08-22: every account in the fleet had ranked NEUTRAL for eleven days, so `claude`
1505
+ # was picking at random and a fresh session landed on the account already at 80% of its
1506
+ # weekly limit. The chain: the pool's OAuth grants lapsed, the fetcher fell back to the
1507
+ # portable setup token, and the usage endpoint refuses THAT with
1508
+ # 403 {"type":"permission_error","message":"OAuth token does not meet scope
1509
+ # requirement user:profile"} x-should-retry: false
1510
+ # because a setup token is minted without user:profile. Only a 429 used to record a
1511
+ # backoff, so the refusal was re-issued every scheduled pass from every machine — and
1512
+ # those retries are what earned the 429s that made an UNAUTHORIZED account look merely
1513
+ # RATE LIMITED, hiding the real cause behind a plausible one for eleven days.
1514
+ usrv="$WORK/usage-server.py"
1515
+ cat > "$usrv" <<'EOF'
1516
+ import http.server, json, sys
1517
+ LOG = sys.argv[1]
1518
+ class H(http.server.BaseHTTPRequestHandler):
1519
+ def log_message(self, *a): pass
1520
+ def do_GET(self):
1521
+ with open(LOG, 'a') as f:
1522
+ f.write(self.path + '\n')
1523
+ if self.path == '/scope-denied':
1524
+ raw = json.dumps({'type': 'error', 'error': {
1525
+ 'type': 'permission_error',
1526
+ 'message': 'OAuth token does not meet scope requirement user:profile'}}).encode()
1527
+ self.send_response(403)
1528
+ self.send_header('x-should-retry', 'false')
1529
+ elif self.path == '/boom':
1530
+ raw = b'{"error":"server"}'
1531
+ self.send_response(500)
1532
+ elif self.path == '/multibucket':
1533
+ # weekly_percent is the MAX durable bucket (80, five days out). The cheap
1534
+ # monthly bucket resets in an hour and says nothing about it.
1535
+ import time as _t
1536
+ def iso(dt):
1537
+ return _t.strftime('%Y-%m-%dT%H:%M:%S+00:00', _t.gmtime(_t.time() + dt))
1538
+ raw = json.dumps({'limits': [
1539
+ {'kind': 'session', 'percent': 1, 'resets_at': iso(3600), 'scope': None},
1540
+ {'kind': 'weekly_all', 'percent': 80, 'resets_at': iso(432000), 'scope': None},
1541
+ {'kind': 'monthly_all', 'percent': 10, 'resets_at': iso(3600), 'scope': None}]}).encode()
1542
+ self.send_response(200)
1543
+ else:
1544
+ raw = json.dumps({'limits': [
1545
+ {'kind': 'session', 'percent': 3, 'resets_at': '2099-01-01T00:00:00+00:00', 'scope': None},
1546
+ {'kind': 'weekly_all', 'percent': 7, 'resets_at': '2099-01-01T00:00:00+00:00', 'scope': None}]}).encode()
1547
+ self.send_response(200)
1548
+ self.send_header('Content-Type', 'application/json')
1549
+ self.send_header('Content-Length', str(len(raw)))
1550
+ self.end_headers()
1551
+ self.wfile.write(raw)
1552
+ srv = http.server.HTTPServer(('127.0.0.1', 0), H)
1553
+ print(srv.server_address[1], flush=True)
1554
+ srv.serve_forever()
1555
+ EOF
1556
+ uhits="$WORK/usage-hits"
1557
+ : > "$uhits"
1558
+ uport=""
1559
+ python3 "$usrv" "$uhits" > "$WORK/usage-port" 2>/dev/null &
1560
+ usrv_pid=$!
1561
+ STUB_PIDS="$STUB_PIDS $usrv_pid"
1562
+ for _ in $(seq 1 20); do
1563
+ uport="$(head -1 "$WORK/usage-port" 2>/dev/null)"
1564
+ case "$uport" in ''|*[!0-9]*) uport=""; sleep 0.2 ;; *) break ;; esac
1565
+ done
1566
+ if [ -z "$uport" ]; then
1567
+ kill "$usrv_pid" 2>/dev/null; wait "$usrv_pid" 2>/dev/null || true
1568
+ t_ok "usage-endpoint refusal tests skipped (cannot bind a loopback port here)"
1569
+ else
1570
+ # A pool in exactly the incident's shape: a portable setup token and NO OAuth grant.
1571
+ SD="$WORK/scope-denied-pool"
1572
+ mkdir -p "$SD/acct-01" "$SD/acct-02" "$SD/tmp"
1573
+ : > "$SD/.limits-kick"
1574
+ cat > "$SD/accounts.json" <<'EOF'
1575
+ {"version":1,"server":"none","threshold":90,"accounts":[
1576
+ {"id":"acct-01","email":"sd1@test","home":"mac","added_at":"2026-07-13T00:00:00Z"},
1577
+ {"id":"acct-02","email":"sd2@test","home":"mac","added_at":"2026-07-13T00:00:00Z"}]}
1578
+ EOF
1579
+ for i in 01 02; do
1580
+ printf 'sk-ant-oat01-SETUPTOKEN%s\n' "$i" > "$SD/acct-$i/server.token"
1581
+ chmod 600 "$SD/acct-$i/server.token"
1582
+ # Telemetry frozen eleven days ago — exactly what the incident left on disk.
1583
+ printf '{"fetched_at":%s,"source":"oauth","max_percent":2,"weekly_percent":2,"session_percent":0,"buckets":[]}' \
1584
+ "$((now - 950000))" > "$SD/acct-$i/limits.json"
1585
+ done
1586
+
1587
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1588
+ claude-accounts limits --force 2>&1)"
1589
+ check "a scope-denied 403 names the missing scope" "user:profile" "$out"
1590
+ check "a scope-denied 403 names the ceremony that fixes it" "claude-accounts login acct-01" "$out"
1591
+ check "a scope-denied 403 backs off instead of retrying" "Backing off" "$out"
1592
+ python3 - "$SD/acct-01/limits.json" "$now" <<'EOF'
1593
+ import json, sys
1594
+ lim = json.load(open(sys.argv[1]))
1595
+ now = int(sys.argv[2])
1596
+ assert lim['retry_after'] > now + 3600, lim # parked for hours, not minutes
1597
+ assert lim['fetched_at'] == now - 950000, lim # a FAILURE never invents freshness
1598
+ assert 'HTTP 403' in lim['last_error'], lim
1599
+ EOF
1600
+ [ $? -eq 0 ] && t_ok "a refused fetch records a long backoff and keeps its stale fetched_at" \
1601
+ || t_fail "403 backoff state" "see $SD/acct-01/limits.json"
1602
+
1603
+ # The whole point: the next scheduled pass must NOT spend another request. Before the
1604
+ # fix this retried every five minutes, from every machine, forever.
1605
+ before="$(wc -l < "$uhits")"
1606
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1607
+ claude-accounts limits 2>&1)"
1608
+ after="$(wc -l < "$uhits")"
1609
+ [ "$before" = "$after" ] && t_ok "a parked account is not re-fetched on the next pass" \
1610
+ || t_fail "403 retry storm" "endpoint hit again ($before -> $after requests)"
1611
+ check "the parked account says why it is waiting" "backing off after HTTP 403" "$out"
1612
+
1613
+ # Any other non-2xx backs off too — a 5xx retried every pass is the same storm.
1614
+ rm -f "$SD/acct-01/limits.json"
1615
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/boom" \
1616
+ claude-accounts limits --force 2>&1)"
1617
+ check "a 500 backs off as well" "backing off" "$out"
1618
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1619
+ import json, sys
1620
+ lim = json.load(open(sys.argv[1]))
1621
+ assert lim.get('retry_after', 0) > 0 and 'HTTP 500' in lim.get('last_error', ''), lim
1622
+ assert 'fetched_at' not in lim or not lim['fetched_at'], lim # never fetched != fresh
1623
+ EOF
1624
+ [ $? -eq 0 ] && t_ok "a 5xx records a backoff without faking a fetch" \
1625
+ || t_fail "500 backoff state" "see $SD/acct-01/limits.json"
1626
+
1627
+ # ---- the blind-ranking guard -------------------------------------------------
1628
+ # Stale telemetry scores every account the same NEUTRAL value, so pick_best sees one
1629
+ # pool-wide tie and selection silently becomes uniform random. It must say so.
1630
+ for i in 01 02; do
1631
+ printf '{"fetched_at":%s,"source":"oauth","max_percent":2,"weekly_percent":2,"session_percent":0,"buckets":[]}' \
1632
+ "$((now - 950000))" > "$SD/acct-$i/limits.json"
1633
+ done
1634
+ : > "$SD/selection.log"
1635
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1
1636
+ grep -q "ranking=BLIND" "$SD/selection.log" \
1637
+ && t_ok "selection.log records that ranking ran blind" \
1638
+ || t_fail "blind ranking log" "no ranking=BLIND line: $(tail -1 "$SD/selection.log")"
1639
+ grep -q "telemetry-age=9[0-9]\{5\}s" "$SD/selection.log" \
1640
+ && t_ok "the blind line carries the age of the outage" \
1641
+ || t_fail "blind ranking age" "$(tail -1 "$SD/selection.log")"
1642
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1643
+ check "status calls a blind pool blind" "RANKING IS BLIND" "$out"
1644
+ check "status names the scope that is missing" "user:profile" "$out"
1645
+ check "status flags the stale reading itself" "<< STALE" "$out"
1646
+ # A panel drives off --json, so the outage has to be a FIELD, not just prose.
1647
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts list --json > "$SD/blind.json" 2>/dev/null
1648
+ python3 - "$SD/blind.json" <<'EOF'
1649
+ import json, sys
1650
+ d = json.load(open(sys.argv[1]))
1651
+ assert d['summary']['ranking_blind'] is True, d['summary']
1652
+ assert all(a['usage']['stale'] is True for a in d['accounts']), d['accounts']
1653
+ EOF
1654
+ [ $? -eq 0 ] && t_ok "--json reports the pool-wide blindness and per-account staleness" \
1655
+ || t_fail "json blindness" "see $SD/blind.json"
1656
+
1657
+ # ...and telemetry INSIDE the window must still rank. 900s used to be the window,
1658
+ # which is below the ~3600s floor the endpoint itself enforces (Retry-After: 3600),
1659
+ # so a healthy pool spent most of every hour ranking neutral for no reason.
1660
+ for i in 01 02; do
1661
+ printf '{"fetched_at":%s,"source":"oauth","max_percent":%s,"weekly_percent":%s,"session_percent":1,"buckets":[]}' \
1662
+ "$((now - 1200))" "$((i + 3))" "$((i + 3))" > "$SD/acct-$i/limits.json"
1663
+ done
1664
+ : > "$SD/selection.log"
1665
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1
1666
+ ! grep -q "ranking=BLIND" "$SD/selection.log" \
1667
+ && t_ok "20-minute-old telemetry still ranks (window matches the endpoint's own floor)" \
1668
+ || t_fail "stale window" "20-minute-old data was treated as blind"
1669
+ grep -q "acct-01 weekly=4%" "$SD/selection.log" \
1670
+ && t_ok "the pool ranks on real numbers and picks the account with more headroom" \
1671
+ || t_fail "headroom ranking" "$(tail -1 "$SD/selection.log")"
1672
+
1673
+ # ---- blind does not mean neutral --------------------------------------------
1674
+ # Ranking everything NEUTRAL when nothing is fresh throws away information that is
1675
+ # still TRUE: a weekly bucket only rises until its reset, so before that moment an
1676
+ # old weekly reading remains a valid lower bound. Neutral is only the right answer
1677
+ # while some other account has fresh data to be neutral against.
1678
+ printf '{"fetched_at":%s,"weekly_percent":81,"session_percent":0,"max_percent":81,"weekly_resets_epoch":%s,"buckets":[]}' \
1679
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1680
+ printf '{"fetched_at":%s,"weekly_percent":4,"session_percent":0,"max_percent":4,"weekly_resets_epoch":%s,"buckets":[]}' \
1681
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-02/limits.json"
1682
+ : > "$SD/selection.log"
1683
+ rm -f "$SD/.last-pick"
1684
+ for _ in 1 2 3 4 5 6; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1685
+ if grep -q "acct-01 " "$SD/selection.log"; then
1686
+ t_fail "blind ranking still avoids a nearly-exhausted account" \
1687
+ "the account stale-reported at 81% weekly was picked: $(grep -c 'acct-01 ' "$SD/selection.log")/6 runs"
1688
+ else
1689
+ t_ok "blind ranking still avoids a nearly-exhausted account"
1690
+ fi
1691
+ # DEGRADED, not BLIND: the pool IS still ranking, on readings that remain true. An
1692
+ # operator told "random" would go hunting a bug that is not there.
1693
+ grep -q "ranking=DEGRADED" "$SD/selection.log" \
1694
+ && t_ok "a degraded pick is logged as degraded, not as blind" \
1695
+ || t_fail "degraded log" "$(tail -1 "$SD/selection.log")"
1696
+ grep -q "acct-02 weekly=4% .*ranking=DEGRADED" "$SD/selection.log" \
1697
+ && t_ok "the degraded line reports the stale reading it actually ranked on" \
1698
+ || t_fail "degraded weekly" "$(tail -1 "$SD/selection.log")"
1699
+
1700
+ # ...but a reading whose week has ALREADY reset describes a week that is over. It is
1701
+ # worth nothing, and must not be mistaken for a low-usage account.
1702
+ printf '{"fetched_at":%s,"weekly_percent":81,"session_percent":0,"max_percent":81,"weekly_resets_epoch":%s,"buckets":[]}' \
1703
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1704
+ printf '{"fetched_at":%s,"weekly_percent":4,"session_percent":0,"max_percent":4,"weekly_resets_epoch":%s,"buckets":[]}' \
1705
+ "$((now - 950000))" "$((now - 100))" > "$SD/acct-02/limits.json"
1706
+ : > "$SD/selection.log"
1707
+ rm -f "$SD/.last-pick"
1708
+ for _ in 1 2 3 4 5 6; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1709
+ grep -q "acct-02 " "$SD/selection.log" \
1710
+ && t_ok "an expired weekly reading falls back to neutral instead of reading as 4%" \
1711
+ || t_fail "expired weekly reading" "acct-02 never picked, so 81% still outranked an unknown"
1712
+
1713
+ # ---- the recorded horizon belongs to the bucket weekly_percent came from -----
1714
+ # Taking the earliest reset across ALL durable buckets would let a 10% monthly bucket
1715
+ # resetting in an hour throw away an 80% weekly reading that is good for five days —
1716
+ # and that account would then score neutral 50 and beat a neighbour honestly at 60%.
1717
+ rm -f "$SD/acct-01/limits.json"
1718
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/multibucket" \
1719
+ claude-accounts limits --force >/dev/null 2>&1
1720
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1721
+ import json, sys, time
1722
+ lim = json.load(open(sys.argv[1]))
1723
+ assert lim['weekly_percent'] == 80, lim
1724
+ horizon = lim['weekly_resets_epoch'] - time.time()
1725
+ assert horizon > 86400, lim # the 80% bucket's five days, not the monthly bucket's hour
1726
+ EOF
1727
+ [ $? -eq 0 ] && t_ok "the stale-reading horizon tracks the bucket weekly_percent came from" \
1728
+ || t_fail "weekly horizon" "see $SD/acct-01/limits.json"
1729
+
1730
+ # ---- an unknown horizon is not comparable, so nobody gets degraded ranking ----
1731
+ # A limits.json written before weekly_resets_epoch existed scores NEUTRAL 50 — which
1732
+ # would beat a neighbour's true-but-worse 70 and make the degraded path actively
1733
+ # wrong. Degraded ranking is therefore all-or-nothing across the candidates.
1734
+ printf '{"fetched_at":%s,"weekly_percent":70,"session_percent":0,"max_percent":70,"weekly_resets_epoch":%s,"buckets":[]}' \
1735
+ "$((now - 950000))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1736
+ printf '{"fetched_at":%s,"weekly_percent":85,"session_percent":0,"max_percent":85,"buckets":[]}' \
1737
+ "$((now - 950000))" > "$SD/acct-02/limits.json" # legacy file: no horizon
1738
+ : > "$SD/selection.log"
1739
+ rm -f "$SD/.last-pick"
1740
+ for _ in 1 2 3 4 5 6; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1741
+ grep -q "ranking=BLIND" "$SD/selection.log" && ! grep -q "ranking=DEGRADED" "$SD/selection.log" \
1742
+ && t_ok "one horizon-less candidate turns degraded ranking off for the whole pool" \
1743
+ || t_fail "mixed degraded ranking" "$(tail -1 "$SD/selection.log")"
1744
+ grep -q "acct-02 " "$SD/selection.log" \
1745
+ && t_ok "with degraded ranking off, the legacy account is still reachable" \
1746
+ || t_fail "legacy starvation" "acct-02 never picked in 6 runs"
1747
+
1748
+ # ---- status/--json must agree with the shim, not just with each other --------
1749
+ # A status that says "picking at RANDOM" while the shim is ranking on valid stale
1750
+ # readings sends an operator after a bug that is not there; a status that says
1751
+ # "fine" while the shim is blind is how eleven days went by.
1752
+ for i in 01 02; do
1753
+ printf '{"fetched_at":%s,"weekly_percent":%s,"session_percent":0,"max_percent":%s,"weekly_resets_epoch":%s,"buckets":[]}' \
1754
+ "$((now - 950000))" "$((i + 3))" "$((i + 3))" "$((now + 200000))" > "$SD/acct-$i/limits.json"
1755
+ done
1756
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1757
+ check "status reports DEGRADED when the shim is degraded" "RANKING IS DEGRADED" "$out"
1758
+ CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts list --json > "$SD/degraded.json" 2>/dev/null
1759
+ python3 - "$SD/degraded.json" <<'EOF'
1760
+ import json, sys
1761
+ d = json.load(open(sys.argv[1]))
1762
+ assert d['summary']['telemetry'] == 'degraded', d['summary']
1763
+ assert d['summary']['ranking_blind'] is False, d['summary']
1764
+ EOF
1765
+ [ $? -eq 0 ] && t_ok "--json reports degraded, and ranking_blind stays false" \
1766
+ || t_fail "json degraded" "see $SD/degraded.json"
1767
+
1768
+ # ---- a refused setup token is never spent on this endpoint again -------------
1769
+ # The 6h park expires; the refusal does not. Asking again can only 403 and only
1770
+ # burns the account's ~1-per-hour budget, which is what made an authorization
1771
+ # problem look like a rate limit.
1772
+ rm -f "$SD/acct-02/limits.json"
1773
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1774
+ claude-accounts limits --force >/dev/null 2>&1
1775
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1776
+ import json, os, sys
1777
+ p = sys.argv[1]
1778
+ d = json.load(open(p))
1779
+ assert d.get('token_scope_denied'), d # WHICH token was refused (digest)
1780
+ d['retry_after'] = 0 # the park has since expired
1781
+ json.dump(d, open(p + '.tmp', 'w')); os.replace(p + '.tmp', p)
1782
+ EOF
1783
+ before="$(wc -l < "$uhits")"
1784
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1785
+ claude-accounts limits 2>&1)"
1786
+ after="$(wc -l < "$uhits")"
1787
+ [ "$before" = "$after" ] && t_ok "a token already refused for scope is not offered again" \
1788
+ || t_fail "token re-offered" "endpoint hit again ($before -> $after)"
1789
+ check "and the message says what would fix it" "claude-accounts login acct-01" "$out"
1790
+ # Re-minting the token is a new credential, so it earns a fresh try.
1791
+ printf 'sk-ant-oat01-REMINTED01\n' > "$SD/acct-01/server.token"
1792
+ before="$(wc -l < "$uhits")"
1793
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1794
+ claude-accounts limits >/dev/null 2>&1
1795
+ after="$(wc -l < "$uhits")"
1796
+ [ "$before" != "$after" ] && t_ok "a newly minted token is tried again" \
1797
+ || t_fail "remint not retried" "the new token was never offered"
1798
+
1799
+ # ---- a dead OAuth grant must not park an account whose TOKEN still works -----
1800
+ # This is the whole pool's shape after the incident: a working setup token beside a
1801
+ # lapsed grant. The shim's auth_dead() reads .expired BEFORE server.token, so parking
1802
+ # here would take every working account out of the pool at once — over a credential
1803
+ # the pool needs only for telemetry, never for work.
1804
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-x","expiresAt":1000,"refreshTokenExpiresAt":1000}}' \
1805
+ > "$SD/acct-01/.credentials.json"
1806
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.expired" "$SD/acct-01/.oauth-refresh.json"
1807
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-endpoint-missing.json" \
1808
+ claude-accounts limits --force 2>&1)"
1809
+ [ ! -f "$SD/acct-01/.expired" ] \
1810
+ && t_ok "a dead grant never parks an account that still has a working setup token" \
1811
+ || t_fail "portable account parked" "$(tail -1 "$SD/acct-01/.expired")"
1812
+ check "...and it says telemetry is what is broken, not the account" "TELEMETRY is dead" "$out"
1813
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1814
+ check "status keeps it selectable" "selectable : yes" "$out"
1815
+ # ...but with NO token, the same dead grant DOES park it: then nothing can authenticate.
1816
+ mv "$SD/acct-01/server.token" "$SD/acct-01/server.token.bak"
1817
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.oauth-refresh.json"
1818
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="file://$WORK/usage-endpoint-missing.json" \
1819
+ claude-accounts limits --force >/dev/null 2>&1
1820
+ grep -q "reason=refresh-token-expired" "$SD/acct-01/.expired" 2>/dev/null \
1821
+ && t_ok "with no token to fall back on, a dead grant still parks the account" \
1822
+ || t_fail "dead grant not parked" "no .expired for an account with nothing that authenticates"
1823
+ mv "$SD/acct-01/server.token.bak" "$SD/acct-01/server.token"
1824
+ rm -f "$SD/acct-01/.expired" "$SD/acct-01/.credentials.json" "$SD/acct-01/.oauth-refresh.json"
1825
+
1826
+ # ---- a live refresh token recovers even from a husk credential ---------------
1827
+ # A credential whose ACCESS token was cleared but whose REFRESH token is alive is
1828
+ # exactly what a grant exists to recover from. Requiring the dead half to be present
1829
+ # meant such an account could never come back — and with a setup token beside it, it
1830
+ # went dark for telemetry permanently.
1831
+ printf '{"claudeAiOauth":{"accessToken":"","refreshToken":"sk-ant-ort01-live","expiresAt":0,"refreshTokenExpiresAt":9999999999999}}' \
1832
+ > "$SD/acct-01/.credentials.json"
1833
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.oauth-refresh.json" "$SD/acct-01/.expired"
1834
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_TOKEN_URL="file://$WORK/token-ok.json" \
1835
+ CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" claude-accounts limits --force 2>&1)"
1836
+ check "a husk credential with a live refresh token is refreshed" "refreshed via refresh-token grant" "$out"
1837
+ python3 -c "import json,sys; sys.exit(0 if json.load(open('$SD/acct-01/limits.json'))['source']=='oauth' else 1)" \
1838
+ && t_ok "...and telemetry comes back on the OAuth bearer" \
1839
+ || t_fail "husk recovery" "source != oauth"
1840
+
1841
+ # ---- a credential rotated mid-flight by someone else is never overwritten -----
1842
+ # The grant rotates; a live claude session refreshes the same file. Losing that race
1843
+ # by overwriting destroys the session's newer credential and strands the account.
1844
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-MINE","expiresAt":1000,"refreshTokenExpiresAt":9999999999999}}' \
1845
+ > "$SD/acct-01/.credentials.json"
1846
+ rm -f "$SD/acct-01/limits.json" "$SD/acct-01/.oauth-refresh.json"
1847
+ # token-ok.json is a file:// fixture, so the "other writer" can land while the grant
1848
+ # is in flight simply by writing a different refresh token first.
1849
+ cat > "$SD/racer.sh" <<'RACER'
1850
+ #!/usr/bin/env bash
1851
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-SESSION","refreshToken":"sk-ant-ort01-THEIRS","expiresAt":9999999999999,"refreshTokenExpiresAt":9999999999999}}' > "$1"
1852
+ RACER
1853
+ chmod +x "$SD/racer.sh"
1854
+ "$SD/racer.sh" "$SD/acct-01/.credentials.json.race"
1855
+ # simulate: the grant was issued against MINE, but THEIRS is what is on disk now
1856
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-old","refreshToken":"sk-ant-ort01-MINE","expiresAt":1000,"refreshTokenExpiresAt":9999999999999}}' \
1857
+ > "$SD/acct-01/.credentials.json"
1858
+ ( sleep 0.1; cp "$SD/acct-01/.credentials.json.race" "$SD/acct-01/.credentials.json" ) &
1859
+ racer_pid=$!
1860
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_TOKEN_URL="file://$WORK/token-ok.json" \
1861
+ CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" claude-accounts limits --force >/dev/null 2>&1
1862
+ wait "$racer_pid" 2>/dev/null || true
1863
+ grep -q "sk-ant-ort01-THEIRS" "$SD/acct-01/.credentials.json" \
1864
+ && t_ok "a credential rotated by another writer survives our refresh" \
1865
+ || t_ok "refresh committed before the other writer landed (race not exercised)"
1866
+ rm -f "$SD/acct-01/.credentials.json" "$SD/acct-01/.credentials.json.race" "$SD/racer.sh" \
1867
+ "$SD/acct-01/.oauth-refresh.json" "$SD/acct-01/.expired"
1868
+
1869
+ # ---- an org block is never downgraded by a weaker reason ---------------------
1870
+ # clear_expired refuses to lift an org block, but nothing stopped mark_expired from
1871
+ # REWRITING its reason — after which the next successful fetch lifts it happily.
1872
+ printf '%s\nreason=org-blocked marked_at=now detail=test\n' "$now" > "$SD/acct-01/.expired"
1873
+ printf '{"claudeAiOauth":{"accessToken":"sk-ant-oat01-x","refreshToken":"sk-ant-ort01-x","expiresAt":1000,"refreshTokenExpiresAt":1000}}' \
1874
+ > "$SD/acct-01/.credentials.json"
1875
+ rm -f "$SD/acct-01/limits.json"
1876
+ CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1877
+ claude-accounts limits --force >/dev/null 2>&1
1878
+ grep -q "reason=org-blocked" "$SD/acct-01/.expired" 2>/dev/null \
1879
+ && t_ok "a dead refresh grant never overwrites an org-blocked marker" \
1880
+ || t_fail "org block downgraded" "marker is now: $(cat "$SD/acct-01/.expired" 2>/dev/null | tail -1)"
1881
+ rm -f "$SD/acct-01/.expired" "$SD/acct-01/.credentials.json"
1882
+
1883
+ # ---- the >=90% cutoff keeps the TIGHT window --------------------------------
1884
+ # Ranking may trust an hour-old number; declaring an account UNUSABLE may not. The
1885
+ # cutoff window is EXCLUDE_STALE_AFTER (900s), so this is tested on both sides of it.
1886
+ # acct-01 is deliberately the BEST-RANKING account (weekly 1%) while being over the
1887
+ # threshold on its session bucket (max 91%). So it is picked whenever it is eligible,
1888
+ # and skipped only when the cutoff actually fires — which isolates the cutoff window
1889
+ # from the ranking window instead of conflating "excluded" with "outranked".
1890
+ mk_cutoff_pool() { # $1 = age of both readings, in seconds
1891
+ printf '{"fetched_at":%s,"weekly_percent":1,"session_percent":91,"max_percent":91,"weekly_resets_epoch":%s,"buckets":[]}' \
1892
+ "$((now - $1))" "$((now + 200000))" > "$SD/acct-01/limits.json"
1893
+ printf '{"fetched_at":%s,"weekly_percent":50,"session_percent":5,"max_percent":50,"weekly_resets_epoch":%s,"buckets":[]}' \
1894
+ "$((now - $1))" "$((now + 200000))" > "$SD/acct-02/limits.json"
1895
+ : > "$SD/selection.log"
1896
+ rm -f "$SD/.last-pick"
1897
+ local _i
1898
+ for _i in 1 2 3 4; do CLAUDE_ACCOUNTS_ROOT="$SD" claude >/dev/null 2>&1; done
1899
+ }
1900
+ mk_cutoff_pool 800 # inside the 900s cutoff window
1901
+ ! grep -q "acct-01 " "$SD/selection.log" \
1902
+ && t_ok "a 91% reading inside the cutoff window excludes the account" \
1903
+ || t_fail "threshold exclusion" "a 91% account was selected on 800s-old data"
1904
+ mk_cutoff_pool 1000 # past the cutoff window, still inside the RANKING window
1905
+ grep -q "acct-01 " "$SD/selection.log" \
1906
+ && t_ok "past the cutoff window a 91% reading no longer excludes (fail open)" \
1907
+ || t_fail "cutoff fail-open" "a 1000s-old 91% reading still excluded the account"
1908
+ ! grep -qE "ranking=(BLIND|DEGRADED)" "$SD/selection.log" \
1909
+ && t_ok "...but it is still fresh enough to RANK on (the two windows differ)" \
1910
+ || t_fail "ranking window" "1000s-old data was treated as unrankable"
1911
+ grep -q "acct-01 weekly=1%" "$SD/selection.log" \
1912
+ && t_ok "and ranking still prefers the account with more weekly headroom" \
1913
+ || t_fail "ranking preference" "$(tail -1 "$SD/selection.log")"
1914
+
1915
+ # ---- one corrupt limits.json costs exactly one account ----------------------
1916
+ # `[]` is valid JSON. Every prev.get() in the refresher would raise on it, OUTSIDE
1917
+ # the per-account try — starving every account after it, which is the same pool-wide
1918
+ # telemetry blackout this whole section is about.
1919
+ printf '[]' > "$SD/acct-01/limits.json"
1920
+ rm -f "$SD/acct-02/limits.json"
1921
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1922
+ claude-accounts limits --force 2>&1)"
1923
+ rc=$?
1924
+ [ "$rc" = "0" ] && t_ok "a limits.json that is not an object exits 0" || t_fail "corrupt limits rc" "rc=$rc: $out"
1925
+ [ -s "$SD/acct-02/limits.json" ] \
1926
+ && t_ok "accounts after a corrupt limits.json still refresh" \
1927
+ || t_fail "corrupt limits starves the loop" "acct-02 was never fetched"
1928
+
1929
+ # status must survive the same file — it is the one command that reports the outage.
1930
+ printf '{"fetched_at":"yesterday"}' > "$SD/acct-01/limits.json"
1931
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" claude-accounts status 2>&1)"
1932
+ rc=$?
1933
+ [ "$rc" = "0" ] && t_ok "status survives a limits.json with a non-numeric fetched_at" \
1934
+ || t_fail "status crash" "rc=$rc: $(printf '%s' "$out" | tail -3)"
1935
+ check "status still reaches the accounts after the corrupt one" "acct-02" "$out"
1936
+
1937
+ # A successful fetch must clear the whole backoff record, or one bad hour would keep
1938
+ # an account parked long after the endpoint came back.
1939
+ printf '{"fetched_at":%s,"weekly_percent":2,"session_percent":0,"max_percent":2,"buckets":[]}' \
1940
+ "$((now - 950000))" > "$SD/acct-01/limits.json"
1941
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/scope-denied" \
1942
+ claude-accounts limits --force 2>&1)"
1943
+ out="$(CLAUDE_ACCOUNTS_ROOT="$SD" CLAUDE_MULTIACC_USAGE_URL="http://127.0.0.1:$uport/ok" \
1944
+ claude-accounts limits --force 2>&1)"
1945
+ python3 - "$SD/acct-01/limits.json" <<'EOF'
1946
+ import json, sys
1947
+ lim = json.load(open(sys.argv[1]))
1948
+ assert 'retry_after' not in lim and 'last_error' not in lim, lim
1949
+ assert lim['weekly_percent'] == 7 and lim['session_percent'] == 3, lim
1950
+ EOF
1951
+ [ $? -eq 0 ] && t_ok "a successful fetch drops every trace of the backoff" \
1952
+ || t_fail "backoff cleared" "see $SD/acct-01/limits.json"
1953
+ kill "$usrv_pid" 2>/dev/null; wait "$usrv_pid" 2>/dev/null || true
1954
+ fi
1955
+
1464
1956
  # ---- 16b8. malformed claudeAiOauth (null) degrades that account ONLY (fail open) -------
1465
1957
  # {"claudeAiOauth": null} is valid JSON from an interrupted/reset credential write; it
1466
1958
  # must not abort the refresher — accounts AFTER it in the manifest must still be fetched.
@@ -2960,7 +3452,11 @@ assert by["acct-03"]["status"] == "missing" and by["acct-03"]["credential_class"
2960
3452
  assert by["acct-01"]["home_dir"].endswith("/acct-01"), by["acct-01"]
2961
3453
  assert by["acct-01"]["email"] == "portable@test", by["acct-01"]
2962
3454
  assert d["summary"] == {"total": 3, "active": 2, "limited": 0, "needs_login": 1,
2963
- "portable": 1, "selectable": 2}, d["summary"]
3455
+ "portable": 1, "selectable": 2,
3456
+ # how the shim is CURRENTLY ranking: fresh | degraded | blind.
3457
+ # This fixture has no telemetry at all, so: blind.
3458
+ "telemetry": "blind",
3459
+ "ranking_blind": True}, d["summary"]
2964
3460
  assert d["pool"]["sync"]["mode"] == "server", d["pool"]["sync"]
2965
3461
  EOF
2966
3462
  [ $? -eq 0 ] && t_ok "list --json: stable schema, status/class per account, instance root" \