handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
r"""Does each key actually work? Checked without spending a token.
|
|
2
|
+
|
|
3
|
+
Every provider exposes a metadata endpoint that authenticates the caller and
|
|
4
|
+
returns no completion — listing models, or reporting the key's own status. So
|
|
5
|
+
31 credentials can be validated for **zero model tokens and zero quota**, which
|
|
6
|
+
matters because the alternative (a one-token completion each) would burn 31 of
|
|
7
|
+
a 50-request daily allowance just to ask whether the keys are real.
|
|
8
|
+
|
|
9
|
+
Two rules this module keeps:
|
|
10
|
+
|
|
11
|
+
* **The key travels in a header, never a URL.** Gemini accepts
|
|
12
|
+
`?key=` and that would put a live credential into proxy logs, shell history
|
|
13
|
+
and any error message containing the URL. `x-goog-api-key` does the same job
|
|
14
|
+
and leaves no trace.
|
|
15
|
+
* **A key is never echoed, including in failures.** An error body can contain
|
|
16
|
+
the key that was rejected, so responses are reduced to a status and a short
|
|
17
|
+
reason before they go anywhere near a terminal.
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import json
|
|
22
|
+
import time
|
|
23
|
+
import urllib.error
|
|
24
|
+
import urllib.request
|
|
25
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
|
|
28
|
+
from .providers import Account, all_accounts
|
|
29
|
+
|
|
30
|
+
TIMEOUT = 20
|
|
31
|
+
|
|
32
|
+
# Identify the client honestly. Groq and Cerebras sit behind Cloudflare, which
|
|
33
|
+
# refuses Python's default `User-Agent: Python-urllib/3.x` with error 1010 --
|
|
34
|
+
# a 403 that looks exactly like a rejected credential. Twelve perfectly good
|
|
35
|
+
# keys were reported as broken before this line existed (`docs/0034` §3).
|
|
36
|
+
USER_AGENT = "agentctl/0.2 (+https://github.com/csdeepak/HandCode)"
|
|
37
|
+
|
|
38
|
+
LIVE, BAD_KEY, LIMITED, UNREACHABLE = "live", "rejected", "limited", "unreachable"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class Result:
|
|
43
|
+
account: Account
|
|
44
|
+
status: str
|
|
45
|
+
detail: str = ""
|
|
46
|
+
#: Registry model ids the provider rejected as unknown. A catalogue entry
|
|
47
|
+
#: that has disappeared is a fact about the registry, not about the key,
|
|
48
|
+
#: and `proxy` leaves these out of the pool (`docs/0044` N6).
|
|
49
|
+
gone: tuple[str, ...] = ()
|
|
50
|
+
#: The registry model id that actually served, when one did. `init`
|
|
51
|
+
#: records it as the user's default: proven by a completion, not assumed.
|
|
52
|
+
model: str = ""
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def ok(self) -> bool:
|
|
56
|
+
# A rate-limited key is a WORKING key with no allowance left. Reporting
|
|
57
|
+
# it as broken would send someone to rotate a credential that is fine.
|
|
58
|
+
return self.status in (LIVE, LIMITED)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
# (url, header builder). Metadata only -- none of these generate tokens.
|
|
62
|
+
ENDPOINTS: dict[str, tuple[str, callable]] = {
|
|
63
|
+
"openrouter": ("https://openrouter.ai/api/v1/auth/key",
|
|
64
|
+
lambda k: {"Authorization": f"Bearer {k}"}),
|
|
65
|
+
"gemini": ("https://generativelanguage.googleapis.com/v1beta/models",
|
|
66
|
+
# Header, not `?key=`: a URL leaks into logs and error text.
|
|
67
|
+
lambda k: {"x-goog-api-key": k}),
|
|
68
|
+
"mistral": ("https://api.mistral.ai/v1/models",
|
|
69
|
+
lambda k: {"Authorization": f"Bearer {k}"}),
|
|
70
|
+
"cerebras": ("https://api.cerebras.ai/v1/models",
|
|
71
|
+
lambda k: {"Authorization": f"Bearer {k}"}),
|
|
72
|
+
"groq": ("https://api.groq.com/openai/v1/models",
|
|
73
|
+
lambda k: {"Authorization": f"Bearer {k}"}),
|
|
74
|
+
"anthropic": ("https://api.anthropic.com/v1/models",
|
|
75
|
+
lambda k: {"x-api-key": k, "anthropic-version": "2023-06-01"}),
|
|
76
|
+
"openai": ("https://api.openai.com/v1/models",
|
|
77
|
+
lambda k: {"Authorization": f"Bearer {k}"}),
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def check(account: Account) -> Result:
|
|
82
|
+
"""One credential. Never raises, never echoes the key."""
|
|
83
|
+
spec = ENDPOINTS.get(account.provider.name)
|
|
84
|
+
if spec is None:
|
|
85
|
+
return Result(account, UNREACHABLE, "no metadata endpoint known")
|
|
86
|
+
|
|
87
|
+
url, headers = spec
|
|
88
|
+
key = account.value
|
|
89
|
+
if not key:
|
|
90
|
+
return Result(account, BAD_KEY, "empty")
|
|
91
|
+
|
|
92
|
+
req = urllib.request.Request(url, headers={**headers(key),
|
|
93
|
+
"User-Agent": USER_AGENT})
|
|
94
|
+
try:
|
|
95
|
+
with urllib.request.urlopen(req, timeout=TIMEOUT) as fh:
|
|
96
|
+
body = fh.read(20_000)
|
|
97
|
+
return Result(account, LIVE, _describe(account, body))
|
|
98
|
+
except urllib.error.HTTPError as e:
|
|
99
|
+
body = _safe_body(e, key)
|
|
100
|
+
# 403 is NOT automatically a bad key. Cloudflare returns 403 with
|
|
101
|
+
# "error code: 1010" when it dislikes the client, which is a transport
|
|
102
|
+
# problem wearing an authentication problem's clothes.
|
|
103
|
+
if "error code: 101" in body or "cloudflare" in body.lower():
|
|
104
|
+
return Result(account, UNREACHABLE,
|
|
105
|
+
f"HTTP {e.code} from a CDN, not the API "
|
|
106
|
+
f"({body[:40]}) — the key was never checked")
|
|
107
|
+
if e.code == 401:
|
|
108
|
+
return Result(account, BAD_KEY, "HTTP 401 unauthorized")
|
|
109
|
+
if e.code == 403:
|
|
110
|
+
return Result(account, BAD_KEY, f"HTTP 403 forbidden ({body[:60]})")
|
|
111
|
+
if e.code == 429:
|
|
112
|
+
return Result(account, LIMITED, "rate limited (the key is valid)")
|
|
113
|
+
return Result(account, UNREACHABLE, f"HTTP {e.code}")
|
|
114
|
+
except Exception as e: # noqa: BLE001
|
|
115
|
+
return Result(account, UNREACHABLE, type(e).__name__)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _describe(account: Account, body: bytes) -> str:
|
|
119
|
+
"""Something useful from the response. Never the key."""
|
|
120
|
+
try:
|
|
121
|
+
d = json.loads(body)
|
|
122
|
+
except Exception: # noqa: BLE001
|
|
123
|
+
return "ok"
|
|
124
|
+
if account.provider.name == "openrouter":
|
|
125
|
+
data = d.get("data") or {}
|
|
126
|
+
return "free tier" if data.get("is_free_tier") else "paid"
|
|
127
|
+
for field in ("data", "models"):
|
|
128
|
+
if isinstance(d.get(field), list):
|
|
129
|
+
return f"{len(d[field])} models"
|
|
130
|
+
return "ok"
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
# ── the one honest quota number (`research/phase-10-3` §2.3, §5.3) ─────
|
|
134
|
+
@dataclass
|
|
135
|
+
class Quota:
|
|
136
|
+
account: Account
|
|
137
|
+
ok: bool
|
|
138
|
+
checked_at: float
|
|
139
|
+
used: int | None = None
|
|
140
|
+
limit: int | None = None
|
|
141
|
+
remaining: int | None = None
|
|
142
|
+
detail: str = ""
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def openrouter_quota(account: Account) -> Quota:
|
|
146
|
+
"""`GET /api/v1/key` -- the sibling of `check()`'s `/api/v1/auth/key`,
|
|
147
|
+
same header discipline, same never-echo-the-key rule.
|
|
148
|
+
|
|
149
|
+
Every other provider in the registry exposes nothing usable to an
|
|
150
|
+
ordinary inference key (`docs/0038` §5.3, `research/phase-10-3-model-
|
|
151
|
+
selection.md` §2.3): this is the only function in this module that
|
|
152
|
+
returns a quota NUMBER rather than a connectivity verdict, and it exists
|
|
153
|
+
only for `openrouter`. Measured live against six accounts on 2026-09-21:
|
|
154
|
+
`free_model_daily_requests: {"used": 0, "limit": 50, "remaining": 50}`
|
|
155
|
+
for every one of them.
|
|
156
|
+
|
|
157
|
+
Callers decide when this runs. Nothing in this module calls it on its
|
|
158
|
+
own, and it must never be polled -- one explicit call per refresh.
|
|
159
|
+
"""
|
|
160
|
+
now = time.time()
|
|
161
|
+
key = account.value
|
|
162
|
+
if not key:
|
|
163
|
+
return Quota(account, False, now, detail="empty")
|
|
164
|
+
|
|
165
|
+
req = urllib.request.Request(
|
|
166
|
+
"https://openrouter.ai/api/v1/key",
|
|
167
|
+
headers={"Authorization": f"Bearer {key}", "User-Agent": USER_AGENT})
|
|
168
|
+
try:
|
|
169
|
+
with urllib.request.urlopen(req, timeout=TIMEOUT) as fh:
|
|
170
|
+
body = fh.read(20_000)
|
|
171
|
+
except urllib.error.HTTPError as e:
|
|
172
|
+
detail = _safe_body(e, key) or f"HTTP {e.code}"
|
|
173
|
+
return Quota(account, False, now, detail=detail[:80])
|
|
174
|
+
except Exception as e: # noqa: BLE001
|
|
175
|
+
return Quota(account, False, now, detail=type(e).__name__)
|
|
176
|
+
|
|
177
|
+
try:
|
|
178
|
+
d = json.loads(body)
|
|
179
|
+
except Exception: # noqa: BLE001
|
|
180
|
+
return Quota(account, False, now, detail="response was not JSON")
|
|
181
|
+
|
|
182
|
+
# The sibling `/api/v1/auth/key` wraps its payload in "data" (see
|
|
183
|
+
# `_describe` above); defend against either shape rather than assume.
|
|
184
|
+
data = d.get("data", d) if isinstance(d, dict) else {}
|
|
185
|
+
fmdr = data.get("free_model_daily_requests") if isinstance(data, dict) else None
|
|
186
|
+
if not isinstance(fmdr, dict):
|
|
187
|
+
return Quota(account, False, now,
|
|
188
|
+
detail="no free_model_daily_requests in the response "
|
|
189
|
+
"(paid key, or the shape changed)")
|
|
190
|
+
return Quota(account, True, now, used=fmdr.get("used"),
|
|
191
|
+
limit=fmdr.get("limit"), remaining=fmdr.get("remaining"))
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def check_all(accounts: list[Account] | None = None,
|
|
195
|
+
workers: int = 8) -> list[Result]:
|
|
196
|
+
"""Every credential, in parallel. Order follows `all_accounts`."""
|
|
197
|
+
accts = accounts if accounts is not None else all_accounts()
|
|
198
|
+
if not accts:
|
|
199
|
+
return []
|
|
200
|
+
with ThreadPoolExecutor(max_workers=workers) as pool:
|
|
201
|
+
return list(pool.map(check, accts))
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def summarise(results: list[Result]) -> dict:
|
|
205
|
+
by = {}
|
|
206
|
+
for r in results:
|
|
207
|
+
by[r.status] = by.get(r.status, 0) + 1
|
|
208
|
+
working = [r for r in results if r.ok]
|
|
209
|
+
return {
|
|
210
|
+
"total": len(results),
|
|
211
|
+
"working": len(working),
|
|
212
|
+
"by_status": by,
|
|
213
|
+
"providers_working": len({r.account.provider.name for r in working}),
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _safe_body(e: urllib.error.HTTPError, key: str) -> str:
|
|
218
|
+
"""The error text with the credential stripped.
|
|
219
|
+
|
|
220
|
+
An error payload can quote the key that was rejected, and the whole point
|
|
221
|
+
of this module is that a key never reaches a terminal.
|
|
222
|
+
"""
|
|
223
|
+
try:
|
|
224
|
+
text = e.read(400).decode("utf-8", "replace")
|
|
225
|
+
except Exception: # noqa: BLE001
|
|
226
|
+
return ""
|
|
227
|
+
return text.replace(key, "<REDACTED>").strip()
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
NO_CREDIT = "no-credit"
|
|
231
|
+
# A paid provider deliberately not called. Not a failure -- a choice.
|
|
232
|
+
SKIPPED_PAID = "skipped"
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def check_inference(provider_name: str, allow_paid: bool = False,
|
|
236
|
+
all_models: bool = False) -> Result | None:
|
|
237
|
+
"""Can this provider actually COMPLETE, not merely authenticate?
|
|
238
|
+
|
|
239
|
+
Metadata endpoints answer "is the credential real". They do not answer "may
|
|
240
|
+
it infer", and the two came apart in practice: six Cerebras keys listed
|
|
241
|
+
models happily and every completion returned *"Payment required to access
|
|
242
|
+
this resource"*. Reporting those as `live` would be a true statement about
|
|
243
|
+
authentication and a false one about usability — the misleading-verdict
|
|
244
|
+
shape this project keeps finding (`docs/0034` §6).
|
|
245
|
+
|
|
246
|
+
One call per PROVIDER, not per key, because the tier is an account-level
|
|
247
|
+
property and 31 completions to learn six facts is not a trade worth making.
|
|
248
|
+
"""
|
|
249
|
+
import os
|
|
250
|
+
import warnings
|
|
251
|
+
|
|
252
|
+
from .providers import BY_NAME, accounts_for
|
|
253
|
+
|
|
254
|
+
p = BY_NAME.get(provider_name)
|
|
255
|
+
if p is None or not p.models:
|
|
256
|
+
return None
|
|
257
|
+
accts = accounts_for(p)
|
|
258
|
+
if not accts:
|
|
259
|
+
return None
|
|
260
|
+
|
|
261
|
+
# `docs/0002` §5: never silently spend. A diagnostic that bills you is a
|
|
262
|
+
# bad diagnostic, and the first version of this function called Anthropic
|
|
263
|
+
# without being asked. Paid providers are checked only on request.
|
|
264
|
+
if not p.free_tier and not allow_paid:
|
|
265
|
+
return Result(accts[0], SKIPPED_PAID,
|
|
266
|
+
"paid — not called. Use --check-paid to bill a few tokens.")
|
|
267
|
+
|
|
268
|
+
warnings.filterwarnings("ignore")
|
|
269
|
+
os.environ.setdefault("LITELLM_LOG", "ERROR")
|
|
270
|
+
|
|
271
|
+
# Ask each ACCOUNT until one serves, and stop there.
|
|
272
|
+
#
|
|
273
|
+
# This used to call `accts[0]` once and rule for the provider, on the
|
|
274
|
+
# reasoning that "the tier is an account-level property" -- which is the
|
|
275
|
+
# argument AGAINST doing that, not for it. A daily cap is account-level
|
|
276
|
+
# too, so one capped key spoke for five working ones and `--verify`
|
|
277
|
+
# dropped all eighteen of that provider's deployments (`docs/0039`).
|
|
278
|
+
#
|
|
279
|
+
# The cost concern behind the original was real and is preserved: a
|
|
280
|
+
# healthy provider still costs exactly one call, because the loop stops
|
|
281
|
+
# at the first success. Only a provider that is actually failing pays for
|
|
282
|
+
# more, which is when you want to know.
|
|
283
|
+
#
|
|
284
|
+
# And ask each MODEL until one serves, but only past a rejected model id.
|
|
285
|
+
# Testing `models[0]` alone meant one model leaving the catalogue made a
|
|
286
|
+
# working key read "0 provider(s) can serve": `nex-agi/nex-n2.5-pro:free`
|
|
287
|
+
# was chosen on 2026-09-21 and gone by 2026-10-01 (`docs/0044` N6). A cap
|
|
288
|
+
# or a bad key is about the ACCOUNT, so those stop the model loop -- the
|
|
289
|
+
# next model would only spend a request to learn the same thing.
|
|
290
|
+
try:
|
|
291
|
+
import litellm
|
|
292
|
+
except ImportError:
|
|
293
|
+
return Result(accts[0], UNREACHABLE,
|
|
294
|
+
'litellm is not installed: pip install -e ".[openhands]"')
|
|
295
|
+
litellm.suppress_debug_info = True
|
|
296
|
+
|
|
297
|
+
best: Result | None = None
|
|
298
|
+
gone: list[str] = []
|
|
299
|
+
for acct in accts:
|
|
300
|
+
for model in p.models:
|
|
301
|
+
if model in gone:
|
|
302
|
+
continue
|
|
303
|
+
try:
|
|
304
|
+
litellm.completion(model=f"{p.prefix}{model}", max_tokens=4,
|
|
305
|
+
api_key=acct.value,
|
|
306
|
+
messages=[{"role": "user", "content": "ok"}])
|
|
307
|
+
except Exception as e: # noqa: BLE001
|
|
308
|
+
r = _classify_inference(e, acct, p, model)
|
|
309
|
+
best = _worse_of(best, r)
|
|
310
|
+
if r.detail.startswith(MODEL_GONE):
|
|
311
|
+
gone.append(model)
|
|
312
|
+
continue
|
|
313
|
+
break
|
|
314
|
+
if all_models:
|
|
315
|
+
# A pool is built from EVERY model id, so every id needs one
|
|
316
|
+
# completion, not just the first that answers. Stopping at the
|
|
317
|
+
# first success left a dead id later in the list in the pool,
|
|
318
|
+
# and its 404 ended a live run (`docs/0047`; `docs/0042` I-11).
|
|
319
|
+
gone += _dead_models(litellm, acct, p, after=model, gone=gone)
|
|
320
|
+
note = f"inference ok ({model})"
|
|
321
|
+
if acct is not accts[0]:
|
|
322
|
+
note += f" via {acct.label}"
|
|
323
|
+
if gone:
|
|
324
|
+
note += f"; no longer served: {', '.join(gone)}"
|
|
325
|
+
return Result(acct, LIVE, note, gone=tuple(gone), model=model)
|
|
326
|
+
assert best is not None
|
|
327
|
+
if gone and len(gone) == len(p.models):
|
|
328
|
+
return Result(best.account, UNREACHABLE,
|
|
329
|
+
f"the key may be fine, but none of the {len(gone)} "
|
|
330
|
+
f"registry model(s) is served: {', '.join(gone)}",
|
|
331
|
+
gone=tuple(gone))
|
|
332
|
+
best.gone = tuple(gone)
|
|
333
|
+
return best
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
#: Most usable first. A LIMITED provider is still in the pool; a NO_CREDIT one
|
|
337
|
+
#: is not, so collapsing the two loses a working provider.
|
|
338
|
+
_RANK = (LIMITED, NO_CREDIT, UNREACHABLE)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def _worse_of(a: "Result | None", b: "Result") -> "Result":
|
|
342
|
+
"""Keep the most *encouraging* verdict seen across a provider's accounts.
|
|
343
|
+
|
|
344
|
+
If one key is rate limited and another has no credit, the provider is
|
|
345
|
+
rate limited -- the capped key will come back. Reporting the bleaker of
|
|
346
|
+
the two would drop a provider that works tomorrow.
|
|
347
|
+
"""
|
|
348
|
+
if a is None:
|
|
349
|
+
return b
|
|
350
|
+
rank = {s: i for i, s in enumerate(_RANK)}
|
|
351
|
+
return a if rank.get(a.status, 9) <= rank.get(b.status, 9) else b
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _dead_models(litellm, acct, p, after: str, gone: list[str]) -> list[str]:
|
|
355
|
+
"""Every registry model after `after` that this working key cannot call.
|
|
356
|
+
|
|
357
|
+
Only a rejected model id counts as dead. A rate limit or an overload says
|
|
358
|
+
nothing about the id, so that model stays in the pool -- dropping it would
|
|
359
|
+
make a busy minute look like a shrinking catalogue.
|
|
360
|
+
"""
|
|
361
|
+
dead: list[str] = []
|
|
362
|
+
for model in p.models[p.models.index(after) + 1:]:
|
|
363
|
+
if model in gone:
|
|
364
|
+
continue
|
|
365
|
+
try:
|
|
366
|
+
litellm.completion(model=f"{p.prefix}{model}", max_tokens=4,
|
|
367
|
+
api_key=acct.value,
|
|
368
|
+
messages=[{"role": "user", "content": "ok"}])
|
|
369
|
+
except Exception as e: # noqa: BLE001
|
|
370
|
+
if _classify_inference(e, acct, p, model).detail.startswith(MODEL_GONE):
|
|
371
|
+
dead.append(model)
|
|
372
|
+
return dead
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
MODEL_GONE = "model id rejected"
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def _classify_inference(e: Exception, acct, p, model: str = "") -> "Result":
|
|
379
|
+
"""Why a completion failed, from the provider's own words.
|
|
380
|
+
|
|
381
|
+
**Rate limiting is checked before payment, and the order is the point.**
|
|
382
|
+
OpenRouter's daily-cap error reads *"Rate limit exceeded:
|
|
383
|
+
free-models-per-day. Add 10 credits to unlock 1000 free model requests
|
|
384
|
+
per day"* -- it contains the word `credits`, so a payment-first match
|
|
385
|
+
classified a temporary cap as a billing failure and removed the provider
|
|
386
|
+
from the pool for the rest of the day. A fifth way a check can lie
|
|
387
|
+
(`docs/0034`).
|
|
388
|
+
"""
|
|
389
|
+
msg = str(e).replace(acct.value, "<REDACTED>")
|
|
390
|
+
low = msg.lower()
|
|
391
|
+
if "rate limit" in low or "429" in msg or "quota" in low:
|
|
392
|
+
return Result(acct, LIMITED, "rate limited (usable later)")
|
|
393
|
+
if "payment" in low or "credit" in low or "billing" in low:
|
|
394
|
+
return Result(acct, NO_CREDIT,
|
|
395
|
+
"authenticates, but inference needs payment")
|
|
396
|
+
if "not found" in low or "404" in msg or "not a valid model" in low:
|
|
397
|
+
return Result(acct, UNREACHABLE,
|
|
398
|
+
f"{MODEL_GONE}: {model or p.models[0]}")
|
|
399
|
+
return Result(acct, UNREACHABLE, msg.splitlines()[0][:80])
|