handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,399 @@
1
+ r"""Does each key actually work? Checked without spending a token.
2
+
3
+ Every provider exposes a metadata endpoint that authenticates the caller and
4
+ returns no completion — listing models, or reporting the key's own status. So
5
+ 31 credentials can be validated for **zero model tokens and zero quota**, which
6
+ matters because the alternative (a one-token completion each) would burn 31 of
7
+ a 50-request daily allowance just to ask whether the keys are real.
8
+
9
+ Two rules this module keeps:
10
+
11
+ * **The key travels in a header, never a URL.** Gemini accepts
12
+ `?key=` and that would put a live credential into proxy logs, shell history
13
+ and any error message containing the URL. `x-goog-api-key` does the same job
14
+ and leaves no trace.
15
+ * **A key is never echoed, including in failures.** An error body can contain
16
+ the key that was rejected, so responses are reduced to a status and a short
17
+ reason before they go anywhere near a terminal.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import json
22
+ import time
23
+ import urllib.error
24
+ import urllib.request
25
+ from concurrent.futures import ThreadPoolExecutor
26
+ from dataclasses import dataclass
27
+
28
+ from .providers import Account, all_accounts
29
+
30
+ TIMEOUT = 20
31
+
32
+ # Identify the client honestly. Groq and Cerebras sit behind Cloudflare, which
33
+ # refuses Python's default `User-Agent: Python-urllib/3.x` with error 1010 --
34
+ # a 403 that looks exactly like a rejected credential. Twelve perfectly good
35
+ # keys were reported as broken before this line existed (`docs/0034` §3).
36
+ USER_AGENT = "agentctl/0.2 (+https://github.com/csdeepak/HandCode)"
37
+
38
+ LIVE, BAD_KEY, LIMITED, UNREACHABLE = "live", "rejected", "limited", "unreachable"
39
+
40
+
41
+ @dataclass
42
+ class Result:
43
+ account: Account
44
+ status: str
45
+ detail: str = ""
46
+ #: Registry model ids the provider rejected as unknown. A catalogue entry
47
+ #: that has disappeared is a fact about the registry, not about the key,
48
+ #: and `proxy` leaves these out of the pool (`docs/0044` N6).
49
+ gone: tuple[str, ...] = ()
50
+ #: The registry model id that actually served, when one did. `init`
51
+ #: records it as the user's default: proven by a completion, not assumed.
52
+ model: str = ""
53
+
54
+ @property
55
+ def ok(self) -> bool:
56
+ # A rate-limited key is a WORKING key with no allowance left. Reporting
57
+ # it as broken would send someone to rotate a credential that is fine.
58
+ return self.status in (LIVE, LIMITED)
59
+
60
+
61
+ # (url, header builder). Metadata only -- none of these generate tokens.
62
+ ENDPOINTS: dict[str, tuple[str, callable]] = {
63
+ "openrouter": ("https://openrouter.ai/api/v1/auth/key",
64
+ lambda k: {"Authorization": f"Bearer {k}"}),
65
+ "gemini": ("https://generativelanguage.googleapis.com/v1beta/models",
66
+ # Header, not `?key=`: a URL leaks into logs and error text.
67
+ lambda k: {"x-goog-api-key": k}),
68
+ "mistral": ("https://api.mistral.ai/v1/models",
69
+ lambda k: {"Authorization": f"Bearer {k}"}),
70
+ "cerebras": ("https://api.cerebras.ai/v1/models",
71
+ lambda k: {"Authorization": f"Bearer {k}"}),
72
+ "groq": ("https://api.groq.com/openai/v1/models",
73
+ lambda k: {"Authorization": f"Bearer {k}"}),
74
+ "anthropic": ("https://api.anthropic.com/v1/models",
75
+ lambda k: {"x-api-key": k, "anthropic-version": "2023-06-01"}),
76
+ "openai": ("https://api.openai.com/v1/models",
77
+ lambda k: {"Authorization": f"Bearer {k}"}),
78
+ }
79
+
80
+
81
+ def check(account: Account) -> Result:
82
+ """One credential. Never raises, never echoes the key."""
83
+ spec = ENDPOINTS.get(account.provider.name)
84
+ if spec is None:
85
+ return Result(account, UNREACHABLE, "no metadata endpoint known")
86
+
87
+ url, headers = spec
88
+ key = account.value
89
+ if not key:
90
+ return Result(account, BAD_KEY, "empty")
91
+
92
+ req = urllib.request.Request(url, headers={**headers(key),
93
+ "User-Agent": USER_AGENT})
94
+ try:
95
+ with urllib.request.urlopen(req, timeout=TIMEOUT) as fh:
96
+ body = fh.read(20_000)
97
+ return Result(account, LIVE, _describe(account, body))
98
+ except urllib.error.HTTPError as e:
99
+ body = _safe_body(e, key)
100
+ # 403 is NOT automatically a bad key. Cloudflare returns 403 with
101
+ # "error code: 1010" when it dislikes the client, which is a transport
102
+ # problem wearing an authentication problem's clothes.
103
+ if "error code: 101" in body or "cloudflare" in body.lower():
104
+ return Result(account, UNREACHABLE,
105
+ f"HTTP {e.code} from a CDN, not the API "
106
+ f"({body[:40]}) — the key was never checked")
107
+ if e.code == 401:
108
+ return Result(account, BAD_KEY, "HTTP 401 unauthorized")
109
+ if e.code == 403:
110
+ return Result(account, BAD_KEY, f"HTTP 403 forbidden ({body[:60]})")
111
+ if e.code == 429:
112
+ return Result(account, LIMITED, "rate limited (the key is valid)")
113
+ return Result(account, UNREACHABLE, f"HTTP {e.code}")
114
+ except Exception as e: # noqa: BLE001
115
+ return Result(account, UNREACHABLE, type(e).__name__)
116
+
117
+
118
+ def _describe(account: Account, body: bytes) -> str:
119
+ """Something useful from the response. Never the key."""
120
+ try:
121
+ d = json.loads(body)
122
+ except Exception: # noqa: BLE001
123
+ return "ok"
124
+ if account.provider.name == "openrouter":
125
+ data = d.get("data") or {}
126
+ return "free tier" if data.get("is_free_tier") else "paid"
127
+ for field in ("data", "models"):
128
+ if isinstance(d.get(field), list):
129
+ return f"{len(d[field])} models"
130
+ return "ok"
131
+
132
+
133
+ # ── the one honest quota number (`research/phase-10-3` §2.3, §5.3) ─────
134
+ @dataclass
135
+ class Quota:
136
+ account: Account
137
+ ok: bool
138
+ checked_at: float
139
+ used: int | None = None
140
+ limit: int | None = None
141
+ remaining: int | None = None
142
+ detail: str = ""
143
+
144
+
145
+ def openrouter_quota(account: Account) -> Quota:
146
+ """`GET /api/v1/key` -- the sibling of `check()`'s `/api/v1/auth/key`,
147
+ same header discipline, same never-echo-the-key rule.
148
+
149
+ Every other provider in the registry exposes nothing usable to an
150
+ ordinary inference key (`docs/0038` §5.3, `research/phase-10-3-model-
151
+ selection.md` §2.3): this is the only function in this module that
152
+ returns a quota NUMBER rather than a connectivity verdict, and it exists
153
+ only for `openrouter`. Measured live against six accounts on 2026-09-21:
154
+ `free_model_daily_requests: {"used": 0, "limit": 50, "remaining": 50}`
155
+ for every one of them.
156
+
157
+ Callers decide when this runs. Nothing in this module calls it on its
158
+ own, and it must never be polled -- one explicit call per refresh.
159
+ """
160
+ now = time.time()
161
+ key = account.value
162
+ if not key:
163
+ return Quota(account, False, now, detail="empty")
164
+
165
+ req = urllib.request.Request(
166
+ "https://openrouter.ai/api/v1/key",
167
+ headers={"Authorization": f"Bearer {key}", "User-Agent": USER_AGENT})
168
+ try:
169
+ with urllib.request.urlopen(req, timeout=TIMEOUT) as fh:
170
+ body = fh.read(20_000)
171
+ except urllib.error.HTTPError as e:
172
+ detail = _safe_body(e, key) or f"HTTP {e.code}"
173
+ return Quota(account, False, now, detail=detail[:80])
174
+ except Exception as e: # noqa: BLE001
175
+ return Quota(account, False, now, detail=type(e).__name__)
176
+
177
+ try:
178
+ d = json.loads(body)
179
+ except Exception: # noqa: BLE001
180
+ return Quota(account, False, now, detail="response was not JSON")
181
+
182
+ # The sibling `/api/v1/auth/key` wraps its payload in "data" (see
183
+ # `_describe` above); defend against either shape rather than assume.
184
+ data = d.get("data", d) if isinstance(d, dict) else {}
185
+ fmdr = data.get("free_model_daily_requests") if isinstance(data, dict) else None
186
+ if not isinstance(fmdr, dict):
187
+ return Quota(account, False, now,
188
+ detail="no free_model_daily_requests in the response "
189
+ "(paid key, or the shape changed)")
190
+ return Quota(account, True, now, used=fmdr.get("used"),
191
+ limit=fmdr.get("limit"), remaining=fmdr.get("remaining"))
192
+
193
+
194
+ def check_all(accounts: list[Account] | None = None,
195
+ workers: int = 8) -> list[Result]:
196
+ """Every credential, in parallel. Order follows `all_accounts`."""
197
+ accts = accounts if accounts is not None else all_accounts()
198
+ if not accts:
199
+ return []
200
+ with ThreadPoolExecutor(max_workers=workers) as pool:
201
+ return list(pool.map(check, accts))
202
+
203
+
204
+ def summarise(results: list[Result]) -> dict:
205
+ by = {}
206
+ for r in results:
207
+ by[r.status] = by.get(r.status, 0) + 1
208
+ working = [r for r in results if r.ok]
209
+ return {
210
+ "total": len(results),
211
+ "working": len(working),
212
+ "by_status": by,
213
+ "providers_working": len({r.account.provider.name for r in working}),
214
+ }
215
+
216
+
217
+ def _safe_body(e: urllib.error.HTTPError, key: str) -> str:
218
+ """The error text with the credential stripped.
219
+
220
+ An error payload can quote the key that was rejected, and the whole point
221
+ of this module is that a key never reaches a terminal.
222
+ """
223
+ try:
224
+ text = e.read(400).decode("utf-8", "replace")
225
+ except Exception: # noqa: BLE001
226
+ return ""
227
+ return text.replace(key, "<REDACTED>").strip()
228
+
229
+
230
+ NO_CREDIT = "no-credit"
231
+ # A paid provider deliberately not called. Not a failure -- a choice.
232
+ SKIPPED_PAID = "skipped"
233
+
234
+
235
+ def check_inference(provider_name: str, allow_paid: bool = False,
236
+ all_models: bool = False) -> Result | None:
237
+ """Can this provider actually COMPLETE, not merely authenticate?
238
+
239
+ Metadata endpoints answer "is the credential real". They do not answer "may
240
+ it infer", and the two came apart in practice: six Cerebras keys listed
241
+ models happily and every completion returned *"Payment required to access
242
+ this resource"*. Reporting those as `live` would be a true statement about
243
+ authentication and a false one about usability — the misleading-verdict
244
+ shape this project keeps finding (`docs/0034` §6).
245
+
246
+ One call per PROVIDER, not per key, because the tier is an account-level
247
+ property and 31 completions to learn six facts is not a trade worth making.
248
+ """
249
+ import os
250
+ import warnings
251
+
252
+ from .providers import BY_NAME, accounts_for
253
+
254
+ p = BY_NAME.get(provider_name)
255
+ if p is None or not p.models:
256
+ return None
257
+ accts = accounts_for(p)
258
+ if not accts:
259
+ return None
260
+
261
+ # `docs/0002` §5: never silently spend. A diagnostic that bills you is a
262
+ # bad diagnostic, and the first version of this function called Anthropic
263
+ # without being asked. Paid providers are checked only on request.
264
+ if not p.free_tier and not allow_paid:
265
+ return Result(accts[0], SKIPPED_PAID,
266
+ "paid — not called. Use --check-paid to bill a few tokens.")
267
+
268
+ warnings.filterwarnings("ignore")
269
+ os.environ.setdefault("LITELLM_LOG", "ERROR")
270
+
271
+ # Ask each ACCOUNT until one serves, and stop there.
272
+ #
273
+ # This used to call `accts[0]` once and rule for the provider, on the
274
+ # reasoning that "the tier is an account-level property" -- which is the
275
+ # argument AGAINST doing that, not for it. A daily cap is account-level
276
+ # too, so one capped key spoke for five working ones and `--verify`
277
+ # dropped all eighteen of that provider's deployments (`docs/0039`).
278
+ #
279
+ # The cost concern behind the original was real and is preserved: a
280
+ # healthy provider still costs exactly one call, because the loop stops
281
+ # at the first success. Only a provider that is actually failing pays for
282
+ # more, which is when you want to know.
283
+ #
284
+ # And ask each MODEL until one serves, but only past a rejected model id.
285
+ # Testing `models[0]` alone meant one model leaving the catalogue made a
286
+ # working key read "0 provider(s) can serve": `nex-agi/nex-n2.5-pro:free`
287
+ # was chosen on 2026-09-21 and gone by 2026-10-01 (`docs/0044` N6). A cap
288
+ # or a bad key is about the ACCOUNT, so those stop the model loop -- the
289
+ # next model would only spend a request to learn the same thing.
290
+ try:
291
+ import litellm
292
+ except ImportError:
293
+ return Result(accts[0], UNREACHABLE,
294
+ 'litellm is not installed: pip install -e ".[openhands]"')
295
+ litellm.suppress_debug_info = True
296
+
297
+ best: Result | None = None
298
+ gone: list[str] = []
299
+ for acct in accts:
300
+ for model in p.models:
301
+ if model in gone:
302
+ continue
303
+ try:
304
+ litellm.completion(model=f"{p.prefix}{model}", max_tokens=4,
305
+ api_key=acct.value,
306
+ messages=[{"role": "user", "content": "ok"}])
307
+ except Exception as e: # noqa: BLE001
308
+ r = _classify_inference(e, acct, p, model)
309
+ best = _worse_of(best, r)
310
+ if r.detail.startswith(MODEL_GONE):
311
+ gone.append(model)
312
+ continue
313
+ break
314
+ if all_models:
315
+ # A pool is built from EVERY model id, so every id needs one
316
+ # completion, not just the first that answers. Stopping at the
317
+ # first success left a dead id later in the list in the pool,
318
+ # and its 404 ended a live run (`docs/0047`; `docs/0042` I-11).
319
+ gone += _dead_models(litellm, acct, p, after=model, gone=gone)
320
+ note = f"inference ok ({model})"
321
+ if acct is not accts[0]:
322
+ note += f" via {acct.label}"
323
+ if gone:
324
+ note += f"; no longer served: {', '.join(gone)}"
325
+ return Result(acct, LIVE, note, gone=tuple(gone), model=model)
326
+ assert best is not None
327
+ if gone and len(gone) == len(p.models):
328
+ return Result(best.account, UNREACHABLE,
329
+ f"the key may be fine, but none of the {len(gone)} "
330
+ f"registry model(s) is served: {', '.join(gone)}",
331
+ gone=tuple(gone))
332
+ best.gone = tuple(gone)
333
+ return best
334
+
335
+
336
+ #: Most usable first. A LIMITED provider is still in the pool; a NO_CREDIT one
337
+ #: is not, so collapsing the two loses a working provider.
338
+ _RANK = (LIMITED, NO_CREDIT, UNREACHABLE)
339
+
340
+
341
+ def _worse_of(a: "Result | None", b: "Result") -> "Result":
342
+ """Keep the most *encouraging* verdict seen across a provider's accounts.
343
+
344
+ If one key is rate limited and another has no credit, the provider is
345
+ rate limited -- the capped key will come back. Reporting the bleaker of
346
+ the two would drop a provider that works tomorrow.
347
+ """
348
+ if a is None:
349
+ return b
350
+ rank = {s: i for i, s in enumerate(_RANK)}
351
+ return a if rank.get(a.status, 9) <= rank.get(b.status, 9) else b
352
+
353
+
354
+ def _dead_models(litellm, acct, p, after: str, gone: list[str]) -> list[str]:
355
+ """Every registry model after `after` that this working key cannot call.
356
+
357
+ Only a rejected model id counts as dead. A rate limit or an overload says
358
+ nothing about the id, so that model stays in the pool -- dropping it would
359
+ make a busy minute look like a shrinking catalogue.
360
+ """
361
+ dead: list[str] = []
362
+ for model in p.models[p.models.index(after) + 1:]:
363
+ if model in gone:
364
+ continue
365
+ try:
366
+ litellm.completion(model=f"{p.prefix}{model}", max_tokens=4,
367
+ api_key=acct.value,
368
+ messages=[{"role": "user", "content": "ok"}])
369
+ except Exception as e: # noqa: BLE001
370
+ if _classify_inference(e, acct, p, model).detail.startswith(MODEL_GONE):
371
+ dead.append(model)
372
+ return dead
373
+
374
+
375
+ MODEL_GONE = "model id rejected"
376
+
377
+
378
+ def _classify_inference(e: Exception, acct, p, model: str = "") -> "Result":
379
+ """Why a completion failed, from the provider's own words.
380
+
381
+ **Rate limiting is checked before payment, and the order is the point.**
382
+ OpenRouter's daily-cap error reads *"Rate limit exceeded:
383
+ free-models-per-day. Add 10 credits to unlock 1000 free model requests
384
+ per day"* -- it contains the word `credits`, so a payment-first match
385
+ classified a temporary cap as a billing failure and removed the provider
386
+ from the pool for the rest of the day. A fifth way a check can lie
387
+ (`docs/0034`).
388
+ """
389
+ msg = str(e).replace(acct.value, "<REDACTED>")
390
+ low = msg.lower()
391
+ if "rate limit" in low or "429" in msg or "quota" in low:
392
+ return Result(acct, LIMITED, "rate limited (usable later)")
393
+ if "payment" in low or "credit" in low or "billing" in low:
394
+ return Result(acct, NO_CREDIT,
395
+ "authenticates, but inference needs payment")
396
+ if "not found" in low or "404" in msg or "not a valid model" in low:
397
+ return Result(acct, UNREACHABLE,
398
+ f"{MODEL_GONE}: {model or p.models[0]}")
399
+ return Result(acct, UNREACHABLE, msg.splitlines()[0][:80])