handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,536 @@
1
+ r"""Generate a LiteLLM proxy config from the keys that actually exist.
2
+
3
+ The project's charter is surviving a rate limit without work stopping. That
4
+ requires somewhere to fail over TO, and a config listing providers you have no
5
+ key for does not start -- litellm resolves `os.environ/X` at load and refuses.
6
+ So the config is generated from the environment rather than hand-written, and
7
+ adding a provider means exporting a key, not editing YAML.
8
+
9
+ ## What failover can and cannot do for you
10
+
11
+ Two different failures, and a pool only fixes one of them with one account:
12
+
13
+ | failure | one OpenRouter key | plus a second provider |
14
+ |---|---|---|
15
+ | transient overload (*"Upstream error from Nvidia"*) | **fixed** — another model, another upstream | fixed |
16
+ | per-model rate limit | **fixed** | fixed |
17
+ | free-models-per-**day** cap | **not fixed** — the cap is account-wide | fixed |
18
+
19
+ Saying a pool fixes the daily cap would be the comfortable answer and the
20
+ wrong one: OpenRouter counts free requests per account, not per model. The
21
+ `:free` models below buy resilience against the first two rows, which is what
22
+ actually interrupted a real run, and nothing against the third.
23
+
24
+ ## Fallbacks are ordered, and the order is a cost decision
25
+
26
+ `fallbacks` sends the next attempt to the next entry. Free models come first
27
+ and paid providers last, so an outage degrades toward *slower*, never silently
28
+ toward *billed*. `docs/0002` §5: never silently spend.
29
+ """
30
+ from __future__ import annotations
31
+
32
+ import os
33
+ import re
34
+ from pathlib import Path
35
+
36
+ # (env var, litellm model, short name, is_free). Order is fallback order:
37
+ # free first, paid last, so degrading never silently costs money.
38
+ POOL = "pool"
39
+ # A separate model group, never an automatic fallback target (`docs/0002` §5).
40
+ PAID = "paid"
41
+ # `pool-openrouter`, `pool-gemini`, ... One source, deliberately narrower than
42
+ # `pool`. The prefix is what makes a source group recognisable as one rather
43
+ # than as some unrelated model group someone hand-added (`docs/0039`).
44
+ SOURCE_PREFIX = "pool-"
45
+
46
+
47
+ def available(env: dict | None = None) -> list[tuple[str, str, str, bool]]:
48
+ """One deployment per ACCOUNT per model: (env, model, id, is_free).
49
+
50
+ The cross product is the point. Two OpenRouter accounts times three models
51
+ is six deployments, and when one account hits its daily cap the other three
52
+ keep serving — which a single-account pool cannot do however many models it
53
+ lists (`docs/0033`).
54
+
55
+ Ordered free-first, paid-last, because that is also the fallback order: an
56
+ outage degrades toward slower, never silently toward billed (`docs/0002`
57
+ §5).
58
+ """
59
+ from .providers import all_accounts
60
+
61
+ out = []
62
+ seen: dict[str, int] = {}
63
+ for acct in all_accounts(env):
64
+ n = seen[acct.provider.name] = seen.get(acct.provider.name, 0) + 1
65
+ for i, model in enumerate(acct.provider.models):
66
+ # `provider-aN-mI`, both parts always present and labelled. The
67
+ # obvious scheme -- label plus model index -- produced
68
+ # `openrouter-2` (account 1, model 2) alongside `openrouter-2-0`
69
+ # (account 2, model 0): unique, and unreadable in a fallback map
70
+ # exactly when you are debugging one. Nothing is abbreviated
71
+ # either, since `openrouter` and `openai` share a prefix.
72
+ short = f"{acct.provider.name}-a{n}-m{i}"
73
+ out.append((acct.env, f"{acct.provider.prefix}{model}", short,
74
+ acct.provider.free_tier))
75
+ return out
76
+
77
+
78
+ # The optional `-only` suffix marks a source-group copy. It is the SAME
79
+ # deployment under a narrower name, so it parses to the same triple --
80
+ # a caller counting deployments must not count it twice (`docs/0039`).
81
+ _DEPLOYMENT_ID = re.compile(r"^([a-z][a-z0-9]*)-a(\d+)-m(\d+)(?:-only)?$")
82
+
83
+
84
+ def parse_deployment_id(short: str) -> tuple[str, int, int] | None:
85
+ """The inverse of the `short` id `available()` builds above: one place
86
+ defines the format, this parses it, and a round-trip test ties them
87
+ together so they cannot drift apart silently.
88
+
89
+ Returns `(provider_name, account_ordinal, model_index)`, 1-indexed for
90
+ the account to match the id text itself (`a1` -> `1`). `None` for
91
+ anything that does not match -- callers must not guess at a foreign or
92
+ malformed id, only count it as unrecognised.
93
+ """
94
+ m = _DEPLOYMENT_ID.match(short)
95
+ if not m:
96
+ return None
97
+ provider, n, i = m.groups()
98
+ return provider, int(n), int(i)
99
+
100
+
101
+ # Kept for callers that want the catalogue rather than what is configured.
102
+ def _candidates() -> list[tuple[str, str, str, bool]]:
103
+ from .providers import PROVIDERS as _P
104
+ return [(p.key, f"{p.prefix}{m}", f"{p.name}-{i}", p.free_tier)
105
+ for p in sorted(_P, key=lambda x: not x.free_tier)
106
+ for i, m in enumerate(p.models)]
107
+
108
+
109
+ CANDIDATES: list[tuple[str, str, str, bool]] = _candidates()
110
+
111
+
112
+ def accounts(entries: list[tuple[str, str, str, bool]]) -> set[str]:
113
+ """Distinct credentials behind the pool, by environment variable.
114
+
115
+ Not providers: two keys at one provider are two accounts with two quotas,
116
+ and counting them as one would under-report the failover you actually have.
117
+ """
118
+ return {c[0] for c in entries}
119
+
120
+
121
+ def _by_source(entries: list[tuple[str, str, str, bool]]) -> dict[str, list]:
122
+ """Free deployments grouped by the provider serving them.
123
+
124
+ Paid deployments are left out: `paid` is already a group you ask for by
125
+ name, and a source group that silently included it would turn "route to
126
+ one provider" into "spend money" (`docs/0002` §5).
127
+ """
128
+ out: dict[str, list] = {}
129
+ for e in entries:
130
+ if not e[3]:
131
+ continue
132
+ parsed = parse_deployment_id(e[2])
133
+ if parsed is None:
134
+ continue
135
+ out.setdefault(parsed[0], []).append(e)
136
+ return out
137
+
138
+
139
+ def sources(env: dict | None = None) -> list[dict]:
140
+ """What a source picker needs, and the cost of each choice.
141
+
142
+ The cost is not decoration. `pool` is 48 deployments across 24 accounts;
143
+ asking for one source can leave you with six behind a single account-wide
144
+ daily cap, which is precisely the failure the multi-account design exists
145
+ to escape (`docs/0033`). A picker that shows only the names is offering a
146
+ downgrade without saying so, so every row carries what it gives up
147
+ (`research/phase-10-3-model-selection.md` §6.5).
148
+
149
+ Ordered widest-first, so the safe choice reads first.
150
+ """
151
+ entries = available(env)
152
+ total = len([e for e in entries if e[3]])
153
+ rows = []
154
+ for name, mine in _by_source(entries).items():
155
+ from .providers import BY_NAME
156
+ prov = BY_NAME.get(name)
157
+ n_acct = len({e[0] for e in mine})
158
+ # Keys are not allowances. Gemini bills per Google project, so six
159
+ # keys there are one quota -- and a picker that offered "6 accounts"
160
+ # would be selling failover this source does not have.
161
+ n_quota = n_acct if (prov is None or prov.quota_per_key) else min(n_acct, 1)
162
+ rows.append({
163
+ "source": name,
164
+ "group": f"{SOURCE_PREFIX}{name}",
165
+ "deployments": len(mine),
166
+ "accounts": n_acct,
167
+ "quotas": n_quota,
168
+ "models": sorted({e[1] for e in mine}),
169
+ "gives_up": total - len(mine),
170
+ })
171
+ rows.sort(key=lambda r: (-r["deployments"], r["source"]))
172
+ return rows
173
+
174
+
175
+ def verified_providers(allow_paid: bool = False) -> tuple[set[str], dict]:
176
+ """Which providers can actually complete a request right now.
177
+
178
+ Authenticating is not the same as being allowed to infer: six Cerebras
179
+ keys list models happily and every completion returns "Payment required".
180
+ Those deployments sit in the pool failing ~11% of requests with an HTTP 402
181
+ that litellm does not treat as retryable, so a pool built from credentials
182
+ is worse than one built from what works (`docs/0034` §7).
183
+ """
184
+ from .probe import LIMITED, LIVE, check_inference
185
+ from .providers import PROVIDERS
186
+
187
+ ok, report = set(), {}
188
+ for p in PROVIDERS:
189
+ r = check_inference(p.name, allow_paid=allow_paid, all_models=True)
190
+ if r is None:
191
+ continue
192
+ report[p.name] = r
193
+ if r.status in (LIVE, LIMITED):
194
+ ok.add(p.name)
195
+ return ok, report
196
+
197
+
198
+ def _rejected(env_var: str) -> list[str]:
199
+ """`additional_drop_params` for a provider that 400s on a forwarded param."""
200
+ from .providers import BY_KEY
201
+ reject = BY_KEY[env_var.split("_API_KEY")[0] + "_API_KEY"].reject_params
202
+ return [f" additional_drop_params: [{', '.join(reject)}]"] if reject else []
203
+
204
+
205
+ def build(env: dict | None = None, telemetry: str | Path | None = None,
206
+ only: set[str] | None = None,
207
+ drop_models: set[str] = frozenset()) -> str:
208
+ """The proxy config, as YAML text. `only` restricts it to named providers.
209
+
210
+ `drop_models` holds full model strings verification found the provider no
211
+ longer serves. Left in, each is a deployment that 404s on every call.
212
+ """
213
+ entries = [e for e in available(env) if e[1] not in drop_models]
214
+ if only is not None:
215
+ from .providers import BY_KEY
216
+ entries = [e for e in entries
217
+ if BY_KEY[e[0].split("_API_KEY")[0] + "_API_KEY"].name in only]
218
+ if not entries:
219
+ raise SystemExit("no provider passed verification -- nothing to route to")
220
+ if not entries:
221
+ raise SystemExit(
222
+ "no provider keys in the environment — nothing to route to.\n"
223
+ " set at least one, e.g. OPENROUTER_API_KEY")
224
+
225
+ lines = [
226
+ "# GENERATED by `agentctl proxy` from the keys in your environment.",
227
+ "# Re-run it after exporting a new provider key; do not hand-edit.",
228
+ "#",
229
+ f"# {len(entries)} deployment(s) across {len(accounts(entries))} "
230
+ f"account(s).",
231
+ "",
232
+ "model_list:",
233
+ ]
234
+ for env_var, model, short, is_free in entries:
235
+ lines += [
236
+ f" - model_name: {POOL if is_free else PAID}",
237
+ " litellm_params:",
238
+ f" model: {model}",
239
+ f" api_key: os.environ/{env_var}",
240
+ *_rejected(env_var),
241
+ " model_info:",
242
+ f" id: {short}",
243
+ f" free: {str(is_free).lower()}",
244
+ ]
245
+
246
+ # ── the source groups ───────────────────────────────────────────────
247
+ #
248
+ # Everything above is one group, `pool`, and that is what makes failover
249
+ # work. These add a second, narrower way to ask: `pool-openrouter` routes
250
+ # only to OpenRouter.
251
+ #
252
+ # It is a duplicated entry, not a second deployment -- the same key, the
253
+ # same model, reachable under two names. The id carries `-only` so it
254
+ # stays unique, because a duplicate id would send a fallback to whichever
255
+ # entry litellm resolved first (`docs/0032`).
256
+ #
257
+ # The cost is real and belongs next to the choice: asking for one source
258
+ # narrows the pool to that source's share, and an account-wide daily cap
259
+ # then has nothing to fail over to. `agentctl models` prints what each
260
+ # choice leaves before you make it (`research/phase-10-3` §6.5).
261
+ by_source = _by_source(entries)
262
+ if len(by_source) > 1:
263
+ lines += ["", " # ── source groups: narrower on purpose ──"]
264
+ for source in sorted(by_source):
265
+ group = f"{SOURCE_PREFIX}{source}"
266
+ mine = by_source[source]
267
+ lines += [
268
+ f" # {group}: {len(mine)} deployment(s), "
269
+ f"{len({e[0] for e in mine})} account(s). "
270
+ f"Choosing it gives up the other "
271
+ f"{len(entries) - len(mine)}.",
272
+ ]
273
+ for env_var, model, short, is_free in mine:
274
+ lines += [
275
+ f" - model_name: {group}",
276
+ " litellm_params:",
277
+ f" model: {model}",
278
+ f" api_key: os.environ/{env_var}",
279
+ *_rejected(env_var),
280
+ " model_info:",
281
+ f" id: {short}-only",
282
+ f" free: {str(is_free).lower()}",
283
+ ]
284
+
285
+ n_free = sum(1 for c in entries if c[3])
286
+ lines += [
287
+ "",
288
+ "litellm_settings:",
289
+ " # Seam A. Cost attribution and turn affinity; it cannot block a",
290
+ " # request (`docs/0021`).",
291
+ " callbacks: agentctl_hook.proxy_handler_instance",
292
+ " # A provider config's supported params differ (`tools`,",
293
+ " # `parallel_tool_calls`, ...) and the pool changes providers on every",
294
+ " # retry. Without this an unsupported param 400s the whole request",
295
+ " # instead of being silently omitted for that one deployment",
296
+ " # (`research/phase-10-3-model-selection.md` §2.1 V7).",
297
+ " drop_params: true",
298
+ " # Repairs cross-provider tool-call history when a turn hops",
299
+ " # providers: drops an orphaned tool result, dedups a duplicated",
300
+ " # one, and synthesises a placeholder result for an orphaned tool",
301
+ " # call -- gates `sanitize_messages_for_tool_calling` (litellm",
302
+ " # 1.100.0). Schema/role translation and thought-signature handling",
303
+ " # already run unconditionally either way; this only turns on those",
304
+ " # three repairs (`research/phase-10-3-model-selection.md` §2.1 V6,",
305
+ " # `docs/0038` §5.1). The synthesised placeholder is content the",
306
+ " # agent never produced -- Seam A already pins a turn to one",
307
+ " # deployment to keep this rare (`kernel/hook.py::TurnAffinity`),",
308
+ " # but does not eliminate it.",
309
+ " modify_params: true",
310
+ "",
311
+ "router_settings:",
312
+ " # Every free deployment shares the model_name `pool`, and THAT is",
313
+ " # what produces failover: the router retries across deployments in",
314
+ " # one model group, cooling down whichever just failed. A key that",
315
+ " # hits its daily cap is skipped for `cooldown_time` while the other",
316
+ f" # {max(n_free - 1, 0)} keep serving.",
317
+ " routing_strategy: simple-shuffle",
318
+ # Enough retries to leave a capped account and land on another one.
319
+ f" num_retries: {min(max(n_free - 1, 2), 5)}",
320
+ " allowed_fails: 1",
321
+ " cooldown_time: 300",
322
+ "",
323
+ " # RETRY ACROSS DEPLOYMENTS ON DEPLOYMENT-SHAPED ERRORS.",
324
+ " #",
325
+ " # litellm gives an AuthenticationError or a BadRequestError zero",
326
+ " # retries, which is right for one endpoint and wrong for a pool:",
327
+ " # here the deployments are interchangeable, so a 401/402 means",
328
+ " # THAT KEY is bad and a 404 means THAT MODEL ID is gone -- neither",
329
+ " # says anything about the request, and the next deployment would",
330
+ " # have served it.",
331
+ " #",
332
+ " # Measured: a live run died outright when one Cerebras deployment",
333
+ " # returned 402 `Payment required`, with 47 working deployments in",
334
+ " # the same group (`docs/0041`). `docs/0034` §7 predicted it.",
335
+ " #",
336
+ " # Bounded deliberately. A BadRequestError can also mean the request",
337
+ " # itself is malformed, and that fails on every deployment -- so it",
338
+ " # gets the smallest budget, enough to cross a dead model id and not",
339
+ " # enough to spend the pool on a bad payload.",
340
+ " retry_policy:",
341
+ " AuthenticationErrorRetries: 3",
342
+ " BadRequestErrorRetries: 2",
343
+ " RateLimitErrorRetries: 3",
344
+ " InternalServerErrorRetries: 3",
345
+ " TimeoutErrorRetries: 2",
346
+ "",
347
+ " # And cool a broken deployment down on the FIRST failure rather",
348
+ " # than the second: a key that cannot authenticate will not start",
349
+ " # working on the next request, so the second attempt is a wasted",
350
+ " # one. Rate limits keep the default, because those do recover.",
351
+ " allowed_fails_policy:",
352
+ " AuthenticationErrorAllowedFails: 0",
353
+ " NotFoundErrorAllowedFails: 0",
354
+ "",
355
+ " # NO automatic fallback to `paid`. `fallbacks` maps between model",
356
+ " # GROUPS, not deployment ids -- litellm's own example is",
357
+ " # [{\"azure-gpt-3.5-turbo\": \"openai-gpt-3.5-turbo\"}] -- and an",
358
+ " # earlier version of this file pointed it at model_info ids, which",
359
+ " # resolve to nothing. It is left out rather than corrected because",
360
+ " # `docs/0002` §5 says never silently spend: paid is a group you ask",
361
+ " # for by name, not one you arrive at by failing.",
362
+ ]
363
+ if any(not c[3] for c in entries):
364
+ lines += [
365
+ " # to use it deliberately: --model openai/paid",
366
+ ]
367
+ lines.append("")
368
+ text = "\n".join(l for l in lines if l is not None)
369
+
370
+ # Parse our own output before handing it over. The first version of this
371
+ # function emitted `["a" "b"]` -- a missing comma -- and returned it
372
+ # happily; the proxy would have failed to start with a YAML error pointing
373
+ # at a file the user never wrote. A generator that cannot detect its own
374
+ # malformed output is the same shape as a guard that cannot fail
375
+ # (`docs/0028` §5).
376
+ _validate(text)
377
+ return text
378
+
379
+
380
+ def _validate(text: str) -> None:
381
+ import yaml
382
+
383
+ try:
384
+ doc = yaml.safe_load(text)
385
+ except yaml.YAMLError as e:
386
+ raise AssertionError(f"generated an invalid proxy config: {e}") from e
387
+
388
+ assert isinstance(doc, dict), "config is not a mapping"
389
+ models = doc.get("model_list") or []
390
+ assert models, "config has no model_list"
391
+ names = {m["model_name"] for m in models}
392
+ # Free deployments must share ONE name. That shared model group is where
393
+ # failover actually comes from: the router retries across the group and
394
+ # cools down whichever deployment just failed.
395
+ unexpected = {n for n in names
396
+ if n not in (POOL, PAID) and not n.startswith(SOURCE_PREFIX)}
397
+ assert not unexpected, f"unexpected model groups: {unexpected}"
398
+ assert POOL in names or names == {PAID}, "no free pool was built"
399
+ for m in models:
400
+ assert m["litellm_params"].get("api_key", "").startswith("os.environ/"), \
401
+ "a key was inlined instead of referenced from the environment"
402
+
403
+ # Deployment ids are what `fallbacks` points at. A duplicate would send a
404
+ # failover to whichever entry litellm resolved first -- silently the wrong
405
+ # provider. Two providers abbreviating to the same prefix caused exactly
406
+ # this once (`docs/0032`).
407
+ ids = [m["model_info"]["id"] for m in models]
408
+ assert len(set(ids)) == len(ids), f"duplicate deployment ids: {ids}"
409
+ # If fallbacks are ever reintroduced they must name model GROUPS. An
410
+ # earlier version pointed them at `model_info.id` values, which resolve to
411
+ # nothing -- litellm's own example maps one model_name to another.
412
+ groups = {m["model_name"] for m in models}
413
+ for entry in (doc.get("router_settings") or {}).get("fallbacks") or []:
414
+ assert isinstance(entry, dict), f"fallbacks is malformed: {entry!r}"
415
+ for source, targets in entry.items():
416
+ assert source in groups, \
417
+ f"fallback source {source!r} is not a model group"
418
+ for t in ([targets] if isinstance(targets, str) else targets):
419
+ assert t in groups, (
420
+ f"fallback target {t!r} is not a model group -- litellm "
421
+ f"maps groups, not deployment ids")
422
+
423
+
424
+ HOOK_MODULE = '''"""Seam A for the proxy. Loaded by `litellm_settings.callbacks`.
425
+
426
+ Generated by `agentctl proxy`. The hook records cost telemetry and stamps turn
427
+ affinity; it never blocks a request -- effects are gated at Seams B and C
428
+ inside the agent, not here (`docs/0008` §3).
429
+ """
430
+ import os
431
+
432
+ from agentctl.adapters.litellm.hook import AgentctlHook
433
+
434
+ proxy_handler_instance = AgentctlHook(
435
+ telemetry_path=os.environ.get("AGENTCTL_TELEMETRY", "hook_telemetry.json"))
436
+ '''
437
+
438
+
439
+ def write(out_dir: str | Path = ".", env: dict | None = None,
440
+ only: set[str] | None = None,
441
+ drop_models: set[str] = frozenset()) -> tuple[Path, Path]:
442
+ d = Path(out_dir)
443
+ d.mkdir(parents=True, exist_ok=True)
444
+ cfg = d / "proxy_config.yaml"
445
+ hook = d / "agentctl_hook.py"
446
+ cfg.write_text(build(env, only=only, drop_models=drop_models),
447
+ encoding="utf-8")
448
+ hook.write_text(HOOK_MODULE, encoding="utf-8")
449
+ write_start_scripts(d, Path(__file__).resolve().parent.parent.parent)
450
+ return cfg, hook
451
+
452
+
453
+ # The proxy prints a banner containing box-drawing characters. On Windows a
454
+ # redirected stdout defaults to cp1252, `click.echo` raises UnicodeEncodeError
455
+ # inside the startup event, and the server exits with "Application startup
456
+ # failed" -- which looks like a config error and is an encoding one. Fifth
457
+ # occurrence in this project; `docs/0012` §6 already required this and the
458
+ # generated output did not do it (`docs/0034` §8).
459
+ # The config says `api_key: os.environ/MISTRAL_API_KEY`, and litellm resolves
460
+ # that against the environment of the process it is started in. The keys live
461
+ # in `keys.env`, which is read by `agentctl`'s own Python and by nothing else
462
+ # -- so a launcher that does not load it starts a proxy holding no credentials
463
+ # for any provider whose key is only in that file.
464
+ #
465
+ # It does not fail at startup either. It serves, and every request returns
466
+ # "Invalid API Key" from the upstream, which reads like a bad key rather than
467
+ # an empty one. Found by the first live fan-out (`docs/0041`).
468
+ START_SH = """#!/usr/bin/env bash
469
+ # Generated by `agentctl proxy`. Start the pool.
470
+ set -euo pipefail
471
+ export PYTHONUTF8=1
472
+ export PYTHONIOENCODING=utf-8
473
+ export OPENHANDS_SUPPRESS_BANNER=1
474
+ export PYTHONPATH="{root}${{PYTHONPATH:+:$PYTHONPATH}}"
475
+
476
+ # The keys. `set -a` exports every assignment the file makes; comments and
477
+ # blank lines are ordinary shell and need no filtering. A local keys.env wins
478
+ # over the home one, matching `agentctl keys`.
479
+ for _kf in "$HOME/.agentctl/keys.env" "./keys.env"; do
480
+ if [ -f "$_kf" ]; then set -a; . "$_kf"; set +a; fi
481
+ done
482
+
483
+ exec "{litellm}" --config "$(dirname "$0")/proxy_config.yaml" --port "${{1:-4000}}"
484
+ """
485
+
486
+ START_PS1 = r"""# Generated by `agentctl proxy`. Start the pool.
487
+ $env:PYTHONUTF8 = "1"
488
+ $env:PYTHONIOENCODING = "utf-8"
489
+ $env:OPENHANDS_SUPPRESS_BANNER = "1"
490
+ $env:PYTHONPATH = "{root};$env:PYTHONPATH"
491
+
492
+ # The keys -- see the note above START_SH. Without this the proxy starts
493
+ # clean and every request returns "Invalid API Key".
494
+ foreach ($kf in @("$env:USERPROFILE\.agentctl\keys.env", ".\keys.env")) {{
495
+ if (Test-Path $kf) {{
496
+ Get-Content $kf | ForEach-Object {{
497
+ if ($_ -match '^\s*([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*)$') {{
498
+ $v = $Matches[2].Trim().Trim('"').Trim("'")
499
+ [Environment]::SetEnvironmentVariable($Matches[1], $v, "Process")
500
+ }}
501
+ }}
502
+ }}
503
+ }}
504
+
505
+ & "{litellm}" --config "$PSScriptRoot\proxy_config.yaml" --port $(if ($args[0]) {{$args[0]}} else {{4000}})
506
+ """
507
+
508
+
509
+ def _litellm_executable() -> str:
510
+ """The litellm CLI belonging to THIS interpreter.
511
+
512
+ A bare `litellm` only resolves when the virtualenv happens to be active,
513
+ and a launcher that requires you to have activated something is a launcher
514
+ that fails the first time you use it — which this one did.
515
+ """
516
+ import shutil
517
+ import sys
518
+
519
+ scripts = Path(sys.executable).parent
520
+ for name in ("litellm.exe", "litellm"):
521
+ if (candidate := scripts / name).exists():
522
+ return str(candidate)
523
+ return shutil.which("litellm") or "litellm"
524
+
525
+
526
+ def write_start_scripts(out_dir: str | Path, root: str | Path) -> list[Path]:
527
+ """Runnable launchers, so the encoding trap cannot be stepped in."""
528
+ d = Path(out_dir)
529
+ exe = _litellm_executable().replace("\\", "/")
530
+ made = []
531
+ for name, body in (("start.sh", START_SH), ("start.ps1", START_PS1)):
532
+ p = d / name
533
+ p.write_text(body.format(root=str(root).replace("\\", "/"), litellm=exe),
534
+ encoding="utf-8", newline="\n")
535
+ made.append(p)
536
+ return made