handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,536 @@
|
|
|
1
|
+
r"""Generate a LiteLLM proxy config from the keys that actually exist.
|
|
2
|
+
|
|
3
|
+
The project's charter is surviving a rate limit without work stopping. That
|
|
4
|
+
requires somewhere to fail over TO, and a config listing providers you have no
|
|
5
|
+
key for does not start -- litellm resolves `os.environ/X` at load and refuses.
|
|
6
|
+
So the config is generated from the environment rather than hand-written, and
|
|
7
|
+
adding a provider means exporting a key, not editing YAML.
|
|
8
|
+
|
|
9
|
+
## What failover can and cannot do for you
|
|
10
|
+
|
|
11
|
+
Two different failures, and a pool only fixes one of them with one account:
|
|
12
|
+
|
|
13
|
+
| failure | one OpenRouter key | plus a second provider |
|
|
14
|
+
|---|---|---|
|
|
15
|
+
| transient overload (*"Upstream error from Nvidia"*) | **fixed** — another model, another upstream | fixed |
|
|
16
|
+
| per-model rate limit | **fixed** | fixed |
|
|
17
|
+
| free-models-per-**day** cap | **not fixed** — the cap is account-wide | fixed |
|
|
18
|
+
|
|
19
|
+
Saying a pool fixes the daily cap would be the comfortable answer and the
|
|
20
|
+
wrong one: OpenRouter counts free requests per account, not per model. The
|
|
21
|
+
`:free` models below buy resilience against the first two rows, which is what
|
|
22
|
+
actually interrupted a real run, and nothing against the third.
|
|
23
|
+
|
|
24
|
+
## Fallbacks are ordered, and the order is a cost decision
|
|
25
|
+
|
|
26
|
+
`fallbacks` sends the next attempt to the next entry. Free models come first
|
|
27
|
+
and paid providers last, so an outage degrades toward *slower*, never silently
|
|
28
|
+
toward *billed*. `docs/0002` §5: never silently spend.
|
|
29
|
+
"""
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import os
|
|
33
|
+
import re
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
|
|
36
|
+
# (env var, litellm model, short name, is_free). Order is fallback order:
|
|
37
|
+
# free first, paid last, so degrading never silently costs money.
|
|
38
|
+
POOL = "pool"
|
|
39
|
+
# A separate model group, never an automatic fallback target (`docs/0002` §5).
|
|
40
|
+
PAID = "paid"
|
|
41
|
+
# `pool-openrouter`, `pool-gemini`, ... One source, deliberately narrower than
|
|
42
|
+
# `pool`. The prefix is what makes a source group recognisable as one rather
|
|
43
|
+
# than as some unrelated model group someone hand-added (`docs/0039`).
|
|
44
|
+
SOURCE_PREFIX = "pool-"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def available(env: dict | None = None) -> list[tuple[str, str, str, bool]]:
|
|
48
|
+
"""One deployment per ACCOUNT per model: (env, model, id, is_free).
|
|
49
|
+
|
|
50
|
+
The cross product is the point. Two OpenRouter accounts times three models
|
|
51
|
+
is six deployments, and when one account hits its daily cap the other three
|
|
52
|
+
keep serving — which a single-account pool cannot do however many models it
|
|
53
|
+
lists (`docs/0033`).
|
|
54
|
+
|
|
55
|
+
Ordered free-first, paid-last, because that is also the fallback order: an
|
|
56
|
+
outage degrades toward slower, never silently toward billed (`docs/0002`
|
|
57
|
+
§5).
|
|
58
|
+
"""
|
|
59
|
+
from .providers import all_accounts
|
|
60
|
+
|
|
61
|
+
out = []
|
|
62
|
+
seen: dict[str, int] = {}
|
|
63
|
+
for acct in all_accounts(env):
|
|
64
|
+
n = seen[acct.provider.name] = seen.get(acct.provider.name, 0) + 1
|
|
65
|
+
for i, model in enumerate(acct.provider.models):
|
|
66
|
+
# `provider-aN-mI`, both parts always present and labelled. The
|
|
67
|
+
# obvious scheme -- label plus model index -- produced
|
|
68
|
+
# `openrouter-2` (account 1, model 2) alongside `openrouter-2-0`
|
|
69
|
+
# (account 2, model 0): unique, and unreadable in a fallback map
|
|
70
|
+
# exactly when you are debugging one. Nothing is abbreviated
|
|
71
|
+
# either, since `openrouter` and `openai` share a prefix.
|
|
72
|
+
short = f"{acct.provider.name}-a{n}-m{i}"
|
|
73
|
+
out.append((acct.env, f"{acct.provider.prefix}{model}", short,
|
|
74
|
+
acct.provider.free_tier))
|
|
75
|
+
return out
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# The optional `-only` suffix marks a source-group copy. It is the SAME
|
|
79
|
+
# deployment under a narrower name, so it parses to the same triple --
|
|
80
|
+
# a caller counting deployments must not count it twice (`docs/0039`).
|
|
81
|
+
_DEPLOYMENT_ID = re.compile(r"^([a-z][a-z0-9]*)-a(\d+)-m(\d+)(?:-only)?$")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def parse_deployment_id(short: str) -> tuple[str, int, int] | None:
|
|
85
|
+
"""The inverse of the `short` id `available()` builds above: one place
|
|
86
|
+
defines the format, this parses it, and a round-trip test ties them
|
|
87
|
+
together so they cannot drift apart silently.
|
|
88
|
+
|
|
89
|
+
Returns `(provider_name, account_ordinal, model_index)`, 1-indexed for
|
|
90
|
+
the account to match the id text itself (`a1` -> `1`). `None` for
|
|
91
|
+
anything that does not match -- callers must not guess at a foreign or
|
|
92
|
+
malformed id, only count it as unrecognised.
|
|
93
|
+
"""
|
|
94
|
+
m = _DEPLOYMENT_ID.match(short)
|
|
95
|
+
if not m:
|
|
96
|
+
return None
|
|
97
|
+
provider, n, i = m.groups()
|
|
98
|
+
return provider, int(n), int(i)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# Kept for callers that want the catalogue rather than what is configured.
|
|
102
|
+
def _candidates() -> list[tuple[str, str, str, bool]]:
|
|
103
|
+
from .providers import PROVIDERS as _P
|
|
104
|
+
return [(p.key, f"{p.prefix}{m}", f"{p.name}-{i}", p.free_tier)
|
|
105
|
+
for p in sorted(_P, key=lambda x: not x.free_tier)
|
|
106
|
+
for i, m in enumerate(p.models)]
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
CANDIDATES: list[tuple[str, str, str, bool]] = _candidates()
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def accounts(entries: list[tuple[str, str, str, bool]]) -> set[str]:
|
|
113
|
+
"""Distinct credentials behind the pool, by environment variable.
|
|
114
|
+
|
|
115
|
+
Not providers: two keys at one provider are two accounts with two quotas,
|
|
116
|
+
and counting them as one would under-report the failover you actually have.
|
|
117
|
+
"""
|
|
118
|
+
return {c[0] for c in entries}
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _by_source(entries: list[tuple[str, str, str, bool]]) -> dict[str, list]:
|
|
122
|
+
"""Free deployments grouped by the provider serving them.
|
|
123
|
+
|
|
124
|
+
Paid deployments are left out: `paid` is already a group you ask for by
|
|
125
|
+
name, and a source group that silently included it would turn "route to
|
|
126
|
+
one provider" into "spend money" (`docs/0002` §5).
|
|
127
|
+
"""
|
|
128
|
+
out: dict[str, list] = {}
|
|
129
|
+
for e in entries:
|
|
130
|
+
if not e[3]:
|
|
131
|
+
continue
|
|
132
|
+
parsed = parse_deployment_id(e[2])
|
|
133
|
+
if parsed is None:
|
|
134
|
+
continue
|
|
135
|
+
out.setdefault(parsed[0], []).append(e)
|
|
136
|
+
return out
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def sources(env: dict | None = None) -> list[dict]:
|
|
140
|
+
"""What a source picker needs, and the cost of each choice.
|
|
141
|
+
|
|
142
|
+
The cost is not decoration. `pool` is 48 deployments across 24 accounts;
|
|
143
|
+
asking for one source can leave you with six behind a single account-wide
|
|
144
|
+
daily cap, which is precisely the failure the multi-account design exists
|
|
145
|
+
to escape (`docs/0033`). A picker that shows only the names is offering a
|
|
146
|
+
downgrade without saying so, so every row carries what it gives up
|
|
147
|
+
(`research/phase-10-3-model-selection.md` §6.5).
|
|
148
|
+
|
|
149
|
+
Ordered widest-first, so the safe choice reads first.
|
|
150
|
+
"""
|
|
151
|
+
entries = available(env)
|
|
152
|
+
total = len([e for e in entries if e[3]])
|
|
153
|
+
rows = []
|
|
154
|
+
for name, mine in _by_source(entries).items():
|
|
155
|
+
from .providers import BY_NAME
|
|
156
|
+
prov = BY_NAME.get(name)
|
|
157
|
+
n_acct = len({e[0] for e in mine})
|
|
158
|
+
# Keys are not allowances. Gemini bills per Google project, so six
|
|
159
|
+
# keys there are one quota -- and a picker that offered "6 accounts"
|
|
160
|
+
# would be selling failover this source does not have.
|
|
161
|
+
n_quota = n_acct if (prov is None or prov.quota_per_key) else min(n_acct, 1)
|
|
162
|
+
rows.append({
|
|
163
|
+
"source": name,
|
|
164
|
+
"group": f"{SOURCE_PREFIX}{name}",
|
|
165
|
+
"deployments": len(mine),
|
|
166
|
+
"accounts": n_acct,
|
|
167
|
+
"quotas": n_quota,
|
|
168
|
+
"models": sorted({e[1] for e in mine}),
|
|
169
|
+
"gives_up": total - len(mine),
|
|
170
|
+
})
|
|
171
|
+
rows.sort(key=lambda r: (-r["deployments"], r["source"]))
|
|
172
|
+
return rows
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def verified_providers(allow_paid: bool = False) -> tuple[set[str], dict]:
|
|
176
|
+
"""Which providers can actually complete a request right now.
|
|
177
|
+
|
|
178
|
+
Authenticating is not the same as being allowed to infer: six Cerebras
|
|
179
|
+
keys list models happily and every completion returns "Payment required".
|
|
180
|
+
Those deployments sit in the pool failing ~11% of requests with an HTTP 402
|
|
181
|
+
that litellm does not treat as retryable, so a pool built from credentials
|
|
182
|
+
is worse than one built from what works (`docs/0034` §7).
|
|
183
|
+
"""
|
|
184
|
+
from .probe import LIMITED, LIVE, check_inference
|
|
185
|
+
from .providers import PROVIDERS
|
|
186
|
+
|
|
187
|
+
ok, report = set(), {}
|
|
188
|
+
for p in PROVIDERS:
|
|
189
|
+
r = check_inference(p.name, allow_paid=allow_paid, all_models=True)
|
|
190
|
+
if r is None:
|
|
191
|
+
continue
|
|
192
|
+
report[p.name] = r
|
|
193
|
+
if r.status in (LIVE, LIMITED):
|
|
194
|
+
ok.add(p.name)
|
|
195
|
+
return ok, report
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _rejected(env_var: str) -> list[str]:
|
|
199
|
+
"""`additional_drop_params` for a provider that 400s on a forwarded param."""
|
|
200
|
+
from .providers import BY_KEY
|
|
201
|
+
reject = BY_KEY[env_var.split("_API_KEY")[0] + "_API_KEY"].reject_params
|
|
202
|
+
return [f" additional_drop_params: [{', '.join(reject)}]"] if reject else []
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def build(env: dict | None = None, telemetry: str | Path | None = None,
|
|
206
|
+
only: set[str] | None = None,
|
|
207
|
+
drop_models: set[str] = frozenset()) -> str:
|
|
208
|
+
"""The proxy config, as YAML text. `only` restricts it to named providers.
|
|
209
|
+
|
|
210
|
+
`drop_models` holds full model strings verification found the provider no
|
|
211
|
+
longer serves. Left in, each is a deployment that 404s on every call.
|
|
212
|
+
"""
|
|
213
|
+
entries = [e for e in available(env) if e[1] not in drop_models]
|
|
214
|
+
if only is not None:
|
|
215
|
+
from .providers import BY_KEY
|
|
216
|
+
entries = [e for e in entries
|
|
217
|
+
if BY_KEY[e[0].split("_API_KEY")[0] + "_API_KEY"].name in only]
|
|
218
|
+
if not entries:
|
|
219
|
+
raise SystemExit("no provider passed verification -- nothing to route to")
|
|
220
|
+
if not entries:
|
|
221
|
+
raise SystemExit(
|
|
222
|
+
"no provider keys in the environment — nothing to route to.\n"
|
|
223
|
+
" set at least one, e.g. OPENROUTER_API_KEY")
|
|
224
|
+
|
|
225
|
+
lines = [
|
|
226
|
+
"# GENERATED by `agentctl proxy` from the keys in your environment.",
|
|
227
|
+
"# Re-run it after exporting a new provider key; do not hand-edit.",
|
|
228
|
+
"#",
|
|
229
|
+
f"# {len(entries)} deployment(s) across {len(accounts(entries))} "
|
|
230
|
+
f"account(s).",
|
|
231
|
+
"",
|
|
232
|
+
"model_list:",
|
|
233
|
+
]
|
|
234
|
+
for env_var, model, short, is_free in entries:
|
|
235
|
+
lines += [
|
|
236
|
+
f" - model_name: {POOL if is_free else PAID}",
|
|
237
|
+
" litellm_params:",
|
|
238
|
+
f" model: {model}",
|
|
239
|
+
f" api_key: os.environ/{env_var}",
|
|
240
|
+
*_rejected(env_var),
|
|
241
|
+
" model_info:",
|
|
242
|
+
f" id: {short}",
|
|
243
|
+
f" free: {str(is_free).lower()}",
|
|
244
|
+
]
|
|
245
|
+
|
|
246
|
+
# ── the source groups ───────────────────────────────────────────────
|
|
247
|
+
#
|
|
248
|
+
# Everything above is one group, `pool`, and that is what makes failover
|
|
249
|
+
# work. These add a second, narrower way to ask: `pool-openrouter` routes
|
|
250
|
+
# only to OpenRouter.
|
|
251
|
+
#
|
|
252
|
+
# It is a duplicated entry, not a second deployment -- the same key, the
|
|
253
|
+
# same model, reachable under two names. The id carries `-only` so it
|
|
254
|
+
# stays unique, because a duplicate id would send a fallback to whichever
|
|
255
|
+
# entry litellm resolved first (`docs/0032`).
|
|
256
|
+
#
|
|
257
|
+
# The cost is real and belongs next to the choice: asking for one source
|
|
258
|
+
# narrows the pool to that source's share, and an account-wide daily cap
|
|
259
|
+
# then has nothing to fail over to. `agentctl models` prints what each
|
|
260
|
+
# choice leaves before you make it (`research/phase-10-3` §6.5).
|
|
261
|
+
by_source = _by_source(entries)
|
|
262
|
+
if len(by_source) > 1:
|
|
263
|
+
lines += ["", " # ── source groups: narrower on purpose ──"]
|
|
264
|
+
for source in sorted(by_source):
|
|
265
|
+
group = f"{SOURCE_PREFIX}{source}"
|
|
266
|
+
mine = by_source[source]
|
|
267
|
+
lines += [
|
|
268
|
+
f" # {group}: {len(mine)} deployment(s), "
|
|
269
|
+
f"{len({e[0] for e in mine})} account(s). "
|
|
270
|
+
f"Choosing it gives up the other "
|
|
271
|
+
f"{len(entries) - len(mine)}.",
|
|
272
|
+
]
|
|
273
|
+
for env_var, model, short, is_free in mine:
|
|
274
|
+
lines += [
|
|
275
|
+
f" - model_name: {group}",
|
|
276
|
+
" litellm_params:",
|
|
277
|
+
f" model: {model}",
|
|
278
|
+
f" api_key: os.environ/{env_var}",
|
|
279
|
+
*_rejected(env_var),
|
|
280
|
+
" model_info:",
|
|
281
|
+
f" id: {short}-only",
|
|
282
|
+
f" free: {str(is_free).lower()}",
|
|
283
|
+
]
|
|
284
|
+
|
|
285
|
+
n_free = sum(1 for c in entries if c[3])
|
|
286
|
+
lines += [
|
|
287
|
+
"",
|
|
288
|
+
"litellm_settings:",
|
|
289
|
+
" # Seam A. Cost attribution and turn affinity; it cannot block a",
|
|
290
|
+
" # request (`docs/0021`).",
|
|
291
|
+
" callbacks: agentctl_hook.proxy_handler_instance",
|
|
292
|
+
" # A provider config's supported params differ (`tools`,",
|
|
293
|
+
" # `parallel_tool_calls`, ...) and the pool changes providers on every",
|
|
294
|
+
" # retry. Without this an unsupported param 400s the whole request",
|
|
295
|
+
" # instead of being silently omitted for that one deployment",
|
|
296
|
+
" # (`research/phase-10-3-model-selection.md` §2.1 V7).",
|
|
297
|
+
" drop_params: true",
|
|
298
|
+
" # Repairs cross-provider tool-call history when a turn hops",
|
|
299
|
+
" # providers: drops an orphaned tool result, dedups a duplicated",
|
|
300
|
+
" # one, and synthesises a placeholder result for an orphaned tool",
|
|
301
|
+
" # call -- gates `sanitize_messages_for_tool_calling` (litellm",
|
|
302
|
+
" # 1.100.0). Schema/role translation and thought-signature handling",
|
|
303
|
+
" # already run unconditionally either way; this only turns on those",
|
|
304
|
+
" # three repairs (`research/phase-10-3-model-selection.md` §2.1 V6,",
|
|
305
|
+
" # `docs/0038` §5.1). The synthesised placeholder is content the",
|
|
306
|
+
" # agent never produced -- Seam A already pins a turn to one",
|
|
307
|
+
" # deployment to keep this rare (`kernel/hook.py::TurnAffinity`),",
|
|
308
|
+
" # but does not eliminate it.",
|
|
309
|
+
" modify_params: true",
|
|
310
|
+
"",
|
|
311
|
+
"router_settings:",
|
|
312
|
+
" # Every free deployment shares the model_name `pool`, and THAT is",
|
|
313
|
+
" # what produces failover: the router retries across deployments in",
|
|
314
|
+
" # one model group, cooling down whichever just failed. A key that",
|
|
315
|
+
" # hits its daily cap is skipped for `cooldown_time` while the other",
|
|
316
|
+
f" # {max(n_free - 1, 0)} keep serving.",
|
|
317
|
+
" routing_strategy: simple-shuffle",
|
|
318
|
+
# Enough retries to leave a capped account and land on another one.
|
|
319
|
+
f" num_retries: {min(max(n_free - 1, 2), 5)}",
|
|
320
|
+
" allowed_fails: 1",
|
|
321
|
+
" cooldown_time: 300",
|
|
322
|
+
"",
|
|
323
|
+
" # RETRY ACROSS DEPLOYMENTS ON DEPLOYMENT-SHAPED ERRORS.",
|
|
324
|
+
" #",
|
|
325
|
+
" # litellm gives an AuthenticationError or a BadRequestError zero",
|
|
326
|
+
" # retries, which is right for one endpoint and wrong for a pool:",
|
|
327
|
+
" # here the deployments are interchangeable, so a 401/402 means",
|
|
328
|
+
" # THAT KEY is bad and a 404 means THAT MODEL ID is gone -- neither",
|
|
329
|
+
" # says anything about the request, and the next deployment would",
|
|
330
|
+
" # have served it.",
|
|
331
|
+
" #",
|
|
332
|
+
" # Measured: a live run died outright when one Cerebras deployment",
|
|
333
|
+
" # returned 402 `Payment required`, with 47 working deployments in",
|
|
334
|
+
" # the same group (`docs/0041`). `docs/0034` §7 predicted it.",
|
|
335
|
+
" #",
|
|
336
|
+
" # Bounded deliberately. A BadRequestError can also mean the request",
|
|
337
|
+
" # itself is malformed, and that fails on every deployment -- so it",
|
|
338
|
+
" # gets the smallest budget, enough to cross a dead model id and not",
|
|
339
|
+
" # enough to spend the pool on a bad payload.",
|
|
340
|
+
" retry_policy:",
|
|
341
|
+
" AuthenticationErrorRetries: 3",
|
|
342
|
+
" BadRequestErrorRetries: 2",
|
|
343
|
+
" RateLimitErrorRetries: 3",
|
|
344
|
+
" InternalServerErrorRetries: 3",
|
|
345
|
+
" TimeoutErrorRetries: 2",
|
|
346
|
+
"",
|
|
347
|
+
" # And cool a broken deployment down on the FIRST failure rather",
|
|
348
|
+
" # than the second: a key that cannot authenticate will not start",
|
|
349
|
+
" # working on the next request, so the second attempt is a wasted",
|
|
350
|
+
" # one. Rate limits keep the default, because those do recover.",
|
|
351
|
+
" allowed_fails_policy:",
|
|
352
|
+
" AuthenticationErrorAllowedFails: 0",
|
|
353
|
+
" NotFoundErrorAllowedFails: 0",
|
|
354
|
+
"",
|
|
355
|
+
" # NO automatic fallback to `paid`. `fallbacks` maps between model",
|
|
356
|
+
" # GROUPS, not deployment ids -- litellm's own example is",
|
|
357
|
+
" # [{\"azure-gpt-3.5-turbo\": \"openai-gpt-3.5-turbo\"}] -- and an",
|
|
358
|
+
" # earlier version of this file pointed it at model_info ids, which",
|
|
359
|
+
" # resolve to nothing. It is left out rather than corrected because",
|
|
360
|
+
" # `docs/0002` §5 says never silently spend: paid is a group you ask",
|
|
361
|
+
" # for by name, not one you arrive at by failing.",
|
|
362
|
+
]
|
|
363
|
+
if any(not c[3] for c in entries):
|
|
364
|
+
lines += [
|
|
365
|
+
" # to use it deliberately: --model openai/paid",
|
|
366
|
+
]
|
|
367
|
+
lines.append("")
|
|
368
|
+
text = "\n".join(l for l in lines if l is not None)
|
|
369
|
+
|
|
370
|
+
# Parse our own output before handing it over. The first version of this
|
|
371
|
+
# function emitted `["a" "b"]` -- a missing comma -- and returned it
|
|
372
|
+
# happily; the proxy would have failed to start with a YAML error pointing
|
|
373
|
+
# at a file the user never wrote. A generator that cannot detect its own
|
|
374
|
+
# malformed output is the same shape as a guard that cannot fail
|
|
375
|
+
# (`docs/0028` §5).
|
|
376
|
+
_validate(text)
|
|
377
|
+
return text
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _validate(text: str) -> None:
|
|
381
|
+
import yaml
|
|
382
|
+
|
|
383
|
+
try:
|
|
384
|
+
doc = yaml.safe_load(text)
|
|
385
|
+
except yaml.YAMLError as e:
|
|
386
|
+
raise AssertionError(f"generated an invalid proxy config: {e}") from e
|
|
387
|
+
|
|
388
|
+
assert isinstance(doc, dict), "config is not a mapping"
|
|
389
|
+
models = doc.get("model_list") or []
|
|
390
|
+
assert models, "config has no model_list"
|
|
391
|
+
names = {m["model_name"] for m in models}
|
|
392
|
+
# Free deployments must share ONE name. That shared model group is where
|
|
393
|
+
# failover actually comes from: the router retries across the group and
|
|
394
|
+
# cools down whichever deployment just failed.
|
|
395
|
+
unexpected = {n for n in names
|
|
396
|
+
if n not in (POOL, PAID) and not n.startswith(SOURCE_PREFIX)}
|
|
397
|
+
assert not unexpected, f"unexpected model groups: {unexpected}"
|
|
398
|
+
assert POOL in names or names == {PAID}, "no free pool was built"
|
|
399
|
+
for m in models:
|
|
400
|
+
assert m["litellm_params"].get("api_key", "").startswith("os.environ/"), \
|
|
401
|
+
"a key was inlined instead of referenced from the environment"
|
|
402
|
+
|
|
403
|
+
# Deployment ids are what `fallbacks` points at. A duplicate would send a
|
|
404
|
+
# failover to whichever entry litellm resolved first -- silently the wrong
|
|
405
|
+
# provider. Two providers abbreviating to the same prefix caused exactly
|
|
406
|
+
# this once (`docs/0032`).
|
|
407
|
+
ids = [m["model_info"]["id"] for m in models]
|
|
408
|
+
assert len(set(ids)) == len(ids), f"duplicate deployment ids: {ids}"
|
|
409
|
+
# If fallbacks are ever reintroduced they must name model GROUPS. An
|
|
410
|
+
# earlier version pointed them at `model_info.id` values, which resolve to
|
|
411
|
+
# nothing -- litellm's own example maps one model_name to another.
|
|
412
|
+
groups = {m["model_name"] for m in models}
|
|
413
|
+
for entry in (doc.get("router_settings") or {}).get("fallbacks") or []:
|
|
414
|
+
assert isinstance(entry, dict), f"fallbacks is malformed: {entry!r}"
|
|
415
|
+
for source, targets in entry.items():
|
|
416
|
+
assert source in groups, \
|
|
417
|
+
f"fallback source {source!r} is not a model group"
|
|
418
|
+
for t in ([targets] if isinstance(targets, str) else targets):
|
|
419
|
+
assert t in groups, (
|
|
420
|
+
f"fallback target {t!r} is not a model group -- litellm "
|
|
421
|
+
f"maps groups, not deployment ids")
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
HOOK_MODULE = '''"""Seam A for the proxy. Loaded by `litellm_settings.callbacks`.
|
|
425
|
+
|
|
426
|
+
Generated by `agentctl proxy`. The hook records cost telemetry and stamps turn
|
|
427
|
+
affinity; it never blocks a request -- effects are gated at Seams B and C
|
|
428
|
+
inside the agent, not here (`docs/0008` §3).
|
|
429
|
+
"""
|
|
430
|
+
import os
|
|
431
|
+
|
|
432
|
+
from agentctl.adapters.litellm.hook import AgentctlHook
|
|
433
|
+
|
|
434
|
+
proxy_handler_instance = AgentctlHook(
|
|
435
|
+
telemetry_path=os.environ.get("AGENTCTL_TELEMETRY", "hook_telemetry.json"))
|
|
436
|
+
'''
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def write(out_dir: str | Path = ".", env: dict | None = None,
|
|
440
|
+
only: set[str] | None = None,
|
|
441
|
+
drop_models: set[str] = frozenset()) -> tuple[Path, Path]:
|
|
442
|
+
d = Path(out_dir)
|
|
443
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
444
|
+
cfg = d / "proxy_config.yaml"
|
|
445
|
+
hook = d / "agentctl_hook.py"
|
|
446
|
+
cfg.write_text(build(env, only=only, drop_models=drop_models),
|
|
447
|
+
encoding="utf-8")
|
|
448
|
+
hook.write_text(HOOK_MODULE, encoding="utf-8")
|
|
449
|
+
write_start_scripts(d, Path(__file__).resolve().parent.parent.parent)
|
|
450
|
+
return cfg, hook
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
# The proxy prints a banner containing box-drawing characters. On Windows a
|
|
454
|
+
# redirected stdout defaults to cp1252, `click.echo` raises UnicodeEncodeError
|
|
455
|
+
# inside the startup event, and the server exits with "Application startup
|
|
456
|
+
# failed" -- which looks like a config error and is an encoding one. Fifth
|
|
457
|
+
# occurrence in this project; `docs/0012` §6 already required this and the
|
|
458
|
+
# generated output did not do it (`docs/0034` §8).
|
|
459
|
+
# The config says `api_key: os.environ/MISTRAL_API_KEY`, and litellm resolves
|
|
460
|
+
# that against the environment of the process it is started in. The keys live
|
|
461
|
+
# in `keys.env`, which is read by `agentctl`'s own Python and by nothing else
|
|
462
|
+
# -- so a launcher that does not load it starts a proxy holding no credentials
|
|
463
|
+
# for any provider whose key is only in that file.
|
|
464
|
+
#
|
|
465
|
+
# It does not fail at startup either. It serves, and every request returns
|
|
466
|
+
# "Invalid API Key" from the upstream, which reads like a bad key rather than
|
|
467
|
+
# an empty one. Found by the first live fan-out (`docs/0041`).
|
|
468
|
+
START_SH = """#!/usr/bin/env bash
|
|
469
|
+
# Generated by `agentctl proxy`. Start the pool.
|
|
470
|
+
set -euo pipefail
|
|
471
|
+
export PYTHONUTF8=1
|
|
472
|
+
export PYTHONIOENCODING=utf-8
|
|
473
|
+
export OPENHANDS_SUPPRESS_BANNER=1
|
|
474
|
+
export PYTHONPATH="{root}${{PYTHONPATH:+:$PYTHONPATH}}"
|
|
475
|
+
|
|
476
|
+
# The keys. `set -a` exports every assignment the file makes; comments and
|
|
477
|
+
# blank lines are ordinary shell and need no filtering. A local keys.env wins
|
|
478
|
+
# over the home one, matching `agentctl keys`.
|
|
479
|
+
for _kf in "$HOME/.agentctl/keys.env" "./keys.env"; do
|
|
480
|
+
if [ -f "$_kf" ]; then set -a; . "$_kf"; set +a; fi
|
|
481
|
+
done
|
|
482
|
+
|
|
483
|
+
exec "{litellm}" --config "$(dirname "$0")/proxy_config.yaml" --port "${{1:-4000}}"
|
|
484
|
+
"""
|
|
485
|
+
|
|
486
|
+
START_PS1 = r"""# Generated by `agentctl proxy`. Start the pool.
|
|
487
|
+
$env:PYTHONUTF8 = "1"
|
|
488
|
+
$env:PYTHONIOENCODING = "utf-8"
|
|
489
|
+
$env:OPENHANDS_SUPPRESS_BANNER = "1"
|
|
490
|
+
$env:PYTHONPATH = "{root};$env:PYTHONPATH"
|
|
491
|
+
|
|
492
|
+
# The keys -- see the note above START_SH. Without this the proxy starts
|
|
493
|
+
# clean and every request returns "Invalid API Key".
|
|
494
|
+
foreach ($kf in @("$env:USERPROFILE\.agentctl\keys.env", ".\keys.env")) {{
|
|
495
|
+
if (Test-Path $kf) {{
|
|
496
|
+
Get-Content $kf | ForEach-Object {{
|
|
497
|
+
if ($_ -match '^\s*([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*)$') {{
|
|
498
|
+
$v = $Matches[2].Trim().Trim('"').Trim("'")
|
|
499
|
+
[Environment]::SetEnvironmentVariable($Matches[1], $v, "Process")
|
|
500
|
+
}}
|
|
501
|
+
}}
|
|
502
|
+
}}
|
|
503
|
+
}}
|
|
504
|
+
|
|
505
|
+
& "{litellm}" --config "$PSScriptRoot\proxy_config.yaml" --port $(if ($args[0]) {{$args[0]}} else {{4000}})
|
|
506
|
+
"""
|
|
507
|
+
|
|
508
|
+
|
|
509
|
+
def _litellm_executable() -> str:
|
|
510
|
+
"""The litellm CLI belonging to THIS interpreter.
|
|
511
|
+
|
|
512
|
+
A bare `litellm` only resolves when the virtualenv happens to be active,
|
|
513
|
+
and a launcher that requires you to have activated something is a launcher
|
|
514
|
+
that fails the first time you use it — which this one did.
|
|
515
|
+
"""
|
|
516
|
+
import shutil
|
|
517
|
+
import sys
|
|
518
|
+
|
|
519
|
+
scripts = Path(sys.executable).parent
|
|
520
|
+
for name in ("litellm.exe", "litellm"):
|
|
521
|
+
if (candidate := scripts / name).exists():
|
|
522
|
+
return str(candidate)
|
|
523
|
+
return shutil.which("litellm") or "litellm"
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
def write_start_scripts(out_dir: str | Path, root: str | Path) -> list[Path]:
|
|
527
|
+
"""Runnable launchers, so the encoding trap cannot be stepped in."""
|
|
528
|
+
d = Path(out_dir)
|
|
529
|
+
exe = _litellm_executable().replace("\\", "/")
|
|
530
|
+
made = []
|
|
531
|
+
for name, body in (("start.sh", START_SH), ("start.ps1", START_PS1)):
|
|
532
|
+
p = d / name
|
|
533
|
+
p.write_text(body.format(root=str(root).replace("\\", "/"), litellm=exe),
|
|
534
|
+
encoding="utf-8", newline="\n")
|
|
535
|
+
made.append(p)
|
|
536
|
+
return made
|