agentic-runner 2.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. agentic_runner/__init__.py +12 -0
  2. agentic_runner/activities.py +4918 -0
  3. agentic_runner/callback.py +342 -0
  4. agentic_runner/child_watcher.py +66 -0
  5. agentic_runner/cli.py +416 -0
  6. agentic_runner/config.py +105 -0
  7. agentic_runner/credentials.py +252 -0
  8. agentic_runner/device_login_activities.py +79 -0
  9. agentic_runner/egress.py +243 -0
  10. agentic_runner/heartbeat_link.py +249 -0
  11. agentic_runner/hooks.py +455 -0
  12. agentic_runner/host_store.py +295 -0
  13. agentic_runner/integrations/__init__.py +0 -0
  14. agentic_runner/integrations/git/__init__.py +1 -0
  15. agentic_runner/integrations/git/contracts.py +198 -0
  16. agentic_runner/integrations/git/evidence.py +442 -0
  17. agentic_runner/integrations/git/fake_workspace.py +339 -0
  18. agentic_runner/integrations/git/workspace.py +921 -0
  19. agentic_runner/integrations/github/__init__.py +53 -0
  20. agentic_runner/integrations/github/auth.py +171 -0
  21. agentic_runner/integrations/github/fake_client.py +494 -0
  22. agentic_runner/integrations/github/gh_client.py +944 -0
  23. agentic_runner/lifecycle.py +48 -0
  24. agentic_runner/llm_proxy.py +937 -0
  25. agentic_runner/mcp.py +342 -0
  26. agentic_runner/message_store.py +341 -0
  27. agentic_runner/py.typed +0 -0
  28. agentic_runner/recipient_key_secret.py +134 -0
  29. agentic_runner/registration.py +363 -0
  30. agentic_runner/runtime/__init__.py +0 -0
  31. agentic_runner/runtime/verifier_command.py +344 -0
  32. agentic_runner/sealed_box.py +509 -0
  33. agentic_runner/service.py +1068 -0
  34. agentic_runner/tiny_http.py +133 -0
  35. agentic_runner/triage_activities.py +113 -0
  36. agentic_runner/user_sources.py +546 -0
  37. agentic_runner/workers/__init__.py +1 -0
  38. agentic_runner/workers/_runtime_support.py +388 -0
  39. agentic_runner/workers/agent_runtime.py +93 -0
  40. agentic_runner/workers/claude_runtime.py +226 -0
  41. agentic_runner/workers/codex_runtime.py +311 -0
  42. agentic_runner/workers/command_policy.py +250 -0
  43. agentic_runner/workers/contract_device_login.py +211 -0
  44. agentic_runner/workers/contract_isolation.py +500 -0
  45. agentic_runner/workers/fastapi_client.py +396 -0
  46. agentic_runner/workers/harness_usage.py +65 -0
  47. agentic_runner/workers/mcp_config.py +111 -0
  48. agentic_runner/workers/settings.py +314 -0
  49. agentic_runner/workstation.py +687 -0
  50. agentic_runner-2.6.0.dist-info/METADATA +49 -0
  51. agentic_runner-2.6.0.dist-info/RECORD +54 -0
  52. agentic_runner-2.6.0.dist-info/WHEEL +4 -0
  53. agentic_runner-2.6.0.dist-info/entry_points.txt +2 -0
  54. agentic_runner-2.6.0.dist-info/licenses/LICENSE +661 -0
@@ -0,0 +1,937 @@
1
+ """The LLM proxy, on the Runner (PRD issue 43, ADR-0013 §4 and §11, ADR-0015 §4).
2
+
3
+ The funder's key never crosses the boundary (map decisions 5 and 8), so the metering
4
+ point moves to the host that holds it. Issue 13 built the Usage Record and the ceiling
5
+ semantics against the backend proxy; this is the same semantics, on the Runner:
6
+
7
+ * **One proxy per Runner process, on loopback TCP.** The CLIs need a URL, not a socket
8
+ (17 A11). Each Directive attempt is registered under its own path segment and is
9
+ admitted only by the per-attempt bearer issue 45 already mints for the callback socket
10
+ — one token per attempt serves both — so another attempt's bearer is a 401 before any
11
+ provider is called. The Agent Runtime subprocess gets that URL and that bearer and
12
+ never a provider key (ADR-0011 §9, unchanged).
13
+ * **The slot is resolved per request, never per Directive** (22 A8). A value the funder
14
+ replaced takes effect on every running Directive at its next call, and no Directive is
15
+ stranded on a key revoked at the provider.
16
+ * **A new value is probe-gated** (22 A10): it enters service only after a zero-cost
17
+ models-list probe answers `valid`; an invalid one is refused, the old one keeps
18
+ serving, and the failure shows on the heartbeat's slot fields.
19
+ * **Ceilings are enforced per call** (map ticket 12 B4), against values the control plane
20
+ pushes. Over either one, the call is refused with issue 13's typed error and nothing is
21
+ spent. **No model degradation** — a silent swap to a cheaper model would make the
22
+ ledger lie (12 B5).
23
+ * **Usage goes out over the heartbeat stream** (ADR-0013 §11), batched, acknowledged and
24
+ retried, idempotent on ``(directive_id, sequence)``.
25
+
26
+ Counts and ids leave this module; prompts and completions do not. The only part of a
27
+ provider response read here is its ``usage`` block, which is what keeps ADR-0010 §4 true
28
+ now that the proxy is no longer control-plane code.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import asyncio
34
+ import contextlib
35
+ import json
36
+ import secrets
37
+ from collections.abc import AsyncIterator, Awaitable, Callable, Iterable, Mapping
38
+ from dataclasses import dataclass, field
39
+ from datetime import UTC, datetime
40
+ from types import TracebackType
41
+ from typing import Any, Final, Self
42
+ from uuid import UUID
43
+
44
+ import httpx
45
+
46
+ from agentic_runner.tiny_http import (
47
+ HttpRequest,
48
+ bearer_matches,
49
+ read_request,
50
+ response_head,
51
+ write_json,
52
+ )
53
+ from agentic_runner_contracts.llm_usage import (
54
+ CeilingExhaustedError,
55
+ ModelPrice,
56
+ UsageRecord,
57
+ ceiling_exhausted_detail,
58
+ resolve_price,
59
+ )
60
+ from agentic_runner_contracts.runner_registration import (
61
+ USAGE_BATCH_MAX,
62
+ SlotProbe,
63
+ SlotStatus,
64
+ )
65
+
66
+ __all__ = [
67
+ "ATTEMPT_PREFIX",
68
+ "COMPLETION_PATHS",
69
+ "CALL_REFUSED_SOURCE",
70
+ "PROXY_ENV_NAMES",
71
+ "AttemptHandle",
72
+ "CeilingStore",
73
+ "Ceilings",
74
+ "CredentialSlot",
75
+ "LlmProxy",
76
+ "SLOT_REFUSED_SOURCE",
77
+ "SLOT_SWAPPED_SOURCE",
78
+ "SlotStore",
79
+ "UsageOutbox",
80
+ "attempt_env",
81
+ ]
82
+
83
+ ATTEMPT_PREFIX: Final[str] = "/a/"
84
+
85
+ # Every name `attempt_env` may set. Reserved on a Directive's environment so no hook
86
+ # can redirect an Agent's traffic away from the metering point (`_runtime_support`).
87
+ PROXY_ENV_NAMES: Final[frozenset[str]] = frozenset(
88
+ {"ANTHROPIC_BASE_URL", "ANTHROPIC_AUTH_TOKEN", "OPENAI_BASE_URL", "OPENAI_API_KEY"}
89
+ )
90
+
91
+ # A Directive's context is the whole reason these bodies are large: Claude Code and Codex
92
+ # both send the assembled prompt on every turn, and 1 MiB (the callback socket's bound)
93
+ # is under what one long Work Record actually sends.
94
+ # ponytail: a flat 8 MiB read into memory. Streaming the request body upstream would
95
+ # avoid the copy — worth doing if a Runner's RSS ever shows it.
96
+ MAX_REQUEST_BODY_BYTES: Final[int] = 8 * 1024 * 1024
97
+
98
+ # The Evidence sources a refusal and a swap append under (issue 43: ids and a reason,
99
+ # never a value and never a body).
100
+ CALL_REFUSED_SOURCE: Final[str] = "llm_call_refused"
101
+ SLOT_SWAPPED_SOURCE: Final[str] = "llm_slot_swapped"
102
+ SLOT_REFUSED_SOURCE: Final[str] = "llm_slot_refused"
103
+
104
+ _TOKEN_BYTES: Final[int] = 32
105
+ _LOOPBACK_HOSTS: Final[frozenset[str]] = frozenset({"127.0.0.1", "::1", "localhost"})
106
+
107
+ # The completion routes, relayed to the same path at the slot's own base URL: the
108
+ # OpenAI-compatible one the backend proxy served, and Anthropic's, because Claude Code
109
+ # is one of the two harnesses the platform ships and it would otherwise still need a
110
+ # provider key of its own (ADR-0011 §9).
111
+ COMPLETION_PATHS: Final[frozenset[str]] = frozenset({"/chat/completions", "/messages"})
112
+ _SSE_DONE: Final[bytes] = b"data: [DONE]"
113
+
114
+ # Anthropic rejects a request without it, whichever credential shape the request carries.
115
+ ANTHROPIC_VERSION: Final[str] = "2023-06-01"
116
+
117
+ # What the proxy does not pass upstream: hop-by-hop headers, the attempt's own bearer
118
+ # (the slot's credential takes its place) and what httpx recomputes for the body it is
119
+ # handed. Everything else is the harness's own and is forwarded -- `anthropic-beta`
120
+ # selects the features Claude Code depends on, and dropping it silently changes the wire
121
+ # the Agent thinks it is speaking.
122
+ _HEADERS_NOT_FORWARDED: Final[frozenset[str]] = frozenset(
123
+ {
124
+ "accept-encoding",
125
+ "authorization",
126
+ "connection",
127
+ "content-length",
128
+ "host",
129
+ "keep-alive",
130
+ "proxy-authenticate",
131
+ "proxy-authorization",
132
+ "te",
133
+ "trailer",
134
+ "transfer-encoding",
135
+ "upgrade",
136
+ "x-api-key",
137
+ }
138
+ )
139
+
140
+ Evidence = Callable[[str, Mapping[str, object]], Awaitable[None]]
141
+
142
+
143
+ @dataclass(frozen=True, slots=True)
144
+ class CredentialSlot:
145
+ """One Contract's LLM credential, as the Runner holds it (ADR-0013 §11).
146
+
147
+ ``value`` is plaintext in this process's memory and nowhere else: it is never
148
+ logged, never written to disk and never put on the heartbeat — what the funder sees
149
+ is ``key_id`` and the probe result (22 A10).
150
+
151
+ ``auth_style`` is two shapes because the platform ships two harnesses: a bearer
152
+ (OpenAI-compatible endpoints, and a ``claude setup-token`` bearer) and Anthropic's
153
+ ``x-api-key``. Not a plugin point — a third provider adds a branch.
154
+ """
155
+
156
+ reference: str
157
+ key_id: str
158
+ provider_name: str
159
+ base_url: str
160
+ value: str
161
+ runtime_kind: str = "codex"
162
+ auth_style: str = "bearer"
163
+ delivered_at: datetime = field(default_factory=lambda: datetime.now(UTC))
164
+
165
+ def headers(self) -> dict[str, str]:
166
+ """The credential, plus the version header Anthropic refuses a request without.
167
+
168
+ ``anthropic-version`` goes on both auth styles, not just ``x-api-key``: a
169
+ ``claude setup-token`` bearer is the same wire, and without the header the
170
+ models-list probe answers 400 and the slot never enters service at all (22 A10).
171
+ """
172
+
173
+ headers = (
174
+ {"x-api-key": self.value}
175
+ if self.auth_style == "x-api-key"
176
+ else {"Authorization": f"Bearer {self.value}"}
177
+ )
178
+ if self.auth_style == "x-api-key" or self.runtime_kind.startswith("claude"):
179
+ headers["anthropic-version"] = ANTHROPIC_VERSION
180
+ return headers
181
+
182
+ def url(self, path: str) -> str:
183
+ return f"{self.base_url.rstrip('/')}/{path.lstrip('/')}"
184
+
185
+
186
+ SlotProber = Callable[[CredentialSlot], Awaitable[bool]]
187
+
188
+
189
+ async def models_list_probe(slot: CredentialSlot) -> bool:
190
+ """The zero-cost probe (22 A10): can this value list models at its provider?
191
+
192
+ A models list bills nothing and answers the only question that matters before a value
193
+ goes into service — whether the provider still accepts it. A network fault answers
194
+ *no*, and the caller keeps the value already serving: putting an unverified value in
195
+ on a timeout is exactly the swap this gate exists to stop.
196
+ """
197
+
198
+ try:
199
+ async with httpx.AsyncClient(timeout=10.0) as client:
200
+ response = await client.get(slot.url("models"), headers=slot.headers())
201
+ except httpx.HTTPError:
202
+ return False
203
+ return response.status_code == 200
204
+
205
+
206
+ class SlotStore:
207
+ """Every Contract's current LLM slot, resolved per request (22 A8).
208
+
209
+ Mutated by delivery, read by every call. The read is a plain dictionary lookup on
210
+ purpose: it happens inside the request, so a replacement lands on every running
211
+ Directive at its next call rather than at its next Directive.
212
+ """
213
+
214
+ def __init__(self, *, probe: SlotProber = models_list_probe) -> None:
215
+ self._probe = probe
216
+ self._slots: dict[UUID, CredentialSlot] = {}
217
+ self._probes: dict[UUID, SlotProbe] = {}
218
+ self._last_used: dict[UUID, datetime] = {}
219
+ # Leaf Contract -> the Contract whose slot funds it (22 A6). Data from issue 34;
220
+ # the proxy follows the mapping and never derives it.
221
+ self._funded_by: dict[UUID, UUID] = {}
222
+
223
+ def fund_from(self, leaf_contract_id: UUID, account_contract_id: UUID) -> None:
224
+ self._funded_by[leaf_contract_id] = account_contract_id
225
+
226
+ async def put(
227
+ self,
228
+ contract_id: UUID,
229
+ slot: CredentialSlot,
230
+ *,
231
+ evidence: Evidence | None = None,
232
+ ) -> SlotProbe:
233
+ """Probe a delivered value and swap it in only if the provider accepts it.
234
+
235
+ Refusing keeps the value already in service. The alternative — trusting delivery
236
+ and discovering the key is dead on the next call — strands every Directive on the
237
+ Contract, which is the failure 22 A10 asks the probe to prevent.
238
+ """
239
+
240
+ if not await self._probe(slot):
241
+ self._probes[contract_id] = SlotProbe.INVALID
242
+ if evidence is not None:
243
+ await evidence(
244
+ SLOT_REFUSED_SOURCE,
245
+ {
246
+ "reason": "probe_invalid",
247
+ "contract_id": str(contract_id),
248
+ "runtime_kind": slot.runtime_kind,
249
+ "key_id": slot.key_id,
250
+ "reference": slot.reference,
251
+ "probe": SlotProbe.INVALID.value,
252
+ },
253
+ )
254
+ return SlotProbe.INVALID
255
+ self._slots[contract_id] = slot
256
+ self._probes[contract_id] = SlotProbe.VALID
257
+ if evidence is not None:
258
+ await evidence(
259
+ SLOT_SWAPPED_SOURCE,
260
+ {
261
+ "contract_id": str(contract_id),
262
+ "runtime_kind": slot.runtime_kind,
263
+ "key_id": slot.key_id,
264
+ "reference": slot.reference,
265
+ "probe": SlotProbe.VALID.value,
266
+ },
267
+ )
268
+ return SlotProbe.VALID
269
+
270
+ def resolve(self, contract_id: UUID | None) -> CredentialSlot | None:
271
+ """The slot this call spends, read at the call (22 A8)."""
272
+
273
+ if contract_id is None:
274
+ return None
275
+ slot = self._slots.get(contract_id)
276
+ if slot is None:
277
+ funder = self._funded_by.get(contract_id)
278
+ if funder is not None:
279
+ slot = self._slots.get(funder)
280
+ if slot is not None:
281
+ self._last_used[contract_id] = datetime.now(UTC)
282
+ return slot
283
+
284
+ def statuses(self) -> list[SlotStatus]:
285
+ """The heartbeat's slot fields (22 A10, issue 31's ``valid|invalid|unprobed``)."""
286
+
287
+ contract_ids = sorted(set(self._slots) | set(self._probes) | set(self._funded_by), key=str)
288
+ return [
289
+ SlotStatus(
290
+ contract_id=contract_id,
291
+ runtime_kind=(
292
+ self._slots[contract_id].runtime_kind if contract_id in self._slots else "codex"
293
+ ),
294
+ present=contract_id in self._slots,
295
+ key_id=(self._slots[contract_id].key_id if contract_id in self._slots else None),
296
+ delivered_at=(
297
+ self._slots[contract_id].delivered_at if contract_id in self._slots else None
298
+ ),
299
+ last_used_at=self._last_used.get(contract_id),
300
+ probe=self._probes.get(contract_id, SlotProbe.UNPROBED),
301
+ )
302
+ for contract_id in contract_ids
303
+ ]
304
+
305
+
306
+ @dataclass(frozen=True, slots=True)
307
+ class Ceilings:
308
+ """One Contract's monthly ceilings as the control plane last stated them (12 B4).
309
+
310
+ ``used`` is the control plane's figure at the moment of the push; the Runner adds
311
+ what it has metered since. Pushing again replaces both, so a raised ceiling — or a
312
+ corrected total — applies at the very next call.
313
+ """
314
+
315
+ contract_limit: int | None = None
316
+ contract_used: int = 0
317
+ organisation_limit: int | None = None
318
+ organisation_used: int = 0
319
+ # The Organisation's ceiling covers its org-funded Contracts only: a user-funded
320
+ # Contract spends the user's own key and is none of the org's budget (map 12 B3).
321
+ org_funded: bool = False
322
+
323
+
324
+ class CeilingStore:
325
+ """What each Contract may still spend, and what it has spent here since the push."""
326
+
327
+ def __init__(self) -> None:
328
+ self._ceilings: dict[UUID, Ceilings] = {}
329
+ self._spent: dict[UUID, int] = {}
330
+
331
+ def push(self, contract_id: UUID, ceilings: Ceilings) -> None:
332
+ self._ceilings[contract_id] = ceilings
333
+ self._spent[contract_id] = 0
334
+
335
+ def spend(self, contract_id: UUID | None, tokens: int) -> None:
336
+ if contract_id is None or tokens <= 0:
337
+ return
338
+ self._spent[contract_id] = self._spent.get(contract_id, 0) + tokens
339
+
340
+ def authorize(self, contract_id: UUID | None) -> None:
341
+ """Refuse the call the month can no longer fund, before any provider is touched.
342
+
343
+ Checked against the month to date: a call may overshoot its ceiling by its own
344
+ spend and no more, the same bound the Work Record Budget carries between
345
+ Directives. A Contract with no pushed ceiling is unbounded here — the control
346
+ plane is the authority on limits, and inventing one would refuse work nobody
347
+ capped.
348
+ """
349
+
350
+ ceilings = self._ceilings.get(contract_id) if contract_id is not None else None
351
+ if ceilings is None:
352
+ return
353
+ local = self._spent.get(contract_id, 0) if contract_id is not None else 0
354
+ if ceilings.contract_limit is not None:
355
+ used = ceilings.contract_used + local
356
+ if used >= ceilings.contract_limit:
357
+ raise CeilingExhaustedError(
358
+ ceiling="contract", limit=ceilings.contract_limit, used=used
359
+ )
360
+ if not ceilings.org_funded or ceilings.organisation_limit is None:
361
+ return
362
+ used = ceilings.organisation_used + local
363
+ if used >= ceilings.organisation_limit:
364
+ raise CeilingExhaustedError(
365
+ ceiling="organisation", limit=ceilings.organisation_limit, used=used
366
+ )
367
+
368
+
369
+ class UsageOutbox:
370
+ """Usage Records waiting for the heartbeat that carries them out (ADR-0013 §11).
371
+
372
+ At-least-once by construction: a record stays here until an acknowledgement names
373
+ it, so a dropped ack costs a re-send. ``(directive_id, sequence)`` is what the ledger
374
+ de-duplicates on, so the re-send lands on the row it already wrote.
375
+ """
376
+
377
+ def __init__(self, *, capacity: int = 10_000) -> None:
378
+ self._capacity = capacity
379
+ self._pending: dict[str, UsageRecord] = {}
380
+ self._sequences: dict[str, int] = {}
381
+
382
+ def next_sequence(self, directive_id: str) -> int:
383
+ # ponytail: one counter per `directive_id`, kept for the process's life and never
384
+ # pruned. A restart resets them, so a Directive still running resumes at 0 and the
385
+ # ledger reads the new rows as duplicates of the ones it already committed --
386
+ # the idempotency key becoming a data-loss key. The upgrade path is a
387
+ # restart-unique prefix (the Runner's registration id) on the key, which changes
388
+ # the wire shape both ends de-duplicate on and so belongs with the heartbeat loop.
389
+ sequence = self._sequences.get(directive_id, 0)
390
+ self._sequences[directive_id] = sequence + 1
391
+ return sequence
392
+
393
+ def record(self, record: UsageRecord) -> None:
394
+ if len(self._pending) >= self._capacity and record.key not in self._pending:
395
+ # Oldest first: a Runner that cannot reach the control plane for long enough
396
+ # to fill this has a bigger problem than the tail of its own ledger, and
397
+ # unbounded growth would take the process down with it.
398
+ self._pending.pop(next(iter(self._pending)))
399
+ self._pending[record.key] = record
400
+
401
+ def pending(self, limit: int = USAGE_BATCH_MAX) -> list[UsageRecord]:
402
+ return list(self._pending.values())[:limit]
403
+
404
+ def acknowledge(self, keys: list[str]) -> int:
405
+ return sum(1 for key in keys if self._pending.pop(key, None) is not None)
406
+
407
+ def __len__(self) -> int:
408
+ return len(self._pending)
409
+
410
+
411
+ @dataclass(frozen=True, slots=True)
412
+ class AttemptHandle:
413
+ """What one Directive attempt is given: a URL under its own path, and its bearer."""
414
+
415
+ attempt_id: str
416
+ token: str
417
+ base_url: str
418
+ directive_id: str
419
+ contract_id: UUID | None
420
+ agent_id: UUID | None
421
+ work_record_id: UUID | None
422
+ # A per-attempt cap on top of the Contract's ceilings: the Learning reserve (PRD issue
423
+ # 55, map 12 B6). None is no cap -- every ordinary Directive.
424
+ reserve_max_tokens: int | None = None
425
+
426
+ def env(self, cli_kind: str) -> dict[str, str]:
427
+ return attempt_env(cli_kind, base_url=self.base_url, token=self.token)
428
+
429
+
430
+ def attempt_env(cli_kind: str, *, base_url: str, token: str) -> dict[str, str]:
431
+ """The names each harness reads its endpoint and credential from.
432
+
433
+ The "credential" here is the attempt's own bearer, which reaches only this Runner's
434
+ loopback proxy and dies with the attempt — so ADR-0011 §9 still holds: the subprocess
435
+ has no provider key, and anything it sends is metered and ceiling-checked on the way
436
+ out.
437
+ """
438
+
439
+ if cli_kind.startswith("claude"):
440
+ # Claude Code appends `/v1/messages` to what it is given; the OpenAI-compatible
441
+ # CLIs are configured with the `/v1` already on. Same attempt, same bearer — only
442
+ # the half of the URL each harness expects to supply differs.
443
+ return {"ANTHROPIC_BASE_URL": base_url, "ANTHROPIC_AUTH_TOKEN": token}
444
+ return {"OPENAI_BASE_URL": f"{base_url}/v1", "OPENAI_API_KEY": token}
445
+
446
+
447
+ class LlmProxy:
448
+ """One Runner process's proxy: loopback TCP, per-attempt bearer, metered per call."""
449
+
450
+ def __init__(
451
+ self,
452
+ *,
453
+ slots: SlotStore,
454
+ ceilings: CeilingStore | None = None,
455
+ outbox: UsageOutbox | None = None,
456
+ prices: dict[str, ModelPrice] | None = None,
457
+ evidence: Evidence | None = None,
458
+ client: httpx.AsyncClient | None = None,
459
+ host: str = "127.0.0.1",
460
+ port: int = 0,
461
+ ) -> None:
462
+ if host not in _LOOPBACK_HOSTS:
463
+ # The proxy resolves the funder's key on every request. Its only legitimate
464
+ # callers are subprocesses of this same process, so a bind anyone else can
465
+ # reach is refused here rather than left to a network policy to catch.
466
+ raise ValueError(f"the LLM proxy binds loopback only, not {host!r}")
467
+ self.slots = slots
468
+ self.ceilings = ceilings or CeilingStore()
469
+ self.outbox = outbox or UsageOutbox()
470
+ self._prices = prices or {}
471
+ self._evidence = evidence
472
+ self._client = client
473
+ self._owns_client = client is None
474
+ self._host = host
475
+ self._port = port
476
+ self._server: asyncio.AbstractServer | None = None
477
+ self._attempts: dict[str, AttemptHandle] = {}
478
+ # Tokens each capped attempt has spent, keyed by attempt id; gone with the attempt.
479
+ self._reserve_spent: dict[str, int] = {}
480
+
481
+ @property
482
+ def base_url(self) -> str:
483
+ return f"http://{self._host}:{self._port}"
484
+
485
+ async def __aenter__(self) -> Self:
486
+ if self._client is None:
487
+ self._client = httpx.AsyncClient(timeout=httpx.Timeout(300.0, connect=10.0))
488
+ self._server = await asyncio.start_server(self._serve, host=self._host, port=self._port)
489
+ # Port 0 by default: the operator configures no port for a surface only this
490
+ # process's own children ever dial.
491
+ self._port = self._server.sockets[0].getsockname()[1]
492
+ return self
493
+
494
+ async def __aexit__(
495
+ self,
496
+ exc_type: type[BaseException] | None,
497
+ exc: BaseException | None,
498
+ traceback: TracebackType | None,
499
+ ) -> None:
500
+ server, self._server = self._server, None
501
+ if server is not None:
502
+ server.close()
503
+ await server.wait_closed()
504
+ if self._owns_client and self._client is not None:
505
+ await self._client.aclose()
506
+ self._client = None
507
+
508
+ @contextlib.asynccontextmanager
509
+ async def attempt(
510
+ self,
511
+ *,
512
+ directive_id: str,
513
+ contract_id: UUID | None,
514
+ agent_id: UUID | None = None,
515
+ work_record_id: UUID | None = None,
516
+ token: str | None = None,
517
+ reserve_max_tokens: int | None = None,
518
+ ) -> AsyncIterator[AttemptHandle]:
519
+ """Admit one Directive attempt for as long as it runs, and no longer.
520
+
521
+ ``token`` is the attempt's callback bearer (issue 45) when there is one: one
522
+ token per attempt serves both surfaces, so the subprocess holds exactly one
523
+ secret and it expires with the attempt either way.
524
+ """
525
+
526
+ attempt_id = secrets.token_urlsafe(9)
527
+ handle = AttemptHandle(
528
+ attempt_id=attempt_id,
529
+ token=token or secrets.token_urlsafe(_TOKEN_BYTES),
530
+ base_url=f"{self.base_url}{ATTEMPT_PREFIX}{attempt_id}",
531
+ directive_id=directive_id,
532
+ contract_id=contract_id,
533
+ agent_id=agent_id,
534
+ work_record_id=work_record_id,
535
+ reserve_max_tokens=reserve_max_tokens,
536
+ )
537
+ self._attempts[attempt_id] = handle
538
+ try:
539
+ yield handle
540
+ finally:
541
+ self._attempts.pop(attempt_id, None)
542
+ self._reserve_spent.pop(attempt_id, None)
543
+
544
+ # ----------------------------------------------------------------- serving
545
+
546
+ async def _serve(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
547
+ try:
548
+ await self._route(reader, writer)
549
+ except Exception as error: # noqa: BLE001 - one bad call never takes the proxy down
550
+ await write_json(writer, 500, {"detail": error.__class__.__name__})
551
+ finally:
552
+ writer.close()
553
+ with contextlib.suppress(ConnectionResetError, BrokenPipeError):
554
+ await writer.wait_closed()
555
+
556
+ async def _route(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
557
+ request = await read_request(reader, max_body_bytes=MAX_REQUEST_BODY_BYTES)
558
+ if not isinstance(request, HttpRequest):
559
+ await write_json(writer, *request)
560
+ return
561
+
562
+ attempt, path = self._authorise(request)
563
+ if attempt is None:
564
+ # One answer for an unknown attempt and for a wrong bearer: which of the two
565
+ # it was is not the caller's business.
566
+ await write_json(writer, 401, {"detail": "a valid attempt bearer is required"})
567
+ return
568
+
569
+ # The route is checked before the slot is resolved: an unknown path is not a
570
+ # funding failure, and answering one with a 503 and an Evidence Event would let a
571
+ # mistyped URL write Evidence.
572
+ models_list = request.method == "GET" and path == "/models"
573
+ if not models_list and (request.method != "POST" or path not in COMPLETION_PATHS):
574
+ await write_json(writer, 404, {"detail": "no such proxy route"})
575
+ return
576
+
577
+ slot = self.slots.resolve(attempt.contract_id)
578
+ if slot is None:
579
+ await self._refused(attempt, reason="slot_missing")
580
+ await write_json(
581
+ writer, 503, {"error": {"message": "no LLM slot", "type": "slot_unavailable"}}
582
+ )
583
+ return
584
+
585
+ if models_list:
586
+ await self._relay_models(writer, slot=slot)
587
+ return
588
+
589
+ try:
590
+ self.ceilings.authorize(attempt.contract_id)
591
+ except CeilingExhaustedError as error:
592
+ await self._refused(attempt, reason="ceiling_exhausted", ceiling=error.ceiling)
593
+ await write_json(writer, 402, ceiling_exhausted_detail(error))
594
+ return
595
+
596
+ spent = self._reserve_spent.get(attempt.attempt_id, 0)
597
+ if attempt.reserve_max_tokens is not None and spent >= attempt.reserve_max_tokens:
598
+ await self._refused(attempt, reason="reserve_exhausted", used=spent)
599
+ await write_json(
600
+ writer,
601
+ 402,
602
+ {
603
+ "error": {
604
+ "message": (
605
+ f"reserve exhausted: {spent} of {attempt.reserve_max_tokens} "
606
+ "tokens spent"
607
+ ),
608
+ "type": "reserve_exhausted",
609
+ }
610
+ },
611
+ )
612
+ return
613
+
614
+ await self._relay_completion(writer, request=request, attempt=attempt, slot=slot, path=path)
615
+
616
+ def _authorise(self, request: HttpRequest) -> tuple[AttemptHandle | None, str]:
617
+ """Which attempt this request is, by path *and* bearer.
618
+
619
+ The path segment is what makes another attempt's bearer a refusal rather than a
620
+ mis-attribution: a token that is valid for some other live attempt does not open
621
+ this one.
622
+ """
623
+
624
+ if not request.target.startswith(ATTEMPT_PREFIX):
625
+ return None, ""
626
+ attempt_id, _, rest = request.target[len(ATTEMPT_PREFIX) :].partition("/")
627
+ attempt = self._attempts.get(attempt_id)
628
+ if attempt is None or not bearer_matches(request.authorization, attempt.token):
629
+ return None, ""
630
+ # The `/v1` is the harness's, not ours: Claude Code appends it and the
631
+ # OpenAI-compatible CLIs carry it in the base URL they were configured with.
632
+ path = "/" + rest.removeprefix("v1/").lstrip("/")
633
+ return attempt, path.partition("?")[0]
634
+
635
+ async def _relay_models(self, writer: asyncio.StreamWriter, *, slot: CredentialSlot) -> None:
636
+ assert self._client is not None
637
+ try:
638
+ upstream = await self._client.get(slot.url("models"), headers=slot.headers())
639
+ except httpx.HTTPError as error:
640
+ await write_json(
641
+ writer,
642
+ 502,
643
+ {"error": {"message": "provider unreachable", "type": error.__class__.__name__}},
644
+ )
645
+ return
646
+ writer.write(
647
+ response_head(
648
+ upstream.status_code,
649
+ content_type="application/json",
650
+ content_length=len(upstream.content),
651
+ )
652
+ )
653
+ writer.write(upstream.content)
654
+ with contextlib.suppress(ConnectionResetError, BrokenPipeError):
655
+ await writer.drain()
656
+
657
+ async def _relay_completion(
658
+ self,
659
+ writer: asyncio.StreamWriter,
660
+ *,
661
+ request: HttpRequest,
662
+ attempt: AttemptHandle,
663
+ slot: CredentialSlot,
664
+ path: str,
665
+ ) -> None:
666
+ assert self._client is not None
667
+ payload = _json_object(request.body)
668
+ model_name = str(payload.get("model") or "unknown")
669
+ streaming = payload.get("stream") is True
670
+ url = slot.url(path)
671
+ headers = _upstream_headers(request, slot)
672
+ body = request.body
673
+ if streaming and path == "/chat/completions":
674
+ # An OpenAI-compatible endpoint omits the usage block from a stream unless it
675
+ # is asked for, and a streamed call metered at zero is a ledger row missing
676
+ # its dominant term and a per-call ceiling that cannot be enforced. Anthropic
677
+ # sends the counts either way, split across two frames (`_sse_usage`).
678
+ body = _with_usage_in_stream(payload)
679
+ started = asyncio.get_running_loop().time()
680
+
681
+ if not streaming:
682
+ try:
683
+ upstream = await self._client.post(url, content=body, headers=headers)
684
+ except httpx.HTTPError as error:
685
+ self._meter(
686
+ attempt,
687
+ slot=slot,
688
+ model_name=model_name,
689
+ usage={},
690
+ started=started,
691
+ error_class=error.__class__.__name__,
692
+ )
693
+ await write_json(
694
+ writer,
695
+ 502,
696
+ {"error": {"message": "provider unreachable", "type": "provider_error"}},
697
+ )
698
+ return
699
+ self._meter(
700
+ attempt,
701
+ slot=slot,
702
+ model_name=model_name,
703
+ usage=_usage_block(_json_object(upstream.content)),
704
+ started=started,
705
+ error_class=None if upstream.status_code < 400 else f"http_{upstream.status_code}",
706
+ )
707
+ writer.write(
708
+ response_head(
709
+ upstream.status_code,
710
+ content_type="application/json",
711
+ content_length=len(upstream.content),
712
+ )
713
+ )
714
+ writer.write(upstream.content)
715
+ with contextlib.suppress(ConnectionResetError, BrokenPipeError):
716
+ await writer.drain()
717
+ return
718
+
719
+ usage: dict[str, object] = {}
720
+ error_class: str | None = None
721
+ started_relaying = False
722
+ tail = b""
723
+ try:
724
+ async with self._client.stream("POST", url, content=body, headers=headers) as upstream:
725
+ writer.write(response_head(upstream.status_code, content_type="text/event-stream"))
726
+ started_relaying = True
727
+ # The same code the non-streaming branch records: a streamed call the
728
+ # provider refused meters no tokens, and a row with none and no error
729
+ # class reads as a successful free call.
730
+ if upstream.status_code >= 400:
731
+ error_class = f"http_{upstream.status_code}"
732
+ async for chunk in upstream.aiter_bytes():
733
+ writer.write(chunk)
734
+ # Lines are re-assembled across chunk boundaries: a `data:` frame the
735
+ # transport split in two carries its counts in neither half.
736
+ *lines, tail = (tail + chunk).split(b"\n")
737
+ usage.update(_sse_usage(lines))
738
+ with contextlib.suppress(ConnectionResetError, BrokenPipeError):
739
+ await writer.drain()
740
+ except httpx.HTTPError as error:
741
+ error_class = error.__class__.__name__
742
+ usage.update(_sse_usage([tail]))
743
+ self._meter(
744
+ attempt,
745
+ slot=slot,
746
+ model_name=model_name,
747
+ usage=usage,
748
+ started=started,
749
+ error_class=error_class,
750
+ )
751
+ if error_class is not None and not started_relaying:
752
+ # Nothing has been written yet, so the caller can still be told why rather
753
+ # than being handed a connection that simply closes.
754
+ await write_json(
755
+ writer,
756
+ 502,
757
+ {"error": {"message": "provider unreachable", "type": "provider_error"}},
758
+ )
759
+
760
+ # ----------------------------------------------------------------- metering
761
+
762
+ def _meter(
763
+ self,
764
+ attempt: AttemptHandle,
765
+ *,
766
+ slot: CredentialSlot,
767
+ model_name: str,
768
+ usage: Mapping[str, object],
769
+ started: float,
770
+ error_class: str | None,
771
+ ) -> None:
772
+ """One Usage Record per call, and the tokens the ceiling now counts.
773
+
774
+ Reads the upstream ``usage`` block and nothing else — the completion itself is
775
+ forwarded and forgotten, which is the rule the structural test pins.
776
+ """
777
+
778
+ prompt_tokens = _prompt_tokens(usage)
779
+ completion_tokens = _completion_tokens(usage)
780
+ price = resolve_price(self._prices, provider_name=slot.provider_name, model_name=model_name)
781
+ record = UsageRecord(
782
+ directive_id=attempt.directive_id,
783
+ sequence=self.outbox.next_sequence(attempt.directive_id),
784
+ provider_name=slot.provider_name,
785
+ model_name=model_name,
786
+ prompt_tokens=prompt_tokens,
787
+ completion_tokens=completion_tokens,
788
+ cached_tokens=_cached_tokens(usage),
789
+ applied_prompt_price_usd=price.prompt_usd_per_token,
790
+ applied_completion_price_usd=price.completion_usd_per_token,
791
+ cost_usd=price.cost_usd(
792
+ prompt_tokens=prompt_tokens or 0, completion_tokens=completion_tokens or 0
793
+ ),
794
+ latency_ms=max(0, int((asyncio.get_running_loop().time() - started) * 1000)),
795
+ error_class=error_class,
796
+ agent_id=attempt.agent_id,
797
+ contract_id=attempt.contract_id,
798
+ work_record_id=attempt.work_record_id,
799
+ )
800
+ self.outbox.record(record)
801
+ tokens = (prompt_tokens or 0) + (completion_tokens or 0)
802
+ self.ceilings.spend(attempt.contract_id, tokens)
803
+ if attempt.reserve_max_tokens is not None:
804
+ self._reserve_spent[attempt.attempt_id] = (
805
+ self._reserve_spent.get(attempt.attempt_id, 0) + tokens
806
+ )
807
+
808
+ async def _refused(self, attempt: AttemptHandle, *, reason: str, **extra: object) -> None:
809
+ """A refused call is an Evidence Event with the reason — ids only (issue 43)."""
810
+
811
+ if self._evidence is None:
812
+ return
813
+ await self._evidence(
814
+ CALL_REFUSED_SOURCE,
815
+ {
816
+ "reason": reason,
817
+ "directive_id": attempt.directive_id,
818
+ "contract_id": str(attempt.contract_id) if attempt.contract_id else None,
819
+ "agent_id": str(attempt.agent_id) if attempt.agent_id else None,
820
+ **extra,
821
+ },
822
+ )
823
+
824
+
825
+ def _json_object(raw: bytes) -> dict[str, Any]:
826
+ try:
827
+ parsed = json.loads(raw or b"{}")
828
+ except ValueError:
829
+ return {}
830
+ return parsed if isinstance(parsed, dict) else {}
831
+
832
+
833
+ def _upstream_headers(request: HttpRequest, slot: CredentialSlot) -> dict[str, str]:
834
+ """The harness's own headers, with the slot's credential in place of its bearer.
835
+
836
+ Forwarded rather than rebuilt: ``anthropic-beta`` is how Claude Code asks for the
837
+ features it is built against, and a proxy that drops it answers a different call than
838
+ the one the Agent made.
839
+ """
840
+
841
+ forwarded = {
842
+ name: value for name, value in request.headers.items() if name not in _HEADERS_NOT_FORWARDED
843
+ }
844
+ # The harness's headers win over the slot's: the credential cannot be among them (both
845
+ # auth headers are stripped above), and a harness that asks for a newer
846
+ # `anthropic-version` than the slot's floor must get the wire it asked for.
847
+ return {"content-type": "application/json", **slot.headers(), **forwarded}
848
+
849
+
850
+ def _with_usage_in_stream(payload: dict[str, Any]) -> bytes:
851
+ """``stream_options.include_usage``, without discarding one the caller already set."""
852
+
853
+ options = payload.get("stream_options")
854
+ payload["stream_options"] = {
855
+ **(options if isinstance(options, Mapping) else {}),
856
+ "include_usage": True,
857
+ }
858
+ return json.dumps(payload).encode()
859
+
860
+
861
+ def _usage_block(body: Mapping[str, object]) -> Mapping[str, object]:
862
+ usage = body.get("usage")
863
+ if isinstance(usage, Mapping):
864
+ return usage
865
+ # Anthropic's `message_start` frame nests the input counts one level down, under the
866
+ # message it is starting. `message` is followed only to reach that `usage` block --
867
+ # nothing else in the frame is read (ADR-0010 §4).
868
+ message = body.get("message")
869
+ if isinstance(message, Mapping):
870
+ nested = message.get("usage")
871
+ if isinstance(nested, Mapping):
872
+ return nested
873
+ return {}
874
+
875
+
876
+ def _sse_usage(lines: Iterable[bytes]) -> dict[str, object]:
877
+ """The ``usage`` counts these SSE lines carried, merged rather than replaced.
878
+
879
+ One call's counts can arrive on more than one frame: Anthropic puts the input tokens
880
+ on ``message_start`` and the output tokens on ``message_delta``. Keeping only the last
881
+ block seen would meter a streamed ``/v1/messages`` call with no prompt tokens at all
882
+ -- the dominant term of both the ledger row and the ceiling -- and Claude Code streams
883
+ by default, so that is the normal path for one of the two harnesses.
884
+ """
885
+
886
+ found: dict[str, object] = {}
887
+ for raw in lines:
888
+ line = raw.strip()
889
+ if not line.startswith(b"data:") or line.startswith(_SSE_DONE):
890
+ continue
891
+ found.update(_usage_block(_json_object(line[len(b"data:") :].strip())))
892
+ return found
893
+
894
+
895
+ def _usage_int(usage: Mapping[str, object], key: str) -> int | None:
896
+ value = usage.get(key)
897
+ return value if isinstance(value, int) else None
898
+
899
+
900
+ def _first_int(usage: Mapping[str, object], *keys: str) -> int | None:
901
+ """The first of ``keys`` this usage block carries.
902
+
903
+ Two wires, one Usage Record: OpenAI-compatible endpoints say ``prompt_tokens`` and
904
+ Anthropic says ``input_tokens``. The ledger and the ceilings count tokens, so the
905
+ difference is a spelling and is resolved here rather than in the row.
906
+ """
907
+
908
+ for key in keys:
909
+ value = _usage_int(usage, key)
910
+ if value is not None:
911
+ return value
912
+ return None
913
+
914
+
915
+ def _prompt_tokens(usage: Mapping[str, object]) -> int | None:
916
+ tokens = _first_int(usage, "prompt_tokens", "input_tokens")
917
+ if tokens is None:
918
+ return None
919
+ # Cache *writes* are new input the model had to encode, so they are prompt tokens —
920
+ # the same split `claude_code_harness_usage` already prices (PRD issue 31).
921
+ return tokens + (_usage_int(usage, "cache_creation_input_tokens") or 0)
922
+
923
+
924
+ def _completion_tokens(usage: Mapping[str, object]) -> int | None:
925
+ return _first_int(usage, "completion_tokens", "output_tokens")
926
+
927
+
928
+ def _cached_tokens(usage: Mapping[str, object]) -> int | None:
929
+ """Cached input, wherever the provider puts it. Counted at full weight (map 12 B1)."""
930
+
931
+ direct = _first_int(usage, "cached_tokens", "cache_read_input_tokens")
932
+ if direct is not None:
933
+ return direct
934
+ details = usage.get("prompt_tokens_details")
935
+ if isinstance(details, Mapping):
936
+ return _usage_int(details, "cached_tokens")
937
+ return None