agentic-runner 2.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentic_runner/__init__.py +12 -0
- agentic_runner/activities.py +4918 -0
- agentic_runner/callback.py +342 -0
- agentic_runner/child_watcher.py +66 -0
- agentic_runner/cli.py +416 -0
- agentic_runner/config.py +105 -0
- agentic_runner/credentials.py +252 -0
- agentic_runner/device_login_activities.py +79 -0
- agentic_runner/egress.py +243 -0
- agentic_runner/heartbeat_link.py +249 -0
- agentic_runner/hooks.py +455 -0
- agentic_runner/host_store.py +295 -0
- agentic_runner/integrations/__init__.py +0 -0
- agentic_runner/integrations/git/__init__.py +1 -0
- agentic_runner/integrations/git/contracts.py +198 -0
- agentic_runner/integrations/git/evidence.py +442 -0
- agentic_runner/integrations/git/fake_workspace.py +339 -0
- agentic_runner/integrations/git/workspace.py +921 -0
- agentic_runner/integrations/github/__init__.py +53 -0
- agentic_runner/integrations/github/auth.py +171 -0
- agentic_runner/integrations/github/fake_client.py +494 -0
- agentic_runner/integrations/github/gh_client.py +944 -0
- agentic_runner/lifecycle.py +48 -0
- agentic_runner/llm_proxy.py +937 -0
- agentic_runner/mcp.py +342 -0
- agentic_runner/message_store.py +341 -0
- agentic_runner/py.typed +0 -0
- agentic_runner/recipient_key_secret.py +134 -0
- agentic_runner/registration.py +363 -0
- agentic_runner/runtime/__init__.py +0 -0
- agentic_runner/runtime/verifier_command.py +344 -0
- agentic_runner/sealed_box.py +509 -0
- agentic_runner/service.py +1068 -0
- agentic_runner/tiny_http.py +133 -0
- agentic_runner/triage_activities.py +113 -0
- agentic_runner/user_sources.py +546 -0
- agentic_runner/workers/__init__.py +1 -0
- agentic_runner/workers/_runtime_support.py +388 -0
- agentic_runner/workers/agent_runtime.py +93 -0
- agentic_runner/workers/claude_runtime.py +226 -0
- agentic_runner/workers/codex_runtime.py +311 -0
- agentic_runner/workers/command_policy.py +250 -0
- agentic_runner/workers/contract_device_login.py +211 -0
- agentic_runner/workers/contract_isolation.py +500 -0
- agentic_runner/workers/fastapi_client.py +396 -0
- agentic_runner/workers/harness_usage.py +65 -0
- agentic_runner/workers/mcp_config.py +111 -0
- agentic_runner/workers/settings.py +314 -0
- agentic_runner/workstation.py +687 -0
- agentic_runner-2.6.0.dist-info/METADATA +49 -0
- agentic_runner-2.6.0.dist-info/RECORD +54 -0
- agentic_runner-2.6.0.dist-info/WHEEL +4 -0
- agentic_runner-2.6.0.dist-info/entry_points.txt +2 -0
- agentic_runner-2.6.0.dist-info/licenses/LICENSE +661 -0
|
@@ -0,0 +1,937 @@
|
|
|
1
|
+
"""The LLM proxy, on the Runner (PRD issue 43, ADR-0013 §4 and §11, ADR-0015 §4).
|
|
2
|
+
|
|
3
|
+
The funder's key never crosses the boundary (map decisions 5 and 8), so the metering
|
|
4
|
+
point moves to the host that holds it. Issue 13 built the Usage Record and the ceiling
|
|
5
|
+
semantics against the backend proxy; this is the same semantics, on the Runner:
|
|
6
|
+
|
|
7
|
+
* **One proxy per Runner process, on loopback TCP.** The CLIs need a URL, not a socket
|
|
8
|
+
(17 A11). Each Directive attempt is registered under its own path segment and is
|
|
9
|
+
admitted only by the per-attempt bearer issue 45 already mints for the callback socket
|
|
10
|
+
— one token per attempt serves both — so another attempt's bearer is a 401 before any
|
|
11
|
+
provider is called. The Agent Runtime subprocess gets that URL and that bearer and
|
|
12
|
+
never a provider key (ADR-0011 §9, unchanged).
|
|
13
|
+
* **The slot is resolved per request, never per Directive** (22 A8). A value the funder
|
|
14
|
+
replaced takes effect on every running Directive at its next call, and no Directive is
|
|
15
|
+
stranded on a key revoked at the provider.
|
|
16
|
+
* **A new value is probe-gated** (22 A10): it enters service only after a zero-cost
|
|
17
|
+
models-list probe answers `valid`; an invalid one is refused, the old one keeps
|
|
18
|
+
serving, and the failure shows on the heartbeat's slot fields.
|
|
19
|
+
* **Ceilings are enforced per call** (map ticket 12 B4), against values the control plane
|
|
20
|
+
pushes. Over either one, the call is refused with issue 13's typed error and nothing is
|
|
21
|
+
spent. **No model degradation** — a silent swap to a cheaper model would make the
|
|
22
|
+
ledger lie (12 B5).
|
|
23
|
+
* **Usage goes out over the heartbeat stream** (ADR-0013 §11), batched, acknowledged and
|
|
24
|
+
retried, idempotent on ``(directive_id, sequence)``.
|
|
25
|
+
|
|
26
|
+
Counts and ids leave this module; prompts and completions do not. The only part of a
|
|
27
|
+
provider response read here is its ``usage`` block, which is what keeps ADR-0010 §4 true
|
|
28
|
+
now that the proxy is no longer control-plane code.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import asyncio
|
|
34
|
+
import contextlib
|
|
35
|
+
import json
|
|
36
|
+
import secrets
|
|
37
|
+
from collections.abc import AsyncIterator, Awaitable, Callable, Iterable, Mapping
|
|
38
|
+
from dataclasses import dataclass, field
|
|
39
|
+
from datetime import UTC, datetime
|
|
40
|
+
from types import TracebackType
|
|
41
|
+
from typing import Any, Final, Self
|
|
42
|
+
from uuid import UUID
|
|
43
|
+
|
|
44
|
+
import httpx
|
|
45
|
+
|
|
46
|
+
from agentic_runner.tiny_http import (
|
|
47
|
+
HttpRequest,
|
|
48
|
+
bearer_matches,
|
|
49
|
+
read_request,
|
|
50
|
+
response_head,
|
|
51
|
+
write_json,
|
|
52
|
+
)
|
|
53
|
+
from agentic_runner_contracts.llm_usage import (
|
|
54
|
+
CeilingExhaustedError,
|
|
55
|
+
ModelPrice,
|
|
56
|
+
UsageRecord,
|
|
57
|
+
ceiling_exhausted_detail,
|
|
58
|
+
resolve_price,
|
|
59
|
+
)
|
|
60
|
+
from agentic_runner_contracts.runner_registration import (
|
|
61
|
+
USAGE_BATCH_MAX,
|
|
62
|
+
SlotProbe,
|
|
63
|
+
SlotStatus,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
__all__ = [
|
|
67
|
+
"ATTEMPT_PREFIX",
|
|
68
|
+
"COMPLETION_PATHS",
|
|
69
|
+
"CALL_REFUSED_SOURCE",
|
|
70
|
+
"PROXY_ENV_NAMES",
|
|
71
|
+
"AttemptHandle",
|
|
72
|
+
"CeilingStore",
|
|
73
|
+
"Ceilings",
|
|
74
|
+
"CredentialSlot",
|
|
75
|
+
"LlmProxy",
|
|
76
|
+
"SLOT_REFUSED_SOURCE",
|
|
77
|
+
"SLOT_SWAPPED_SOURCE",
|
|
78
|
+
"SlotStore",
|
|
79
|
+
"UsageOutbox",
|
|
80
|
+
"attempt_env",
|
|
81
|
+
]
|
|
82
|
+
|
|
83
|
+
ATTEMPT_PREFIX: Final[str] = "/a/"
|
|
84
|
+
|
|
85
|
+
# Every name `attempt_env` may set. Reserved on a Directive's environment so no hook
|
|
86
|
+
# can redirect an Agent's traffic away from the metering point (`_runtime_support`).
|
|
87
|
+
PROXY_ENV_NAMES: Final[frozenset[str]] = frozenset(
|
|
88
|
+
{"ANTHROPIC_BASE_URL", "ANTHROPIC_AUTH_TOKEN", "OPENAI_BASE_URL", "OPENAI_API_KEY"}
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
# A Directive's context is the whole reason these bodies are large: Claude Code and Codex
|
|
92
|
+
# both send the assembled prompt on every turn, and 1 MiB (the callback socket's bound)
|
|
93
|
+
# is under what one long Work Record actually sends.
|
|
94
|
+
# ponytail: a flat 8 MiB read into memory. Streaming the request body upstream would
|
|
95
|
+
# avoid the copy — worth doing if a Runner's RSS ever shows it.
|
|
96
|
+
MAX_REQUEST_BODY_BYTES: Final[int] = 8 * 1024 * 1024
|
|
97
|
+
|
|
98
|
+
# The Evidence sources a refusal and a swap append under (issue 43: ids and a reason,
|
|
99
|
+
# never a value and never a body).
|
|
100
|
+
CALL_REFUSED_SOURCE: Final[str] = "llm_call_refused"
|
|
101
|
+
SLOT_SWAPPED_SOURCE: Final[str] = "llm_slot_swapped"
|
|
102
|
+
SLOT_REFUSED_SOURCE: Final[str] = "llm_slot_refused"
|
|
103
|
+
|
|
104
|
+
_TOKEN_BYTES: Final[int] = 32
|
|
105
|
+
_LOOPBACK_HOSTS: Final[frozenset[str]] = frozenset({"127.0.0.1", "::1", "localhost"})
|
|
106
|
+
|
|
107
|
+
# The completion routes, relayed to the same path at the slot's own base URL: the
|
|
108
|
+
# OpenAI-compatible one the backend proxy served, and Anthropic's, because Claude Code
|
|
109
|
+
# is one of the two harnesses the platform ships and it would otherwise still need a
|
|
110
|
+
# provider key of its own (ADR-0011 §9).
|
|
111
|
+
COMPLETION_PATHS: Final[frozenset[str]] = frozenset({"/chat/completions", "/messages"})
|
|
112
|
+
_SSE_DONE: Final[bytes] = b"data: [DONE]"
|
|
113
|
+
|
|
114
|
+
# Anthropic rejects a request without it, whichever credential shape the request carries.
|
|
115
|
+
ANTHROPIC_VERSION: Final[str] = "2023-06-01"
|
|
116
|
+
|
|
117
|
+
# What the proxy does not pass upstream: hop-by-hop headers, the attempt's own bearer
|
|
118
|
+
# (the slot's credential takes its place) and what httpx recomputes for the body it is
|
|
119
|
+
# handed. Everything else is the harness's own and is forwarded -- `anthropic-beta`
|
|
120
|
+
# selects the features Claude Code depends on, and dropping it silently changes the wire
|
|
121
|
+
# the Agent thinks it is speaking.
|
|
122
|
+
_HEADERS_NOT_FORWARDED: Final[frozenset[str]] = frozenset(
|
|
123
|
+
{
|
|
124
|
+
"accept-encoding",
|
|
125
|
+
"authorization",
|
|
126
|
+
"connection",
|
|
127
|
+
"content-length",
|
|
128
|
+
"host",
|
|
129
|
+
"keep-alive",
|
|
130
|
+
"proxy-authenticate",
|
|
131
|
+
"proxy-authorization",
|
|
132
|
+
"te",
|
|
133
|
+
"trailer",
|
|
134
|
+
"transfer-encoding",
|
|
135
|
+
"upgrade",
|
|
136
|
+
"x-api-key",
|
|
137
|
+
}
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
Evidence = Callable[[str, Mapping[str, object]], Awaitable[None]]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
@dataclass(frozen=True, slots=True)
|
|
144
|
+
class CredentialSlot:
|
|
145
|
+
"""One Contract's LLM credential, as the Runner holds it (ADR-0013 §11).
|
|
146
|
+
|
|
147
|
+
``value`` is plaintext in this process's memory and nowhere else: it is never
|
|
148
|
+
logged, never written to disk and never put on the heartbeat — what the funder sees
|
|
149
|
+
is ``key_id`` and the probe result (22 A10).
|
|
150
|
+
|
|
151
|
+
``auth_style`` is two shapes because the platform ships two harnesses: a bearer
|
|
152
|
+
(OpenAI-compatible endpoints, and a ``claude setup-token`` bearer) and Anthropic's
|
|
153
|
+
``x-api-key``. Not a plugin point — a third provider adds a branch.
|
|
154
|
+
"""
|
|
155
|
+
|
|
156
|
+
reference: str
|
|
157
|
+
key_id: str
|
|
158
|
+
provider_name: str
|
|
159
|
+
base_url: str
|
|
160
|
+
value: str
|
|
161
|
+
runtime_kind: str = "codex"
|
|
162
|
+
auth_style: str = "bearer"
|
|
163
|
+
delivered_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
164
|
+
|
|
165
|
+
def headers(self) -> dict[str, str]:
|
|
166
|
+
"""The credential, plus the version header Anthropic refuses a request without.
|
|
167
|
+
|
|
168
|
+
``anthropic-version`` goes on both auth styles, not just ``x-api-key``: a
|
|
169
|
+
``claude setup-token`` bearer is the same wire, and without the header the
|
|
170
|
+
models-list probe answers 400 and the slot never enters service at all (22 A10).
|
|
171
|
+
"""
|
|
172
|
+
|
|
173
|
+
headers = (
|
|
174
|
+
{"x-api-key": self.value}
|
|
175
|
+
if self.auth_style == "x-api-key"
|
|
176
|
+
else {"Authorization": f"Bearer {self.value}"}
|
|
177
|
+
)
|
|
178
|
+
if self.auth_style == "x-api-key" or self.runtime_kind.startswith("claude"):
|
|
179
|
+
headers["anthropic-version"] = ANTHROPIC_VERSION
|
|
180
|
+
return headers
|
|
181
|
+
|
|
182
|
+
def url(self, path: str) -> str:
|
|
183
|
+
return f"{self.base_url.rstrip('/')}/{path.lstrip('/')}"
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
SlotProber = Callable[[CredentialSlot], Awaitable[bool]]
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
async def models_list_probe(slot: CredentialSlot) -> bool:
|
|
190
|
+
"""The zero-cost probe (22 A10): can this value list models at its provider?
|
|
191
|
+
|
|
192
|
+
A models list bills nothing and answers the only question that matters before a value
|
|
193
|
+
goes into service — whether the provider still accepts it. A network fault answers
|
|
194
|
+
*no*, and the caller keeps the value already serving: putting an unverified value in
|
|
195
|
+
on a timeout is exactly the swap this gate exists to stop.
|
|
196
|
+
"""
|
|
197
|
+
|
|
198
|
+
try:
|
|
199
|
+
async with httpx.AsyncClient(timeout=10.0) as client:
|
|
200
|
+
response = await client.get(slot.url("models"), headers=slot.headers())
|
|
201
|
+
except httpx.HTTPError:
|
|
202
|
+
return False
|
|
203
|
+
return response.status_code == 200
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
class SlotStore:
|
|
207
|
+
"""Every Contract's current LLM slot, resolved per request (22 A8).
|
|
208
|
+
|
|
209
|
+
Mutated by delivery, read by every call. The read is a plain dictionary lookup on
|
|
210
|
+
purpose: it happens inside the request, so a replacement lands on every running
|
|
211
|
+
Directive at its next call rather than at its next Directive.
|
|
212
|
+
"""
|
|
213
|
+
|
|
214
|
+
def __init__(self, *, probe: SlotProber = models_list_probe) -> None:
|
|
215
|
+
self._probe = probe
|
|
216
|
+
self._slots: dict[UUID, CredentialSlot] = {}
|
|
217
|
+
self._probes: dict[UUID, SlotProbe] = {}
|
|
218
|
+
self._last_used: dict[UUID, datetime] = {}
|
|
219
|
+
# Leaf Contract -> the Contract whose slot funds it (22 A6). Data from issue 34;
|
|
220
|
+
# the proxy follows the mapping and never derives it.
|
|
221
|
+
self._funded_by: dict[UUID, UUID] = {}
|
|
222
|
+
|
|
223
|
+
def fund_from(self, leaf_contract_id: UUID, account_contract_id: UUID) -> None:
|
|
224
|
+
self._funded_by[leaf_contract_id] = account_contract_id
|
|
225
|
+
|
|
226
|
+
async def put(
|
|
227
|
+
self,
|
|
228
|
+
contract_id: UUID,
|
|
229
|
+
slot: CredentialSlot,
|
|
230
|
+
*,
|
|
231
|
+
evidence: Evidence | None = None,
|
|
232
|
+
) -> SlotProbe:
|
|
233
|
+
"""Probe a delivered value and swap it in only if the provider accepts it.
|
|
234
|
+
|
|
235
|
+
Refusing keeps the value already in service. The alternative — trusting delivery
|
|
236
|
+
and discovering the key is dead on the next call — strands every Directive on the
|
|
237
|
+
Contract, which is the failure 22 A10 asks the probe to prevent.
|
|
238
|
+
"""
|
|
239
|
+
|
|
240
|
+
if not await self._probe(slot):
|
|
241
|
+
self._probes[contract_id] = SlotProbe.INVALID
|
|
242
|
+
if evidence is not None:
|
|
243
|
+
await evidence(
|
|
244
|
+
SLOT_REFUSED_SOURCE,
|
|
245
|
+
{
|
|
246
|
+
"reason": "probe_invalid",
|
|
247
|
+
"contract_id": str(contract_id),
|
|
248
|
+
"runtime_kind": slot.runtime_kind,
|
|
249
|
+
"key_id": slot.key_id,
|
|
250
|
+
"reference": slot.reference,
|
|
251
|
+
"probe": SlotProbe.INVALID.value,
|
|
252
|
+
},
|
|
253
|
+
)
|
|
254
|
+
return SlotProbe.INVALID
|
|
255
|
+
self._slots[contract_id] = slot
|
|
256
|
+
self._probes[contract_id] = SlotProbe.VALID
|
|
257
|
+
if evidence is not None:
|
|
258
|
+
await evidence(
|
|
259
|
+
SLOT_SWAPPED_SOURCE,
|
|
260
|
+
{
|
|
261
|
+
"contract_id": str(contract_id),
|
|
262
|
+
"runtime_kind": slot.runtime_kind,
|
|
263
|
+
"key_id": slot.key_id,
|
|
264
|
+
"reference": slot.reference,
|
|
265
|
+
"probe": SlotProbe.VALID.value,
|
|
266
|
+
},
|
|
267
|
+
)
|
|
268
|
+
return SlotProbe.VALID
|
|
269
|
+
|
|
270
|
+
def resolve(self, contract_id: UUID | None) -> CredentialSlot | None:
|
|
271
|
+
"""The slot this call spends, read at the call (22 A8)."""
|
|
272
|
+
|
|
273
|
+
if contract_id is None:
|
|
274
|
+
return None
|
|
275
|
+
slot = self._slots.get(contract_id)
|
|
276
|
+
if slot is None:
|
|
277
|
+
funder = self._funded_by.get(contract_id)
|
|
278
|
+
if funder is not None:
|
|
279
|
+
slot = self._slots.get(funder)
|
|
280
|
+
if slot is not None:
|
|
281
|
+
self._last_used[contract_id] = datetime.now(UTC)
|
|
282
|
+
return slot
|
|
283
|
+
|
|
284
|
+
def statuses(self) -> list[SlotStatus]:
|
|
285
|
+
"""The heartbeat's slot fields (22 A10, issue 31's ``valid|invalid|unprobed``)."""
|
|
286
|
+
|
|
287
|
+
contract_ids = sorted(set(self._slots) | set(self._probes) | set(self._funded_by), key=str)
|
|
288
|
+
return [
|
|
289
|
+
SlotStatus(
|
|
290
|
+
contract_id=contract_id,
|
|
291
|
+
runtime_kind=(
|
|
292
|
+
self._slots[contract_id].runtime_kind if contract_id in self._slots else "codex"
|
|
293
|
+
),
|
|
294
|
+
present=contract_id in self._slots,
|
|
295
|
+
key_id=(self._slots[contract_id].key_id if contract_id in self._slots else None),
|
|
296
|
+
delivered_at=(
|
|
297
|
+
self._slots[contract_id].delivered_at if contract_id in self._slots else None
|
|
298
|
+
),
|
|
299
|
+
last_used_at=self._last_used.get(contract_id),
|
|
300
|
+
probe=self._probes.get(contract_id, SlotProbe.UNPROBED),
|
|
301
|
+
)
|
|
302
|
+
for contract_id in contract_ids
|
|
303
|
+
]
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
@dataclass(frozen=True, slots=True)
|
|
307
|
+
class Ceilings:
|
|
308
|
+
"""One Contract's monthly ceilings as the control plane last stated them (12 B4).
|
|
309
|
+
|
|
310
|
+
``used`` is the control plane's figure at the moment of the push; the Runner adds
|
|
311
|
+
what it has metered since. Pushing again replaces both, so a raised ceiling — or a
|
|
312
|
+
corrected total — applies at the very next call.
|
|
313
|
+
"""
|
|
314
|
+
|
|
315
|
+
contract_limit: int | None = None
|
|
316
|
+
contract_used: int = 0
|
|
317
|
+
organisation_limit: int | None = None
|
|
318
|
+
organisation_used: int = 0
|
|
319
|
+
# The Organisation's ceiling covers its org-funded Contracts only: a user-funded
|
|
320
|
+
# Contract spends the user's own key and is none of the org's budget (map 12 B3).
|
|
321
|
+
org_funded: bool = False
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
class CeilingStore:
|
|
325
|
+
"""What each Contract may still spend, and what it has spent here since the push."""
|
|
326
|
+
|
|
327
|
+
def __init__(self) -> None:
|
|
328
|
+
self._ceilings: dict[UUID, Ceilings] = {}
|
|
329
|
+
self._spent: dict[UUID, int] = {}
|
|
330
|
+
|
|
331
|
+
def push(self, contract_id: UUID, ceilings: Ceilings) -> None:
|
|
332
|
+
self._ceilings[contract_id] = ceilings
|
|
333
|
+
self._spent[contract_id] = 0
|
|
334
|
+
|
|
335
|
+
def spend(self, contract_id: UUID | None, tokens: int) -> None:
|
|
336
|
+
if contract_id is None or tokens <= 0:
|
|
337
|
+
return
|
|
338
|
+
self._spent[contract_id] = self._spent.get(contract_id, 0) + tokens
|
|
339
|
+
|
|
340
|
+
def authorize(self, contract_id: UUID | None) -> None:
|
|
341
|
+
"""Refuse the call the month can no longer fund, before any provider is touched.
|
|
342
|
+
|
|
343
|
+
Checked against the month to date: a call may overshoot its ceiling by its own
|
|
344
|
+
spend and no more, the same bound the Work Record Budget carries between
|
|
345
|
+
Directives. A Contract with no pushed ceiling is unbounded here — the control
|
|
346
|
+
plane is the authority on limits, and inventing one would refuse work nobody
|
|
347
|
+
capped.
|
|
348
|
+
"""
|
|
349
|
+
|
|
350
|
+
ceilings = self._ceilings.get(contract_id) if contract_id is not None else None
|
|
351
|
+
if ceilings is None:
|
|
352
|
+
return
|
|
353
|
+
local = self._spent.get(contract_id, 0) if contract_id is not None else 0
|
|
354
|
+
if ceilings.contract_limit is not None:
|
|
355
|
+
used = ceilings.contract_used + local
|
|
356
|
+
if used >= ceilings.contract_limit:
|
|
357
|
+
raise CeilingExhaustedError(
|
|
358
|
+
ceiling="contract", limit=ceilings.contract_limit, used=used
|
|
359
|
+
)
|
|
360
|
+
if not ceilings.org_funded or ceilings.organisation_limit is None:
|
|
361
|
+
return
|
|
362
|
+
used = ceilings.organisation_used + local
|
|
363
|
+
if used >= ceilings.organisation_limit:
|
|
364
|
+
raise CeilingExhaustedError(
|
|
365
|
+
ceiling="organisation", limit=ceilings.organisation_limit, used=used
|
|
366
|
+
)
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
class UsageOutbox:
|
|
370
|
+
"""Usage Records waiting for the heartbeat that carries them out (ADR-0013 §11).
|
|
371
|
+
|
|
372
|
+
At-least-once by construction: a record stays here until an acknowledgement names
|
|
373
|
+
it, so a dropped ack costs a re-send. ``(directive_id, sequence)`` is what the ledger
|
|
374
|
+
de-duplicates on, so the re-send lands on the row it already wrote.
|
|
375
|
+
"""
|
|
376
|
+
|
|
377
|
+
def __init__(self, *, capacity: int = 10_000) -> None:
|
|
378
|
+
self._capacity = capacity
|
|
379
|
+
self._pending: dict[str, UsageRecord] = {}
|
|
380
|
+
self._sequences: dict[str, int] = {}
|
|
381
|
+
|
|
382
|
+
def next_sequence(self, directive_id: str) -> int:
|
|
383
|
+
# ponytail: one counter per `directive_id`, kept for the process's life and never
|
|
384
|
+
# pruned. A restart resets them, so a Directive still running resumes at 0 and the
|
|
385
|
+
# ledger reads the new rows as duplicates of the ones it already committed --
|
|
386
|
+
# the idempotency key becoming a data-loss key. The upgrade path is a
|
|
387
|
+
# restart-unique prefix (the Runner's registration id) on the key, which changes
|
|
388
|
+
# the wire shape both ends de-duplicate on and so belongs with the heartbeat loop.
|
|
389
|
+
sequence = self._sequences.get(directive_id, 0)
|
|
390
|
+
self._sequences[directive_id] = sequence + 1
|
|
391
|
+
return sequence
|
|
392
|
+
|
|
393
|
+
def record(self, record: UsageRecord) -> None:
|
|
394
|
+
if len(self._pending) >= self._capacity and record.key not in self._pending:
|
|
395
|
+
# Oldest first: a Runner that cannot reach the control plane for long enough
|
|
396
|
+
# to fill this has a bigger problem than the tail of its own ledger, and
|
|
397
|
+
# unbounded growth would take the process down with it.
|
|
398
|
+
self._pending.pop(next(iter(self._pending)))
|
|
399
|
+
self._pending[record.key] = record
|
|
400
|
+
|
|
401
|
+
def pending(self, limit: int = USAGE_BATCH_MAX) -> list[UsageRecord]:
|
|
402
|
+
return list(self._pending.values())[:limit]
|
|
403
|
+
|
|
404
|
+
def acknowledge(self, keys: list[str]) -> int:
|
|
405
|
+
return sum(1 for key in keys if self._pending.pop(key, None) is not None)
|
|
406
|
+
|
|
407
|
+
def __len__(self) -> int:
|
|
408
|
+
return len(self._pending)
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
@dataclass(frozen=True, slots=True)
|
|
412
|
+
class AttemptHandle:
|
|
413
|
+
"""What one Directive attempt is given: a URL under its own path, and its bearer."""
|
|
414
|
+
|
|
415
|
+
attempt_id: str
|
|
416
|
+
token: str
|
|
417
|
+
base_url: str
|
|
418
|
+
directive_id: str
|
|
419
|
+
contract_id: UUID | None
|
|
420
|
+
agent_id: UUID | None
|
|
421
|
+
work_record_id: UUID | None
|
|
422
|
+
# A per-attempt cap on top of the Contract's ceilings: the Learning reserve (PRD issue
|
|
423
|
+
# 55, map 12 B6). None is no cap -- every ordinary Directive.
|
|
424
|
+
reserve_max_tokens: int | None = None
|
|
425
|
+
|
|
426
|
+
def env(self, cli_kind: str) -> dict[str, str]:
|
|
427
|
+
return attempt_env(cli_kind, base_url=self.base_url, token=self.token)
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def attempt_env(cli_kind: str, *, base_url: str, token: str) -> dict[str, str]:
|
|
431
|
+
"""The names each harness reads its endpoint and credential from.
|
|
432
|
+
|
|
433
|
+
The "credential" here is the attempt's own bearer, which reaches only this Runner's
|
|
434
|
+
loopback proxy and dies with the attempt — so ADR-0011 §9 still holds: the subprocess
|
|
435
|
+
has no provider key, and anything it sends is metered and ceiling-checked on the way
|
|
436
|
+
out.
|
|
437
|
+
"""
|
|
438
|
+
|
|
439
|
+
if cli_kind.startswith("claude"):
|
|
440
|
+
# Claude Code appends `/v1/messages` to what it is given; the OpenAI-compatible
|
|
441
|
+
# CLIs are configured with the `/v1` already on. Same attempt, same bearer — only
|
|
442
|
+
# the half of the URL each harness expects to supply differs.
|
|
443
|
+
return {"ANTHROPIC_BASE_URL": base_url, "ANTHROPIC_AUTH_TOKEN": token}
|
|
444
|
+
return {"OPENAI_BASE_URL": f"{base_url}/v1", "OPENAI_API_KEY": token}
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
class LlmProxy:
|
|
448
|
+
"""One Runner process's proxy: loopback TCP, per-attempt bearer, metered per call."""
|
|
449
|
+
|
|
450
|
+
def __init__(
|
|
451
|
+
self,
|
|
452
|
+
*,
|
|
453
|
+
slots: SlotStore,
|
|
454
|
+
ceilings: CeilingStore | None = None,
|
|
455
|
+
outbox: UsageOutbox | None = None,
|
|
456
|
+
prices: dict[str, ModelPrice] | None = None,
|
|
457
|
+
evidence: Evidence | None = None,
|
|
458
|
+
client: httpx.AsyncClient | None = None,
|
|
459
|
+
host: str = "127.0.0.1",
|
|
460
|
+
port: int = 0,
|
|
461
|
+
) -> None:
|
|
462
|
+
if host not in _LOOPBACK_HOSTS:
|
|
463
|
+
# The proxy resolves the funder's key on every request. Its only legitimate
|
|
464
|
+
# callers are subprocesses of this same process, so a bind anyone else can
|
|
465
|
+
# reach is refused here rather than left to a network policy to catch.
|
|
466
|
+
raise ValueError(f"the LLM proxy binds loopback only, not {host!r}")
|
|
467
|
+
self.slots = slots
|
|
468
|
+
self.ceilings = ceilings or CeilingStore()
|
|
469
|
+
self.outbox = outbox or UsageOutbox()
|
|
470
|
+
self._prices = prices or {}
|
|
471
|
+
self._evidence = evidence
|
|
472
|
+
self._client = client
|
|
473
|
+
self._owns_client = client is None
|
|
474
|
+
self._host = host
|
|
475
|
+
self._port = port
|
|
476
|
+
self._server: asyncio.AbstractServer | None = None
|
|
477
|
+
self._attempts: dict[str, AttemptHandle] = {}
|
|
478
|
+
# Tokens each capped attempt has spent, keyed by attempt id; gone with the attempt.
|
|
479
|
+
self._reserve_spent: dict[str, int] = {}
|
|
480
|
+
|
|
481
|
+
@property
|
|
482
|
+
def base_url(self) -> str:
|
|
483
|
+
return f"http://{self._host}:{self._port}"
|
|
484
|
+
|
|
485
|
+
async def __aenter__(self) -> Self:
|
|
486
|
+
if self._client is None:
|
|
487
|
+
self._client = httpx.AsyncClient(timeout=httpx.Timeout(300.0, connect=10.0))
|
|
488
|
+
self._server = await asyncio.start_server(self._serve, host=self._host, port=self._port)
|
|
489
|
+
# Port 0 by default: the operator configures no port for a surface only this
|
|
490
|
+
# process's own children ever dial.
|
|
491
|
+
self._port = self._server.sockets[0].getsockname()[1]
|
|
492
|
+
return self
|
|
493
|
+
|
|
494
|
+
async def __aexit__(
|
|
495
|
+
self,
|
|
496
|
+
exc_type: type[BaseException] | None,
|
|
497
|
+
exc: BaseException | None,
|
|
498
|
+
traceback: TracebackType | None,
|
|
499
|
+
) -> None:
|
|
500
|
+
server, self._server = self._server, None
|
|
501
|
+
if server is not None:
|
|
502
|
+
server.close()
|
|
503
|
+
await server.wait_closed()
|
|
504
|
+
if self._owns_client and self._client is not None:
|
|
505
|
+
await self._client.aclose()
|
|
506
|
+
self._client = None
|
|
507
|
+
|
|
508
|
+
@contextlib.asynccontextmanager
|
|
509
|
+
async def attempt(
|
|
510
|
+
self,
|
|
511
|
+
*,
|
|
512
|
+
directive_id: str,
|
|
513
|
+
contract_id: UUID | None,
|
|
514
|
+
agent_id: UUID | None = None,
|
|
515
|
+
work_record_id: UUID | None = None,
|
|
516
|
+
token: str | None = None,
|
|
517
|
+
reserve_max_tokens: int | None = None,
|
|
518
|
+
) -> AsyncIterator[AttemptHandle]:
|
|
519
|
+
"""Admit one Directive attempt for as long as it runs, and no longer.
|
|
520
|
+
|
|
521
|
+
``token`` is the attempt's callback bearer (issue 45) when there is one: one
|
|
522
|
+
token per attempt serves both surfaces, so the subprocess holds exactly one
|
|
523
|
+
secret and it expires with the attempt either way.
|
|
524
|
+
"""
|
|
525
|
+
|
|
526
|
+
attempt_id = secrets.token_urlsafe(9)
|
|
527
|
+
handle = AttemptHandle(
|
|
528
|
+
attempt_id=attempt_id,
|
|
529
|
+
token=token or secrets.token_urlsafe(_TOKEN_BYTES),
|
|
530
|
+
base_url=f"{self.base_url}{ATTEMPT_PREFIX}{attempt_id}",
|
|
531
|
+
directive_id=directive_id,
|
|
532
|
+
contract_id=contract_id,
|
|
533
|
+
agent_id=agent_id,
|
|
534
|
+
work_record_id=work_record_id,
|
|
535
|
+
reserve_max_tokens=reserve_max_tokens,
|
|
536
|
+
)
|
|
537
|
+
self._attempts[attempt_id] = handle
|
|
538
|
+
try:
|
|
539
|
+
yield handle
|
|
540
|
+
finally:
|
|
541
|
+
self._attempts.pop(attempt_id, None)
|
|
542
|
+
self._reserve_spent.pop(attempt_id, None)
|
|
543
|
+
|
|
544
|
+
# ----------------------------------------------------------------- serving
|
|
545
|
+
|
|
546
|
+
async def _serve(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
|
|
547
|
+
try:
|
|
548
|
+
await self._route(reader, writer)
|
|
549
|
+
except Exception as error: # noqa: BLE001 - one bad call never takes the proxy down
|
|
550
|
+
await write_json(writer, 500, {"detail": error.__class__.__name__})
|
|
551
|
+
finally:
|
|
552
|
+
writer.close()
|
|
553
|
+
with contextlib.suppress(ConnectionResetError, BrokenPipeError):
|
|
554
|
+
await writer.wait_closed()
|
|
555
|
+
|
|
556
|
+
async def _route(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
|
|
557
|
+
request = await read_request(reader, max_body_bytes=MAX_REQUEST_BODY_BYTES)
|
|
558
|
+
if not isinstance(request, HttpRequest):
|
|
559
|
+
await write_json(writer, *request)
|
|
560
|
+
return
|
|
561
|
+
|
|
562
|
+
attempt, path = self._authorise(request)
|
|
563
|
+
if attempt is None:
|
|
564
|
+
# One answer for an unknown attempt and for a wrong bearer: which of the two
|
|
565
|
+
# it was is not the caller's business.
|
|
566
|
+
await write_json(writer, 401, {"detail": "a valid attempt bearer is required"})
|
|
567
|
+
return
|
|
568
|
+
|
|
569
|
+
# The route is checked before the slot is resolved: an unknown path is not a
|
|
570
|
+
# funding failure, and answering one with a 503 and an Evidence Event would let a
|
|
571
|
+
# mistyped URL write Evidence.
|
|
572
|
+
models_list = request.method == "GET" and path == "/models"
|
|
573
|
+
if not models_list and (request.method != "POST" or path not in COMPLETION_PATHS):
|
|
574
|
+
await write_json(writer, 404, {"detail": "no such proxy route"})
|
|
575
|
+
return
|
|
576
|
+
|
|
577
|
+
slot = self.slots.resolve(attempt.contract_id)
|
|
578
|
+
if slot is None:
|
|
579
|
+
await self._refused(attempt, reason="slot_missing")
|
|
580
|
+
await write_json(
|
|
581
|
+
writer, 503, {"error": {"message": "no LLM slot", "type": "slot_unavailable"}}
|
|
582
|
+
)
|
|
583
|
+
return
|
|
584
|
+
|
|
585
|
+
if models_list:
|
|
586
|
+
await self._relay_models(writer, slot=slot)
|
|
587
|
+
return
|
|
588
|
+
|
|
589
|
+
try:
|
|
590
|
+
self.ceilings.authorize(attempt.contract_id)
|
|
591
|
+
except CeilingExhaustedError as error:
|
|
592
|
+
await self._refused(attempt, reason="ceiling_exhausted", ceiling=error.ceiling)
|
|
593
|
+
await write_json(writer, 402, ceiling_exhausted_detail(error))
|
|
594
|
+
return
|
|
595
|
+
|
|
596
|
+
spent = self._reserve_spent.get(attempt.attempt_id, 0)
|
|
597
|
+
if attempt.reserve_max_tokens is not None and spent >= attempt.reserve_max_tokens:
|
|
598
|
+
await self._refused(attempt, reason="reserve_exhausted", used=spent)
|
|
599
|
+
await write_json(
|
|
600
|
+
writer,
|
|
601
|
+
402,
|
|
602
|
+
{
|
|
603
|
+
"error": {
|
|
604
|
+
"message": (
|
|
605
|
+
f"reserve exhausted: {spent} of {attempt.reserve_max_tokens} "
|
|
606
|
+
"tokens spent"
|
|
607
|
+
),
|
|
608
|
+
"type": "reserve_exhausted",
|
|
609
|
+
}
|
|
610
|
+
},
|
|
611
|
+
)
|
|
612
|
+
return
|
|
613
|
+
|
|
614
|
+
await self._relay_completion(writer, request=request, attempt=attempt, slot=slot, path=path)
|
|
615
|
+
|
|
616
|
+
def _authorise(self, request: HttpRequest) -> tuple[AttemptHandle | None, str]:
|
|
617
|
+
"""Which attempt this request is, by path *and* bearer.
|
|
618
|
+
|
|
619
|
+
The path segment is what makes another attempt's bearer a refusal rather than a
|
|
620
|
+
mis-attribution: a token that is valid for some other live attempt does not open
|
|
621
|
+
this one.
|
|
622
|
+
"""
|
|
623
|
+
|
|
624
|
+
if not request.target.startswith(ATTEMPT_PREFIX):
|
|
625
|
+
return None, ""
|
|
626
|
+
attempt_id, _, rest = request.target[len(ATTEMPT_PREFIX) :].partition("/")
|
|
627
|
+
attempt = self._attempts.get(attempt_id)
|
|
628
|
+
if attempt is None or not bearer_matches(request.authorization, attempt.token):
|
|
629
|
+
return None, ""
|
|
630
|
+
# The `/v1` is the harness's, not ours: Claude Code appends it and the
|
|
631
|
+
# OpenAI-compatible CLIs carry it in the base URL they were configured with.
|
|
632
|
+
path = "/" + rest.removeprefix("v1/").lstrip("/")
|
|
633
|
+
return attempt, path.partition("?")[0]
|
|
634
|
+
|
|
635
|
+
async def _relay_models(self, writer: asyncio.StreamWriter, *, slot: CredentialSlot) -> None:
|
|
636
|
+
assert self._client is not None
|
|
637
|
+
try:
|
|
638
|
+
upstream = await self._client.get(slot.url("models"), headers=slot.headers())
|
|
639
|
+
except httpx.HTTPError as error:
|
|
640
|
+
await write_json(
|
|
641
|
+
writer,
|
|
642
|
+
502,
|
|
643
|
+
{"error": {"message": "provider unreachable", "type": error.__class__.__name__}},
|
|
644
|
+
)
|
|
645
|
+
return
|
|
646
|
+
writer.write(
|
|
647
|
+
response_head(
|
|
648
|
+
upstream.status_code,
|
|
649
|
+
content_type="application/json",
|
|
650
|
+
content_length=len(upstream.content),
|
|
651
|
+
)
|
|
652
|
+
)
|
|
653
|
+
writer.write(upstream.content)
|
|
654
|
+
with contextlib.suppress(ConnectionResetError, BrokenPipeError):
|
|
655
|
+
await writer.drain()
|
|
656
|
+
|
|
657
|
+
async def _relay_completion(
|
|
658
|
+
self,
|
|
659
|
+
writer: asyncio.StreamWriter,
|
|
660
|
+
*,
|
|
661
|
+
request: HttpRequest,
|
|
662
|
+
attempt: AttemptHandle,
|
|
663
|
+
slot: CredentialSlot,
|
|
664
|
+
path: str,
|
|
665
|
+
) -> None:
|
|
666
|
+
assert self._client is not None
|
|
667
|
+
payload = _json_object(request.body)
|
|
668
|
+
model_name = str(payload.get("model") or "unknown")
|
|
669
|
+
streaming = payload.get("stream") is True
|
|
670
|
+
url = slot.url(path)
|
|
671
|
+
headers = _upstream_headers(request, slot)
|
|
672
|
+
body = request.body
|
|
673
|
+
if streaming and path == "/chat/completions":
|
|
674
|
+
# An OpenAI-compatible endpoint omits the usage block from a stream unless it
|
|
675
|
+
# is asked for, and a streamed call metered at zero is a ledger row missing
|
|
676
|
+
# its dominant term and a per-call ceiling that cannot be enforced. Anthropic
|
|
677
|
+
# sends the counts either way, split across two frames (`_sse_usage`).
|
|
678
|
+
body = _with_usage_in_stream(payload)
|
|
679
|
+
started = asyncio.get_running_loop().time()
|
|
680
|
+
|
|
681
|
+
if not streaming:
|
|
682
|
+
try:
|
|
683
|
+
upstream = await self._client.post(url, content=body, headers=headers)
|
|
684
|
+
except httpx.HTTPError as error:
|
|
685
|
+
self._meter(
|
|
686
|
+
attempt,
|
|
687
|
+
slot=slot,
|
|
688
|
+
model_name=model_name,
|
|
689
|
+
usage={},
|
|
690
|
+
started=started,
|
|
691
|
+
error_class=error.__class__.__name__,
|
|
692
|
+
)
|
|
693
|
+
await write_json(
|
|
694
|
+
writer,
|
|
695
|
+
502,
|
|
696
|
+
{"error": {"message": "provider unreachable", "type": "provider_error"}},
|
|
697
|
+
)
|
|
698
|
+
return
|
|
699
|
+
self._meter(
|
|
700
|
+
attempt,
|
|
701
|
+
slot=slot,
|
|
702
|
+
model_name=model_name,
|
|
703
|
+
usage=_usage_block(_json_object(upstream.content)),
|
|
704
|
+
started=started,
|
|
705
|
+
error_class=None if upstream.status_code < 400 else f"http_{upstream.status_code}",
|
|
706
|
+
)
|
|
707
|
+
writer.write(
|
|
708
|
+
response_head(
|
|
709
|
+
upstream.status_code,
|
|
710
|
+
content_type="application/json",
|
|
711
|
+
content_length=len(upstream.content),
|
|
712
|
+
)
|
|
713
|
+
)
|
|
714
|
+
writer.write(upstream.content)
|
|
715
|
+
with contextlib.suppress(ConnectionResetError, BrokenPipeError):
|
|
716
|
+
await writer.drain()
|
|
717
|
+
return
|
|
718
|
+
|
|
719
|
+
usage: dict[str, object] = {}
|
|
720
|
+
error_class: str | None = None
|
|
721
|
+
started_relaying = False
|
|
722
|
+
tail = b""
|
|
723
|
+
try:
|
|
724
|
+
async with self._client.stream("POST", url, content=body, headers=headers) as upstream:
|
|
725
|
+
writer.write(response_head(upstream.status_code, content_type="text/event-stream"))
|
|
726
|
+
started_relaying = True
|
|
727
|
+
# The same code the non-streaming branch records: a streamed call the
|
|
728
|
+
# provider refused meters no tokens, and a row with none and no error
|
|
729
|
+
# class reads as a successful free call.
|
|
730
|
+
if upstream.status_code >= 400:
|
|
731
|
+
error_class = f"http_{upstream.status_code}"
|
|
732
|
+
async for chunk in upstream.aiter_bytes():
|
|
733
|
+
writer.write(chunk)
|
|
734
|
+
# Lines are re-assembled across chunk boundaries: a `data:` frame the
|
|
735
|
+
# transport split in two carries its counts in neither half.
|
|
736
|
+
*lines, tail = (tail + chunk).split(b"\n")
|
|
737
|
+
usage.update(_sse_usage(lines))
|
|
738
|
+
with contextlib.suppress(ConnectionResetError, BrokenPipeError):
|
|
739
|
+
await writer.drain()
|
|
740
|
+
except httpx.HTTPError as error:
|
|
741
|
+
error_class = error.__class__.__name__
|
|
742
|
+
usage.update(_sse_usage([tail]))
|
|
743
|
+
self._meter(
|
|
744
|
+
attempt,
|
|
745
|
+
slot=slot,
|
|
746
|
+
model_name=model_name,
|
|
747
|
+
usage=usage,
|
|
748
|
+
started=started,
|
|
749
|
+
error_class=error_class,
|
|
750
|
+
)
|
|
751
|
+
if error_class is not None and not started_relaying:
|
|
752
|
+
# Nothing has been written yet, so the caller can still be told why rather
|
|
753
|
+
# than being handed a connection that simply closes.
|
|
754
|
+
await write_json(
|
|
755
|
+
writer,
|
|
756
|
+
502,
|
|
757
|
+
{"error": {"message": "provider unreachable", "type": "provider_error"}},
|
|
758
|
+
)
|
|
759
|
+
|
|
760
|
+
# ----------------------------------------------------------------- metering
|
|
761
|
+
|
|
762
|
+
def _meter(
|
|
763
|
+
self,
|
|
764
|
+
attempt: AttemptHandle,
|
|
765
|
+
*,
|
|
766
|
+
slot: CredentialSlot,
|
|
767
|
+
model_name: str,
|
|
768
|
+
usage: Mapping[str, object],
|
|
769
|
+
started: float,
|
|
770
|
+
error_class: str | None,
|
|
771
|
+
) -> None:
|
|
772
|
+
"""One Usage Record per call, and the tokens the ceiling now counts.
|
|
773
|
+
|
|
774
|
+
Reads the upstream ``usage`` block and nothing else — the completion itself is
|
|
775
|
+
forwarded and forgotten, which is the rule the structural test pins.
|
|
776
|
+
"""
|
|
777
|
+
|
|
778
|
+
prompt_tokens = _prompt_tokens(usage)
|
|
779
|
+
completion_tokens = _completion_tokens(usage)
|
|
780
|
+
price = resolve_price(self._prices, provider_name=slot.provider_name, model_name=model_name)
|
|
781
|
+
record = UsageRecord(
|
|
782
|
+
directive_id=attempt.directive_id,
|
|
783
|
+
sequence=self.outbox.next_sequence(attempt.directive_id),
|
|
784
|
+
provider_name=slot.provider_name,
|
|
785
|
+
model_name=model_name,
|
|
786
|
+
prompt_tokens=prompt_tokens,
|
|
787
|
+
completion_tokens=completion_tokens,
|
|
788
|
+
cached_tokens=_cached_tokens(usage),
|
|
789
|
+
applied_prompt_price_usd=price.prompt_usd_per_token,
|
|
790
|
+
applied_completion_price_usd=price.completion_usd_per_token,
|
|
791
|
+
cost_usd=price.cost_usd(
|
|
792
|
+
prompt_tokens=prompt_tokens or 0, completion_tokens=completion_tokens or 0
|
|
793
|
+
),
|
|
794
|
+
latency_ms=max(0, int((asyncio.get_running_loop().time() - started) * 1000)),
|
|
795
|
+
error_class=error_class,
|
|
796
|
+
agent_id=attempt.agent_id,
|
|
797
|
+
contract_id=attempt.contract_id,
|
|
798
|
+
work_record_id=attempt.work_record_id,
|
|
799
|
+
)
|
|
800
|
+
self.outbox.record(record)
|
|
801
|
+
tokens = (prompt_tokens or 0) + (completion_tokens or 0)
|
|
802
|
+
self.ceilings.spend(attempt.contract_id, tokens)
|
|
803
|
+
if attempt.reserve_max_tokens is not None:
|
|
804
|
+
self._reserve_spent[attempt.attempt_id] = (
|
|
805
|
+
self._reserve_spent.get(attempt.attempt_id, 0) + tokens
|
|
806
|
+
)
|
|
807
|
+
|
|
808
|
+
async def _refused(self, attempt: AttemptHandle, *, reason: str, **extra: object) -> None:
|
|
809
|
+
"""A refused call is an Evidence Event with the reason — ids only (issue 43)."""
|
|
810
|
+
|
|
811
|
+
if self._evidence is None:
|
|
812
|
+
return
|
|
813
|
+
await self._evidence(
|
|
814
|
+
CALL_REFUSED_SOURCE,
|
|
815
|
+
{
|
|
816
|
+
"reason": reason,
|
|
817
|
+
"directive_id": attempt.directive_id,
|
|
818
|
+
"contract_id": str(attempt.contract_id) if attempt.contract_id else None,
|
|
819
|
+
"agent_id": str(attempt.agent_id) if attempt.agent_id else None,
|
|
820
|
+
**extra,
|
|
821
|
+
},
|
|
822
|
+
)
|
|
823
|
+
|
|
824
|
+
|
|
825
|
+
def _json_object(raw: bytes) -> dict[str, Any]:
|
|
826
|
+
try:
|
|
827
|
+
parsed = json.loads(raw or b"{}")
|
|
828
|
+
except ValueError:
|
|
829
|
+
return {}
|
|
830
|
+
return parsed if isinstance(parsed, dict) else {}
|
|
831
|
+
|
|
832
|
+
|
|
833
|
+
def _upstream_headers(request: HttpRequest, slot: CredentialSlot) -> dict[str, str]:
|
|
834
|
+
"""The harness's own headers, with the slot's credential in place of its bearer.
|
|
835
|
+
|
|
836
|
+
Forwarded rather than rebuilt: ``anthropic-beta`` is how Claude Code asks for the
|
|
837
|
+
features it is built against, and a proxy that drops it answers a different call than
|
|
838
|
+
the one the Agent made.
|
|
839
|
+
"""
|
|
840
|
+
|
|
841
|
+
forwarded = {
|
|
842
|
+
name: value for name, value in request.headers.items() if name not in _HEADERS_NOT_FORWARDED
|
|
843
|
+
}
|
|
844
|
+
# The harness's headers win over the slot's: the credential cannot be among them (both
|
|
845
|
+
# auth headers are stripped above), and a harness that asks for a newer
|
|
846
|
+
# `anthropic-version` than the slot's floor must get the wire it asked for.
|
|
847
|
+
return {"content-type": "application/json", **slot.headers(), **forwarded}
|
|
848
|
+
|
|
849
|
+
|
|
850
|
+
def _with_usage_in_stream(payload: dict[str, Any]) -> bytes:
|
|
851
|
+
"""``stream_options.include_usage``, without discarding one the caller already set."""
|
|
852
|
+
|
|
853
|
+
options = payload.get("stream_options")
|
|
854
|
+
payload["stream_options"] = {
|
|
855
|
+
**(options if isinstance(options, Mapping) else {}),
|
|
856
|
+
"include_usage": True,
|
|
857
|
+
}
|
|
858
|
+
return json.dumps(payload).encode()
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
def _usage_block(body: Mapping[str, object]) -> Mapping[str, object]:
|
|
862
|
+
usage = body.get("usage")
|
|
863
|
+
if isinstance(usage, Mapping):
|
|
864
|
+
return usage
|
|
865
|
+
# Anthropic's `message_start` frame nests the input counts one level down, under the
|
|
866
|
+
# message it is starting. `message` is followed only to reach that `usage` block --
|
|
867
|
+
# nothing else in the frame is read (ADR-0010 §4).
|
|
868
|
+
message = body.get("message")
|
|
869
|
+
if isinstance(message, Mapping):
|
|
870
|
+
nested = message.get("usage")
|
|
871
|
+
if isinstance(nested, Mapping):
|
|
872
|
+
return nested
|
|
873
|
+
return {}
|
|
874
|
+
|
|
875
|
+
|
|
876
|
+
def _sse_usage(lines: Iterable[bytes]) -> dict[str, object]:
|
|
877
|
+
"""The ``usage`` counts these SSE lines carried, merged rather than replaced.
|
|
878
|
+
|
|
879
|
+
One call's counts can arrive on more than one frame: Anthropic puts the input tokens
|
|
880
|
+
on ``message_start`` and the output tokens on ``message_delta``. Keeping only the last
|
|
881
|
+
block seen would meter a streamed ``/v1/messages`` call with no prompt tokens at all
|
|
882
|
+
-- the dominant term of both the ledger row and the ceiling -- and Claude Code streams
|
|
883
|
+
by default, so that is the normal path for one of the two harnesses.
|
|
884
|
+
"""
|
|
885
|
+
|
|
886
|
+
found: dict[str, object] = {}
|
|
887
|
+
for raw in lines:
|
|
888
|
+
line = raw.strip()
|
|
889
|
+
if not line.startswith(b"data:") or line.startswith(_SSE_DONE):
|
|
890
|
+
continue
|
|
891
|
+
found.update(_usage_block(_json_object(line[len(b"data:") :].strip())))
|
|
892
|
+
return found
|
|
893
|
+
|
|
894
|
+
|
|
895
|
+
def _usage_int(usage: Mapping[str, object], key: str) -> int | None:
|
|
896
|
+
value = usage.get(key)
|
|
897
|
+
return value if isinstance(value, int) else None
|
|
898
|
+
|
|
899
|
+
|
|
900
|
+
def _first_int(usage: Mapping[str, object], *keys: str) -> int | None:
|
|
901
|
+
"""The first of ``keys`` this usage block carries.
|
|
902
|
+
|
|
903
|
+
Two wires, one Usage Record: OpenAI-compatible endpoints say ``prompt_tokens`` and
|
|
904
|
+
Anthropic says ``input_tokens``. The ledger and the ceilings count tokens, so the
|
|
905
|
+
difference is a spelling and is resolved here rather than in the row.
|
|
906
|
+
"""
|
|
907
|
+
|
|
908
|
+
for key in keys:
|
|
909
|
+
value = _usage_int(usage, key)
|
|
910
|
+
if value is not None:
|
|
911
|
+
return value
|
|
912
|
+
return None
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def _prompt_tokens(usage: Mapping[str, object]) -> int | None:
|
|
916
|
+
tokens = _first_int(usage, "prompt_tokens", "input_tokens")
|
|
917
|
+
if tokens is None:
|
|
918
|
+
return None
|
|
919
|
+
# Cache *writes* are new input the model had to encode, so they are prompt tokens —
|
|
920
|
+
# the same split `claude_code_harness_usage` already prices (PRD issue 31).
|
|
921
|
+
return tokens + (_usage_int(usage, "cache_creation_input_tokens") or 0)
|
|
922
|
+
|
|
923
|
+
|
|
924
|
+
def _completion_tokens(usage: Mapping[str, object]) -> int | None:
|
|
925
|
+
return _first_int(usage, "completion_tokens", "output_tokens")
|
|
926
|
+
|
|
927
|
+
|
|
928
|
+
def _cached_tokens(usage: Mapping[str, object]) -> int | None:
|
|
929
|
+
"""Cached input, wherever the provider puts it. Counted at full weight (map 12 B1)."""
|
|
930
|
+
|
|
931
|
+
direct = _first_int(usage, "cached_tokens", "cache_read_input_tokens")
|
|
932
|
+
if direct is not None:
|
|
933
|
+
return direct
|
|
934
|
+
details = usage.get("prompt_tokens_details")
|
|
935
|
+
if isinstance(details, Mapping):
|
|
936
|
+
return _usage_int(details, "cached_tokens")
|
|
937
|
+
return None
|