handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,391 @@
1
+ """The effect gate. Spec: `docs/0011` §3, `docs/0012` §3.2.
2
+
3
+ The correctness core, and deliberately small. One rule dominates:
4
+
5
+ guard() MUST NEVER RAISE.
6
+
7
+ Any internal failure becomes BLOCK — fail closed (`docs/0008` §6.5). A gate
8
+ that crashes open is worse than no gate, because it creates false confidence.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import json
13
+ import logging
14
+
15
+ from .classify import Classifier
16
+ from .reconcile.base import (
17
+ DID_NOT_LAND, INCONCLUSIVE, LANDED, SAFE_TO_RETRY, ProbeRegistry,
18
+ )
19
+ from .ledger.models import (
20
+ EffectClass,
21
+ EffectState,
22
+ GateDecision,
23
+ ToolCall,
24
+ Verdict,
25
+ )
26
+ from .ledger.store import LedgerStore, StaleFence
27
+
28
+ log = logging.getLogger("agentctl.gate")
29
+
30
+
31
+ class EffectGate:
32
+ """Decides whether a tool call may execute.
33
+
34
+ Seam-agnostic by design (`docs/0012` §3.2.1): it returns a decision, and
35
+ the binding acts on it. Seam B can honour BLOCK/ESCALATE; only Seam C can
36
+ honour SUBSTITUTE.
37
+ """
38
+
39
+ def __init__(
40
+ self,
41
+ store: LedgerStore,
42
+ classifier: Classifier | None = None,
43
+ probes: ProbeRegistry | None = None,
44
+ fence: int = 0,
45
+ ):
46
+ self.store = store
47
+ self.classifier = classifier or Classifier()
48
+ self.probes = probes if probes is not None else ProbeRegistry()
49
+ self.fence = fence
50
+
51
+ # ── the decision ───────────────────────────────────────────────────
52
+ def guard(self, call: ToolCall) -> GateDecision:
53
+ try:
54
+ return self._guard(call)
55
+ except StaleFence as exc:
56
+ # Not a bug in the gate: another process took this conversation
57
+ # over (`docs/0046`). Said as such, rather than as "gate error,
58
+ # failing closed: StaleFence(...)" -- accurate, and meaningless to
59
+ # the person reading it (`docs/0048`).
60
+ log.warning("superseded: %s", exc)
61
+ return GateDecision(
62
+ Verdict.BLOCK,
63
+ reason="another process has taken over this conversation, so "
64
+ "this run may not start new actions; let the other "
65
+ "run finish, or stop this one")
66
+ except Exception as exc: # noqa: BLE001
67
+ # Fail closed. Never let a gate bug become an unguarded effect.
68
+ log.exception("gate failure for %s", call.tool_call_id)
69
+ self._try_block(call.tool_call_id, f"gate error: {exc!r}")
70
+ return GateDecision(
71
+ Verdict.BLOCK, reason=f"gate error, failing closed: {exc!r}"
72
+ )
73
+
74
+ def _guard(self, call: ToolCall) -> GateDecision:
75
+ cls = self.classifier.classify(call)
76
+ rec = self.store.lookup(call.tool_call_id)
77
+
78
+ if rec is None and not cls.replay_safe:
79
+ # The id is model-minted, so a different model answering the same
80
+ # question produces a different one for the identical call
81
+ # (`docs/0023` §4). Before calling this a first sighting, ask
82
+ # whether this exact effect is already on record under another id.
83
+ twin = self.store.find_by_intent(call.conversation_id,
84
+ call.intent_hash())
85
+ if twin is not None:
86
+ log.info("matched %s to prior effect %s by intent hash",
87
+ call.tool_call_id, twin.tool_call_id)
88
+ return self._decide_on(call, twin, cls, aliased=True)
89
+
90
+ # First sighting — the common case.
91
+ if rec is None:
92
+ self.store.write_intent(call, cls, self.fence, self._capture(call, cls))
93
+ return GateDecision(Verdict.EXECUTE, cls)
94
+
95
+ # Same id, different arguments. Substituting here would return the
96
+ # wrong observation, so refuse outright.
97
+ if rec.intent_hash != call.intent_hash():
98
+ return GateDecision(
99
+ Verdict.BLOCK, cls,
100
+ reason="tool_call_id reused with different arguments",
101
+ )
102
+ return self._decide_on(call, rec, cls)
103
+
104
+ def _decide_on(self, call: ToolCall, rec, cls: EffectClass,
105
+ aliased: bool = False) -> GateDecision:
106
+ """Decide against a record, which may be under a different id."""
107
+ note = (f" (matched to {rec.tool_call_id} by intent hash)"
108
+ if aliased else "")
109
+
110
+ # OBSERVED reaches here only by its own id: the SDK re-driving the very
111
+ # action whose result it persisted. `find_by_intent` never aliases to
112
+ # one, because a re-minted twin of an observed call is the model's
113
+ # decision to repeat (`docs/0045`).
114
+ if rec.state in (EffectState.COMMITTED, EffectState.OBSERVED):
115
+ return GateDecision(Verdict.SUBSTITUTE, cls,
116
+ observation=rec.observation,
117
+ reason=f"already recorded{note}" if note else None,
118
+ record_id=rec.tool_call_id)
119
+
120
+ if rec.state is EffectState.FAILED:
121
+ # Re-capture: the world may have moved since the failed attempt.
122
+ self.store.write_intent(call, cls, self.fence, self._capture(call, cls))
123
+ return GateDecision(Verdict.EXECUTE, cls)
124
+
125
+ if rec.state is EffectState.BLOCKED:
126
+ return GateDecision(
127
+ Verdict.BLOCK, cls,
128
+ reason=(rec.error or "previously blocked; awaiting human decision")
129
+ + note,
130
+ )
131
+
132
+ # rec.state is INTENT — the irreducible ambiguity (docs/0008 §6.5).
133
+ return self._resolve_ambiguous(call, rec, cls)
134
+
135
+ def _resolve_ambiguous(self, call, rec, cls: EffectClass) -> GateDecision:
136
+ """We cannot tell whether the effect landed. Decide by class.
137
+
138
+ Every ledger write here targets `rec.tool_call_id`, never
139
+ `call.tool_call_id`. Under intent-hash aliasing those differ, and
140
+ writing to the caller's id raises `IllegalTransition` -- which the
141
+ fail-closed wrapper then turns into a BLOCK, producing a correct
142
+ outcome by an incorrect route and stranding the record (`docs/0024`).
143
+ """
144
+ rid = rec.tool_call_id
145
+
146
+ if cls.replay_safe:
147
+ # Repeating is harmless, so the window does not matter.
148
+ return GateDecision(Verdict.EXECUTE, cls)
149
+
150
+ if cls is EffectClass.DESTRUCTIVE:
151
+ self.store.block(rid, "destructive effect, outcome unknown")
152
+ return GateDecision(
153
+ Verdict.ESCALATE, cls,
154
+ reason="destructive effect with unknown outcome; a human must decide",
155
+ )
156
+
157
+ # NON_IDEMPOTENT_WRITE / EXTERNAL: ask the world if it can answer.
158
+ probe = self.probes.for_call(call, self.classifier.probe_for(call))
159
+ verdict = INCONCLUSIVE
160
+ if probe is not None:
161
+ try:
162
+ verdict = probe.probe(call, rec)
163
+ except Exception: # noqa: BLE001
164
+ log.exception("probe %s raised", getattr(probe, "name", "?"))
165
+ verdict = INCONCLUSIVE
166
+
167
+ if verdict == LANDED and not self._sole_writer(rec):
168
+ # The probe reasoned from world state, but this call was not
169
+ # the only thing writing to that world. The change it saw may
170
+ # be a sibling's (`docs/0038` §3). Downgrade rather than
171
+ # attribute: a dropped effect recorded as COMMITTED is the one
172
+ # outcome this system exists to prevent.
173
+ log.warning("%s probe said LANDED, but a sibling effect "
174
+ "committed inside the window; not attributing", probe.name)
175
+ verdict = INCONCLUSIVE
176
+
177
+ if verdict == LANDED:
178
+ self.store.reconcile(rid, verdict, landed=True)
179
+ return GateDecision(
180
+ Verdict.SUBSTITUTE, cls, observation=rec.observation,
181
+ reason=f"{probe.name} probe: the effect already landed",
182
+ record_id=rid,
183
+ )
184
+ if verdict == DID_NOT_LAND:
185
+ self.store.reconcile(rid, verdict, landed=False)
186
+ self.store.write_intent(call, cls, self.fence,
187
+ self._capture(call, cls))
188
+ return GateDecision(
189
+ Verdict.EXECUTE, cls,
190
+ reason=f"{probe.name} probe: the effect did not land",
191
+ )
192
+ if verdict == SAFE_TO_RETRY:
193
+ # We cannot tell whether it landed, and we do not need to: the
194
+ # call carries an idempotency key, so the remote collapses a
195
+ # retry into the original effect (`docs/0020`).
196
+ self.store.reconcile(rid, verdict, landed=False)
197
+ self.store.write_intent(call, cls, self.fence,
198
+ self._capture(call, cls))
199
+ return GateDecision(
200
+ Verdict.EXECUTE, cls,
201
+ reason=(f"{probe.name} probe: outcome unknown, but the call "
202
+ f"is idempotency-keyed so a retry is safe"),
203
+ )
204
+
205
+ # No probe, or inconclusive. Fail closed.
206
+ detail = ("no probe available" if probe is None
207
+ else f"{probe.name} probe inconclusive")
208
+ reason = (f"cannot determine whether {call.tool_name} already "
209
+ f"executed ({detail})")
210
+ self.store.block(rid, reason)
211
+ return GateDecision(Verdict.BLOCK, cls, reason=reason)
212
+
213
+ def _sole_writer(self, rec) -> bool:
214
+ """Was this call the only thing writing the world its probe just read?
215
+
216
+ A probe that reasons from world state — HEAD moved, so my commit
217
+ landed — is sound only under that assumption, and it is the probe's
218
+ own stated one. It holds across conversations. It does not hold inside
219
+ a batch, where a sibling can execute after this call was fingerprinted
220
+ (`docs/0038` §3).
221
+
222
+ Only effects that can change the world count; a read cannot. And when
223
+ the question cannot be answered, the answer is no: a world-state
224
+ verdict we cannot justify is exactly what must not be trusted.
225
+ """
226
+ try:
227
+ siblings = self.store.committed_since(
228
+ rec.conversation_id, rec.started_at, rec.tool_call_id)
229
+ except Exception: # noqa: BLE001
230
+ log.exception("could not check for concurrent effects on %s",
231
+ rec.tool_call_id)
232
+ return False
233
+ return not any(not s.effect_class.replay_safe for s in siblings)
234
+
235
+ def recapture(self, call: ToolCall) -> None:
236
+ """Re-fingerprint the world for `call`, immediately before it runs.
237
+
238
+ Called from Seam C, the only seam that sees a call at its own execution
239
+ moment. Seam B decided for the whole batch; by the time this call is
240
+ actually about to execute, its siblings may already have changed the
241
+ world its probe will later reason about (`docs/0038` §3).
242
+
243
+ Never raises and never decides. A failure here costs a stale
244
+ fingerprint, which the probe reads as INCONCLUSIVE and the gate fails
245
+ closed on — the same direction as every other unknown in this system.
246
+ """
247
+ try:
248
+ cls = self.classifier.classify(call)
249
+ if cls.replay_safe:
250
+ return
251
+ self.store.refresh_pre_state(call.tool_call_id,
252
+ self._capture(call, cls))
253
+ except Exception: # noqa: BLE001
254
+ log.exception("recapture failed for %s", call.tool_call_id)
255
+
256
+ def _capture(self, call: ToolCall, cls: EffectClass) -> str | None:
257
+ """Fingerprint the world before acting, so a probe can compare later.
258
+
259
+ Only for classes that can reach the ambiguous branch -- there is no
260
+ point paying for it on a read.
261
+ """
262
+ if cls.replay_safe:
263
+ return None
264
+ probe = self.probes.for_call(call, self.classifier.probe_for(call))
265
+ if probe is None:
266
+ return None
267
+ try:
268
+ state = probe.capture(call)
269
+ except Exception: # noqa: BLE001
270
+ log.exception("probe %s capture raised", getattr(probe, "name", "?"))
271
+ return None
272
+ return json.dumps(state) if state else None
273
+
274
+ # ── post-execution bookkeeping ─────────────────────────────────────
275
+ def record_success(self, call: ToolCall, observation: bytes | None = None) -> None:
276
+ try:
277
+ self.store.commit(call.tool_call_id, observation)
278
+ except Exception: # noqa: BLE001
279
+ log.exception("could not commit %s", call.tool_call_id)
280
+
281
+ def record_failure(self, call: ToolCall, error: str) -> None:
282
+ try:
283
+ self.store.fail(call.tool_call_id, error)
284
+ except Exception: # noqa: BLE001
285
+ log.exception("could not record failure for %s", call.tool_call_id)
286
+
287
+ def record_success_by_id(self, tool_call_id: str, observation: bytes | None = None) -> None:
288
+ try:
289
+ self.store.commit(tool_call_id, observation)
290
+ except Exception: # noqa: BLE001
291
+ log.exception("could not commit %s", tool_call_id)
292
+
293
+ def record_failure_by_id(self, tool_call_id: str, error: str) -> None:
294
+ try:
295
+ self.store.fail(tool_call_id, error)
296
+ except Exception: # noqa: BLE001
297
+ log.exception("could not record failure for %s", tool_call_id)
298
+
299
+ def record_observation(self, tool_call_id: str,
300
+ observation: bytes | None = None,
301
+ error: str | None = None) -> None:
302
+ """The tool returned, and the harness has PERSISTED its result.
303
+
304
+ Only a caller that knows the result is durably in the conversation
305
+ history may use this -- Seam B qualifies because the SDK persists an
306
+ event before any caller callback runs (`docs/0042` §8.2). Otherwise
307
+ use `record_success` / `record_tool_error`, which assume nothing about
308
+ what the model has seen.
309
+
310
+ The record goes INTENT -> OBSERVED in one write, success or reported
311
+ failure alike. From then on an identical call is the model choosing to
312
+ repeat it, so `find_by_intent` stops aliasing to it (`docs/0045`).
313
+
314
+ One exception keeps the old shape: a replay-safe effect that reported
315
+ failure is FAILED. Repeating it is harmless, and FAILED already lets
316
+ the agent retry without a human.
317
+ """
318
+ try:
319
+ rec = self.store.lookup(tool_call_id)
320
+ if rec is None:
321
+ log.error("no record to deliver for %s", tool_call_id)
322
+ return
323
+ if error is not None and rec.effect_class.replay_safe:
324
+ self.store.fail(tool_call_id, error)
325
+ elif rec.state is EffectState.INTENT:
326
+ self.store.deliver(tool_call_id, observation, error)
327
+ elif rec.state is EffectState.COMMITTED:
328
+ self.store.observed(tool_call_id)
329
+ else:
330
+ log.warning("not delivering %s from state %s",
331
+ tool_call_id, rec.state.value)
332
+ except Exception: # noqa: BLE001
333
+ # Leaves the record INTENT (or COMMITTED): the conservative side.
334
+ # A later identical call is then matched and probed or blocked,
335
+ # never silently repeated.
336
+ log.exception("could not record the observation for %s", tool_call_id)
337
+
338
+ def mark_observed(self, tool_call_id: str) -> None:
339
+ """A substituted result has reached the model. COMMITTED -> OBSERVED.
340
+
341
+ Without this an effect recovered after a crash stays COMMITTED, and
342
+ every later deliberate repeat of it is answered with the old output.
343
+ """
344
+ try:
345
+ rec = self.store.lookup(tool_call_id)
346
+ if rec is not None and rec.state is EffectState.COMMITTED:
347
+ self.store.observed(tool_call_id)
348
+ except Exception: # noqa: BLE001
349
+ log.exception("could not mark %s observed", tool_call_id)
350
+
351
+ def record_tool_error(self, tool_call_id: str, error: str,
352
+ observation: bytes | None = None) -> None:
353
+ """The tool RAN and reported failure. That is a third thing.
354
+
355
+ Not `record_success`: committing a failed effect means a resume treats
356
+ work that never happened as done, and silently skips it. A real run
357
+ recorded five `exit=1` shell commands as COMMITTED (`docs/0031` §10).
358
+
359
+ Not `record_failure` either. `FAILED` means *provably did not land*,
360
+ and a non-zero exit is not proof of that — `echo x > a.txt && bad`
361
+ writes the file and exits 1. Asserting it did not land would be a lie
362
+ in the direction that permits a duplicate.
363
+
364
+ So it splits on whether repeating is safe, which is what the effect
365
+ class already encodes:
366
+
367
+ replay_safe -> FAILED, and the agent may simply try again
368
+ everything -> BLOCKED. Nobody knows whether it landed; that is
369
+ else the definition of the ambiguous case, and
370
+ `docs/0008` §6.5 says effect decisions fail closed.
371
+ """
372
+ rec = None
373
+ try:
374
+ rec = self.store.lookup(tool_call_id)
375
+ except Exception: # noqa: BLE001
376
+ log.exception("could not look up %s", tool_call_id)
377
+
378
+ if rec is not None and not rec.effect_class.replay_safe:
379
+ self._try_block(
380
+ tool_call_id,
381
+ f"the tool reported failure and this effect cannot be safely "
382
+ f"repeated, so whether it landed is unknown: {error[:300]}")
383
+ return
384
+ self.record_failure_by_id(tool_call_id, error)
385
+
386
+ def _try_block(self, tool_call_id: str, reason: str) -> None:
387
+ try:
388
+ if self.store.lookup(tool_call_id):
389
+ self.store.block(tool_call_id, reason)
390
+ except Exception: # noqa: BLE001
391
+ log.exception("could not block %s", tool_call_id)
@@ -0,0 +1,229 @@
1
+ """Seam A — the LiteLLM request hook. Spec: `docs/0012` §3.4.
2
+
3
+ Runs inside the LiteLLM proxy, before the request reaches a provider. It is
4
+ the only seam that is harness-independent: anything speaking the
5
+ OpenAI-compatible format — OpenHands, Claude Code, Aider, Cline — passes
6
+ through it (`docs/0013` §5).
7
+
8
+ What it does:
9
+
10
+ 1. **Turn-atomic routing** — an endpoint may change between turns and never
11
+ within one. A message list ending in unresolved tool calls is pinned to the
12
+ deployment that opened the turn (`docs/0010` §7.3). Model families mint
13
+ `tool_call_id` differently, and the effect ledger is keyed on it, so a
14
+ mid-turn switch could make a committed effect invisible.
15
+ 2. **Attribution** — stamps a turn-scoped trace id. The SDK already sends a
16
+ conversation-scoped `x-litellm-session-id` (`docs/0015` §5); this adds the
17
+ turn.
18
+ 3. **Telemetry** — records cost, tokens, cache hits and the chosen deployment.
19
+
20
+ Two things it deliberately does *not* do: decide anything about effects (that
21
+ is the gate's job at Seams B and C), and depend on the control plane being
22
+ reachable. It reads local state only.
23
+
24
+ **This module holds only the logic.** The LiteLLM binding lives in
25
+ `agentctl.adapters.litellm.hook` — the kernel must not import a data plane any
26
+ more than it imports a harness (`docs/0008` R6), and `tests/test_boundaries.py`
27
+ enforces that. `docs/0012` §1 originally placed the whole hook here; the split
28
+ is recorded in `docs/0021` §7.
29
+
30
+ **Seam A is best-effort and must be verified live.** LiteLLM has an open bug
31
+ where `async_pre_call_hook` is bypassed on the Anthropic `/v1/messages`
32
+ endpoint (#27518) and never fires for `/mcp/` tool calls (#25011) — silently,
33
+ in both cases. `docs/0021` asserts it actually fires rather than assuming.
34
+ """
35
+ from __future__ import annotations
36
+
37
+ import json
38
+ import logging
39
+ import time
40
+ from pathlib import Path
41
+ from typing import Any
42
+
43
+ log = logging.getLogger("agentctl.hook")
44
+
45
+
46
+ def ends_with_unresolved_tool_calls(messages: list[dict] | None) -> bool:
47
+ """Is the conversation mid-turn — awaiting tool results?
48
+
49
+ True when the last assistant message asked for tools and no matching tool
50
+ results follow it. Switching endpoints here is the hazard in `docs/0010`
51
+ §7.3.
52
+ """
53
+ if not messages:
54
+ return False
55
+
56
+ pending: set[str] = set()
57
+ for m in messages:
58
+ if not isinstance(m, dict):
59
+ continue
60
+ role = m.get("role")
61
+ if role == "assistant":
62
+ calls = m.get("tool_calls") or []
63
+ # A fresh assistant turn supersedes anything left dangling.
64
+ pending = {c.get("id") for c in calls if isinstance(c, dict) and c.get("id")}
65
+ elif role == "tool":
66
+ pending.discard(m.get("tool_call_id"))
67
+ elif role == "user":
68
+ # A user message means the turn was abandoned, not continued.
69
+ pending.clear()
70
+ return bool(pending)
71
+
72
+
73
+ def turn_signature(messages: list[dict] | None) -> str | None:
74
+ """Identify the turn: the id of its first outstanding tool call."""
75
+ if not messages:
76
+ return None
77
+ for m in reversed(messages):
78
+ if isinstance(m, dict) and m.get("role") == "assistant":
79
+ calls = m.get("tool_calls") or []
80
+ ids = sorted(c.get("id") for c in calls
81
+ if isinstance(c, dict) and c.get("id"))
82
+ return ids[0] if ids else None
83
+ return None
84
+
85
+
86
+ class TurnAffinity:
87
+ """Remembers which deployment opened each turn. Local, in-memory.
88
+
89
+ Deliberately not shared state: `docs/0008` R2 says the hot path must not
90
+ depend on anything that can be unreachable. Losing this map costs a pinning
91
+ opportunity, never correctness.
92
+ """
93
+
94
+ def __init__(self, ttl_s: float = 900.0):
95
+ self._pins: dict[str, tuple[str, float]] = {}
96
+ self.ttl_s = ttl_s
97
+
98
+ def remember(self, turn: str | None, deployment: str | None) -> None:
99
+ if turn and deployment:
100
+ self._pins[turn] = (deployment, time.time())
101
+
102
+ def pinned(self, turn: str | None) -> str | None:
103
+ if not turn:
104
+ return None
105
+ hit = self._pins.get(turn)
106
+ if not hit:
107
+ return None
108
+ deployment, at = hit
109
+ if time.time() - at > self.ttl_s:
110
+ self._pins.pop(turn, None)
111
+ return None
112
+ return deployment
113
+
114
+ def __len__(self) -> int:
115
+ return len(self._pins)
116
+
117
+
118
+ class RequestHook:
119
+ """Seam A. Subclasses LiteLLM's CustomLogger when it is importable.
120
+
121
+ Written so the logic is testable without the proxy: every method takes
122
+ plain dicts.
123
+ """
124
+
125
+ def __init__(self, telemetry_path: str | Path | None = None,
126
+ affinity: TurnAffinity | None = None):
127
+ self.affinity = affinity or TurnAffinity()
128
+ self.telemetry_path = Path(telemetry_path) if telemetry_path else None
129
+ self.calls: list[dict] = [] # live-fire evidence for M1
130
+ self.records: list[dict] = []
131
+
132
+ # ── pre-call ───────────────────────────────────────────────────────
133
+ def apply(self, data: dict) -> dict:
134
+ """Mutate an outbound request. The testable core of the hook."""
135
+ messages = data.get("messages") or []
136
+ meta = data.setdefault("metadata", {})
137
+
138
+ turn = turn_signature(messages)
139
+ mid_turn = ends_with_unresolved_tool_calls(messages)
140
+
141
+ if mid_turn:
142
+ # THE RULE: the turn is the atomic unit of routing.
143
+ if (pin := self.affinity.pinned(turn)):
144
+ data["model"] = pin
145
+ meta["agentctl_pinned"] = pin
146
+ meta["agentctl_mid_turn"] = True
147
+ else:
148
+ self.affinity.remember(turn, data.get("model"))
149
+ meta["agentctl_mid_turn"] = False
150
+
151
+ # Where the proxy ACTUALLY puts it (`docs/0042` I-09, `docs/0048`).
152
+ # OpenHands sends the conversation id as the `x-litellm-session-id`
153
+ # header, and the proxy maps it to `litellm_session_id` and
154
+ # `metadata.session_id` -- captured from a real pre-call `data` dict,
155
+ # not assumed. `extra_headers` is a CLIENT-side kwarg that never
156
+ # reaches the proxy, so reading it alone gave `unknown:` on every live
157
+ # record (23/23, then 10/10). It stays last, for a direct caller.
158
+ conv = (meta.get("conversation_id")
159
+ or data.get("litellm_session_id")
160
+ or meta.get("session_id")
161
+ or (data.get("extra_headers") or {}).get("x-litellm-session-id")
162
+ or "unknown")
163
+ meta["agentctl_trace_id"] = f"{conv}:{turn or 'turn0'}"
164
+
165
+ self.calls.append({
166
+ "model": data.get("model"),
167
+ "n_messages": len(messages),
168
+ "mid_turn": mid_turn,
169
+ "turn": turn,
170
+ "trace_id": meta["agentctl_trace_id"],
171
+ "ts": time.time(),
172
+ })
173
+ self._flush()
174
+ return data
175
+
176
+ # ── post-call ──────────────────────────────────────────────────────
177
+ def record(self, kwargs: dict, response: Any,
178
+ start: float | None = None, end: float | None = None) -> dict:
179
+ """Extract the telemetry the cost ledger will need (M5)."""
180
+ usage = _get(response, "usage") or {}
181
+ details = _get(usage, "prompt_tokens_details") or {}
182
+ # WHERE THE METADATA ACTUALLY IS (docs/0021 §4). Anything the
183
+ # pre-call hook writes into data["metadata"] arrives on the logging
184
+ # callback under litellm_params.metadata, NOT kwargs["metadata"],
185
+ # which is empty. Reading only the obvious place yields trace_id=None
186
+ # and silently breaks cost attribution.
187
+ meta = kwargs.get("metadata") or {}
188
+ nested = (kwargs.get("litellm_params") or {}).get("metadata") or {}
189
+ trace = (meta.get("agentctl_trace_id")
190
+ or nested.get("agentctl_trace_id")
191
+ or (nested.get("requester_metadata") or {}).get("agentctl_trace_id"))
192
+ info = (kwargs.get("litellm_params") or {}).get("model_info") or {}
193
+ rec = {
194
+ "model": kwargs.get("model"),
195
+ "deployment": info.get("id"),
196
+ # The generated config's own claim (`model_info.free`). The cost
197
+ # ledger's third state reads it (`docs/0042` I-10).
198
+ "free": info.get("free"),
199
+ "trace_id": trace,
200
+ "prompt_tokens": _get(usage, "prompt_tokens"),
201
+ "completion_tokens": _get(usage, "completion_tokens"),
202
+ "cached_tokens": _get(details, "cached_tokens"),
203
+ "cost": kwargs.get("response_cost"),
204
+ "latency_s": (end - start) if (start and end) else None,
205
+ "ts": time.time(),
206
+ }
207
+ self.records.append(rec)
208
+ self._flush()
209
+ return rec
210
+
211
+ def _flush(self) -> None:
212
+ """Evidence on disk, so a subprocess proxy can be inspected."""
213
+ if self.telemetry_path is None:
214
+ return
215
+ try:
216
+ self.telemetry_path.parent.mkdir(parents=True, exist_ok=True)
217
+ self.telemetry_path.write_text(
218
+ json.dumps({"calls": self.calls, "records": self.records},
219
+ indent=2, default=str), encoding="utf-8")
220
+ except Exception: # noqa: BLE001
221
+ log.exception("could not write hook telemetry")
222
+
223
+
224
+ def _get(obj: Any, key: str, default=None):
225
+ if obj is None:
226
+ return default
227
+ if isinstance(obj, dict):
228
+ return obj.get(key, default)
229
+ return getattr(obj, key, default)
File without changes