handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
agentctl/kernel/gate.py
ADDED
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""The effect gate. Spec: `docs/0011` §3, `docs/0012` §3.2.
|
|
2
|
+
|
|
3
|
+
The correctness core, and deliberately small. One rule dominates:
|
|
4
|
+
|
|
5
|
+
guard() MUST NEVER RAISE.
|
|
6
|
+
|
|
7
|
+
Any internal failure becomes BLOCK — fail closed (`docs/0008` §6.5). A gate
|
|
8
|
+
that crashes open is worse than no gate, because it creates false confidence.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
|
|
15
|
+
from .classify import Classifier
|
|
16
|
+
from .reconcile.base import (
|
|
17
|
+
DID_NOT_LAND, INCONCLUSIVE, LANDED, SAFE_TO_RETRY, ProbeRegistry,
|
|
18
|
+
)
|
|
19
|
+
from .ledger.models import (
|
|
20
|
+
EffectClass,
|
|
21
|
+
EffectState,
|
|
22
|
+
GateDecision,
|
|
23
|
+
ToolCall,
|
|
24
|
+
Verdict,
|
|
25
|
+
)
|
|
26
|
+
from .ledger.store import LedgerStore, StaleFence
|
|
27
|
+
|
|
28
|
+
log = logging.getLogger("agentctl.gate")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class EffectGate:
|
|
32
|
+
"""Decides whether a tool call may execute.
|
|
33
|
+
|
|
34
|
+
Seam-agnostic by design (`docs/0012` §3.2.1): it returns a decision, and
|
|
35
|
+
the binding acts on it. Seam B can honour BLOCK/ESCALATE; only Seam C can
|
|
36
|
+
honour SUBSTITUTE.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
store: LedgerStore,
|
|
42
|
+
classifier: Classifier | None = None,
|
|
43
|
+
probes: ProbeRegistry | None = None,
|
|
44
|
+
fence: int = 0,
|
|
45
|
+
):
|
|
46
|
+
self.store = store
|
|
47
|
+
self.classifier = classifier or Classifier()
|
|
48
|
+
self.probes = probes if probes is not None else ProbeRegistry()
|
|
49
|
+
self.fence = fence
|
|
50
|
+
|
|
51
|
+
# ── the decision ───────────────────────────────────────────────────
|
|
52
|
+
def guard(self, call: ToolCall) -> GateDecision:
|
|
53
|
+
try:
|
|
54
|
+
return self._guard(call)
|
|
55
|
+
except StaleFence as exc:
|
|
56
|
+
# Not a bug in the gate: another process took this conversation
|
|
57
|
+
# over (`docs/0046`). Said as such, rather than as "gate error,
|
|
58
|
+
# failing closed: StaleFence(...)" -- accurate, and meaningless to
|
|
59
|
+
# the person reading it (`docs/0048`).
|
|
60
|
+
log.warning("superseded: %s", exc)
|
|
61
|
+
return GateDecision(
|
|
62
|
+
Verdict.BLOCK,
|
|
63
|
+
reason="another process has taken over this conversation, so "
|
|
64
|
+
"this run may not start new actions; let the other "
|
|
65
|
+
"run finish, or stop this one")
|
|
66
|
+
except Exception as exc: # noqa: BLE001
|
|
67
|
+
# Fail closed. Never let a gate bug become an unguarded effect.
|
|
68
|
+
log.exception("gate failure for %s", call.tool_call_id)
|
|
69
|
+
self._try_block(call.tool_call_id, f"gate error: {exc!r}")
|
|
70
|
+
return GateDecision(
|
|
71
|
+
Verdict.BLOCK, reason=f"gate error, failing closed: {exc!r}"
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
def _guard(self, call: ToolCall) -> GateDecision:
|
|
75
|
+
cls = self.classifier.classify(call)
|
|
76
|
+
rec = self.store.lookup(call.tool_call_id)
|
|
77
|
+
|
|
78
|
+
if rec is None and not cls.replay_safe:
|
|
79
|
+
# The id is model-minted, so a different model answering the same
|
|
80
|
+
# question produces a different one for the identical call
|
|
81
|
+
# (`docs/0023` §4). Before calling this a first sighting, ask
|
|
82
|
+
# whether this exact effect is already on record under another id.
|
|
83
|
+
twin = self.store.find_by_intent(call.conversation_id,
|
|
84
|
+
call.intent_hash())
|
|
85
|
+
if twin is not None:
|
|
86
|
+
log.info("matched %s to prior effect %s by intent hash",
|
|
87
|
+
call.tool_call_id, twin.tool_call_id)
|
|
88
|
+
return self._decide_on(call, twin, cls, aliased=True)
|
|
89
|
+
|
|
90
|
+
# First sighting — the common case.
|
|
91
|
+
if rec is None:
|
|
92
|
+
self.store.write_intent(call, cls, self.fence, self._capture(call, cls))
|
|
93
|
+
return GateDecision(Verdict.EXECUTE, cls)
|
|
94
|
+
|
|
95
|
+
# Same id, different arguments. Substituting here would return the
|
|
96
|
+
# wrong observation, so refuse outright.
|
|
97
|
+
if rec.intent_hash != call.intent_hash():
|
|
98
|
+
return GateDecision(
|
|
99
|
+
Verdict.BLOCK, cls,
|
|
100
|
+
reason="tool_call_id reused with different arguments",
|
|
101
|
+
)
|
|
102
|
+
return self._decide_on(call, rec, cls)
|
|
103
|
+
|
|
104
|
+
def _decide_on(self, call: ToolCall, rec, cls: EffectClass,
|
|
105
|
+
aliased: bool = False) -> GateDecision:
|
|
106
|
+
"""Decide against a record, which may be under a different id."""
|
|
107
|
+
note = (f" (matched to {rec.tool_call_id} by intent hash)"
|
|
108
|
+
if aliased else "")
|
|
109
|
+
|
|
110
|
+
# OBSERVED reaches here only by its own id: the SDK re-driving the very
|
|
111
|
+
# action whose result it persisted. `find_by_intent` never aliases to
|
|
112
|
+
# one, because a re-minted twin of an observed call is the model's
|
|
113
|
+
# decision to repeat (`docs/0045`).
|
|
114
|
+
if rec.state in (EffectState.COMMITTED, EffectState.OBSERVED):
|
|
115
|
+
return GateDecision(Verdict.SUBSTITUTE, cls,
|
|
116
|
+
observation=rec.observation,
|
|
117
|
+
reason=f"already recorded{note}" if note else None,
|
|
118
|
+
record_id=rec.tool_call_id)
|
|
119
|
+
|
|
120
|
+
if rec.state is EffectState.FAILED:
|
|
121
|
+
# Re-capture: the world may have moved since the failed attempt.
|
|
122
|
+
self.store.write_intent(call, cls, self.fence, self._capture(call, cls))
|
|
123
|
+
return GateDecision(Verdict.EXECUTE, cls)
|
|
124
|
+
|
|
125
|
+
if rec.state is EffectState.BLOCKED:
|
|
126
|
+
return GateDecision(
|
|
127
|
+
Verdict.BLOCK, cls,
|
|
128
|
+
reason=(rec.error or "previously blocked; awaiting human decision")
|
|
129
|
+
+ note,
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
# rec.state is INTENT — the irreducible ambiguity (docs/0008 §6.5).
|
|
133
|
+
return self._resolve_ambiguous(call, rec, cls)
|
|
134
|
+
|
|
135
|
+
def _resolve_ambiguous(self, call, rec, cls: EffectClass) -> GateDecision:
|
|
136
|
+
"""We cannot tell whether the effect landed. Decide by class.
|
|
137
|
+
|
|
138
|
+
Every ledger write here targets `rec.tool_call_id`, never
|
|
139
|
+
`call.tool_call_id`. Under intent-hash aliasing those differ, and
|
|
140
|
+
writing to the caller's id raises `IllegalTransition` -- which the
|
|
141
|
+
fail-closed wrapper then turns into a BLOCK, producing a correct
|
|
142
|
+
outcome by an incorrect route and stranding the record (`docs/0024`).
|
|
143
|
+
"""
|
|
144
|
+
rid = rec.tool_call_id
|
|
145
|
+
|
|
146
|
+
if cls.replay_safe:
|
|
147
|
+
# Repeating is harmless, so the window does not matter.
|
|
148
|
+
return GateDecision(Verdict.EXECUTE, cls)
|
|
149
|
+
|
|
150
|
+
if cls is EffectClass.DESTRUCTIVE:
|
|
151
|
+
self.store.block(rid, "destructive effect, outcome unknown")
|
|
152
|
+
return GateDecision(
|
|
153
|
+
Verdict.ESCALATE, cls,
|
|
154
|
+
reason="destructive effect with unknown outcome; a human must decide",
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
# NON_IDEMPOTENT_WRITE / EXTERNAL: ask the world if it can answer.
|
|
158
|
+
probe = self.probes.for_call(call, self.classifier.probe_for(call))
|
|
159
|
+
verdict = INCONCLUSIVE
|
|
160
|
+
if probe is not None:
|
|
161
|
+
try:
|
|
162
|
+
verdict = probe.probe(call, rec)
|
|
163
|
+
except Exception: # noqa: BLE001
|
|
164
|
+
log.exception("probe %s raised", getattr(probe, "name", "?"))
|
|
165
|
+
verdict = INCONCLUSIVE
|
|
166
|
+
|
|
167
|
+
if verdict == LANDED and not self._sole_writer(rec):
|
|
168
|
+
# The probe reasoned from world state, but this call was not
|
|
169
|
+
# the only thing writing to that world. The change it saw may
|
|
170
|
+
# be a sibling's (`docs/0038` §3). Downgrade rather than
|
|
171
|
+
# attribute: a dropped effect recorded as COMMITTED is the one
|
|
172
|
+
# outcome this system exists to prevent.
|
|
173
|
+
log.warning("%s probe said LANDED, but a sibling effect "
|
|
174
|
+
"committed inside the window; not attributing", probe.name)
|
|
175
|
+
verdict = INCONCLUSIVE
|
|
176
|
+
|
|
177
|
+
if verdict == LANDED:
|
|
178
|
+
self.store.reconcile(rid, verdict, landed=True)
|
|
179
|
+
return GateDecision(
|
|
180
|
+
Verdict.SUBSTITUTE, cls, observation=rec.observation,
|
|
181
|
+
reason=f"{probe.name} probe: the effect already landed",
|
|
182
|
+
record_id=rid,
|
|
183
|
+
)
|
|
184
|
+
if verdict == DID_NOT_LAND:
|
|
185
|
+
self.store.reconcile(rid, verdict, landed=False)
|
|
186
|
+
self.store.write_intent(call, cls, self.fence,
|
|
187
|
+
self._capture(call, cls))
|
|
188
|
+
return GateDecision(
|
|
189
|
+
Verdict.EXECUTE, cls,
|
|
190
|
+
reason=f"{probe.name} probe: the effect did not land",
|
|
191
|
+
)
|
|
192
|
+
if verdict == SAFE_TO_RETRY:
|
|
193
|
+
# We cannot tell whether it landed, and we do not need to: the
|
|
194
|
+
# call carries an idempotency key, so the remote collapses a
|
|
195
|
+
# retry into the original effect (`docs/0020`).
|
|
196
|
+
self.store.reconcile(rid, verdict, landed=False)
|
|
197
|
+
self.store.write_intent(call, cls, self.fence,
|
|
198
|
+
self._capture(call, cls))
|
|
199
|
+
return GateDecision(
|
|
200
|
+
Verdict.EXECUTE, cls,
|
|
201
|
+
reason=(f"{probe.name} probe: outcome unknown, but the call "
|
|
202
|
+
f"is idempotency-keyed so a retry is safe"),
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
# No probe, or inconclusive. Fail closed.
|
|
206
|
+
detail = ("no probe available" if probe is None
|
|
207
|
+
else f"{probe.name} probe inconclusive")
|
|
208
|
+
reason = (f"cannot determine whether {call.tool_name} already "
|
|
209
|
+
f"executed ({detail})")
|
|
210
|
+
self.store.block(rid, reason)
|
|
211
|
+
return GateDecision(Verdict.BLOCK, cls, reason=reason)
|
|
212
|
+
|
|
213
|
+
def _sole_writer(self, rec) -> bool:
|
|
214
|
+
"""Was this call the only thing writing the world its probe just read?
|
|
215
|
+
|
|
216
|
+
A probe that reasons from world state — HEAD moved, so my commit
|
|
217
|
+
landed — is sound only under that assumption, and it is the probe's
|
|
218
|
+
own stated one. It holds across conversations. It does not hold inside
|
|
219
|
+
a batch, where a sibling can execute after this call was fingerprinted
|
|
220
|
+
(`docs/0038` §3).
|
|
221
|
+
|
|
222
|
+
Only effects that can change the world count; a read cannot. And when
|
|
223
|
+
the question cannot be answered, the answer is no: a world-state
|
|
224
|
+
verdict we cannot justify is exactly what must not be trusted.
|
|
225
|
+
"""
|
|
226
|
+
try:
|
|
227
|
+
siblings = self.store.committed_since(
|
|
228
|
+
rec.conversation_id, rec.started_at, rec.tool_call_id)
|
|
229
|
+
except Exception: # noqa: BLE001
|
|
230
|
+
log.exception("could not check for concurrent effects on %s",
|
|
231
|
+
rec.tool_call_id)
|
|
232
|
+
return False
|
|
233
|
+
return not any(not s.effect_class.replay_safe for s in siblings)
|
|
234
|
+
|
|
235
|
+
def recapture(self, call: ToolCall) -> None:
|
|
236
|
+
"""Re-fingerprint the world for `call`, immediately before it runs.
|
|
237
|
+
|
|
238
|
+
Called from Seam C, the only seam that sees a call at its own execution
|
|
239
|
+
moment. Seam B decided for the whole batch; by the time this call is
|
|
240
|
+
actually about to execute, its siblings may already have changed the
|
|
241
|
+
world its probe will later reason about (`docs/0038` §3).
|
|
242
|
+
|
|
243
|
+
Never raises and never decides. A failure here costs a stale
|
|
244
|
+
fingerprint, which the probe reads as INCONCLUSIVE and the gate fails
|
|
245
|
+
closed on — the same direction as every other unknown in this system.
|
|
246
|
+
"""
|
|
247
|
+
try:
|
|
248
|
+
cls = self.classifier.classify(call)
|
|
249
|
+
if cls.replay_safe:
|
|
250
|
+
return
|
|
251
|
+
self.store.refresh_pre_state(call.tool_call_id,
|
|
252
|
+
self._capture(call, cls))
|
|
253
|
+
except Exception: # noqa: BLE001
|
|
254
|
+
log.exception("recapture failed for %s", call.tool_call_id)
|
|
255
|
+
|
|
256
|
+
def _capture(self, call: ToolCall, cls: EffectClass) -> str | None:
|
|
257
|
+
"""Fingerprint the world before acting, so a probe can compare later.
|
|
258
|
+
|
|
259
|
+
Only for classes that can reach the ambiguous branch -- there is no
|
|
260
|
+
point paying for it on a read.
|
|
261
|
+
"""
|
|
262
|
+
if cls.replay_safe:
|
|
263
|
+
return None
|
|
264
|
+
probe = self.probes.for_call(call, self.classifier.probe_for(call))
|
|
265
|
+
if probe is None:
|
|
266
|
+
return None
|
|
267
|
+
try:
|
|
268
|
+
state = probe.capture(call)
|
|
269
|
+
except Exception: # noqa: BLE001
|
|
270
|
+
log.exception("probe %s capture raised", getattr(probe, "name", "?"))
|
|
271
|
+
return None
|
|
272
|
+
return json.dumps(state) if state else None
|
|
273
|
+
|
|
274
|
+
# ── post-execution bookkeeping ─────────────────────────────────────
|
|
275
|
+
def record_success(self, call: ToolCall, observation: bytes | None = None) -> None:
|
|
276
|
+
try:
|
|
277
|
+
self.store.commit(call.tool_call_id, observation)
|
|
278
|
+
except Exception: # noqa: BLE001
|
|
279
|
+
log.exception("could not commit %s", call.tool_call_id)
|
|
280
|
+
|
|
281
|
+
def record_failure(self, call: ToolCall, error: str) -> None:
|
|
282
|
+
try:
|
|
283
|
+
self.store.fail(call.tool_call_id, error)
|
|
284
|
+
except Exception: # noqa: BLE001
|
|
285
|
+
log.exception("could not record failure for %s", call.tool_call_id)
|
|
286
|
+
|
|
287
|
+
def record_success_by_id(self, tool_call_id: str, observation: bytes | None = None) -> None:
|
|
288
|
+
try:
|
|
289
|
+
self.store.commit(tool_call_id, observation)
|
|
290
|
+
except Exception: # noqa: BLE001
|
|
291
|
+
log.exception("could not commit %s", tool_call_id)
|
|
292
|
+
|
|
293
|
+
def record_failure_by_id(self, tool_call_id: str, error: str) -> None:
|
|
294
|
+
try:
|
|
295
|
+
self.store.fail(tool_call_id, error)
|
|
296
|
+
except Exception: # noqa: BLE001
|
|
297
|
+
log.exception("could not record failure for %s", tool_call_id)
|
|
298
|
+
|
|
299
|
+
def record_observation(self, tool_call_id: str,
|
|
300
|
+
observation: bytes | None = None,
|
|
301
|
+
error: str | None = None) -> None:
|
|
302
|
+
"""The tool returned, and the harness has PERSISTED its result.
|
|
303
|
+
|
|
304
|
+
Only a caller that knows the result is durably in the conversation
|
|
305
|
+
history may use this -- Seam B qualifies because the SDK persists an
|
|
306
|
+
event before any caller callback runs (`docs/0042` §8.2). Otherwise
|
|
307
|
+
use `record_success` / `record_tool_error`, which assume nothing about
|
|
308
|
+
what the model has seen.
|
|
309
|
+
|
|
310
|
+
The record goes INTENT -> OBSERVED in one write, success or reported
|
|
311
|
+
failure alike. From then on an identical call is the model choosing to
|
|
312
|
+
repeat it, so `find_by_intent` stops aliasing to it (`docs/0045`).
|
|
313
|
+
|
|
314
|
+
One exception keeps the old shape: a replay-safe effect that reported
|
|
315
|
+
failure is FAILED. Repeating it is harmless, and FAILED already lets
|
|
316
|
+
the agent retry without a human.
|
|
317
|
+
"""
|
|
318
|
+
try:
|
|
319
|
+
rec = self.store.lookup(tool_call_id)
|
|
320
|
+
if rec is None:
|
|
321
|
+
log.error("no record to deliver for %s", tool_call_id)
|
|
322
|
+
return
|
|
323
|
+
if error is not None and rec.effect_class.replay_safe:
|
|
324
|
+
self.store.fail(tool_call_id, error)
|
|
325
|
+
elif rec.state is EffectState.INTENT:
|
|
326
|
+
self.store.deliver(tool_call_id, observation, error)
|
|
327
|
+
elif rec.state is EffectState.COMMITTED:
|
|
328
|
+
self.store.observed(tool_call_id)
|
|
329
|
+
else:
|
|
330
|
+
log.warning("not delivering %s from state %s",
|
|
331
|
+
tool_call_id, rec.state.value)
|
|
332
|
+
except Exception: # noqa: BLE001
|
|
333
|
+
# Leaves the record INTENT (or COMMITTED): the conservative side.
|
|
334
|
+
# A later identical call is then matched and probed or blocked,
|
|
335
|
+
# never silently repeated.
|
|
336
|
+
log.exception("could not record the observation for %s", tool_call_id)
|
|
337
|
+
|
|
338
|
+
def mark_observed(self, tool_call_id: str) -> None:
|
|
339
|
+
"""A substituted result has reached the model. COMMITTED -> OBSERVED.
|
|
340
|
+
|
|
341
|
+
Without this an effect recovered after a crash stays COMMITTED, and
|
|
342
|
+
every later deliberate repeat of it is answered with the old output.
|
|
343
|
+
"""
|
|
344
|
+
try:
|
|
345
|
+
rec = self.store.lookup(tool_call_id)
|
|
346
|
+
if rec is not None and rec.state is EffectState.COMMITTED:
|
|
347
|
+
self.store.observed(tool_call_id)
|
|
348
|
+
except Exception: # noqa: BLE001
|
|
349
|
+
log.exception("could not mark %s observed", tool_call_id)
|
|
350
|
+
|
|
351
|
+
def record_tool_error(self, tool_call_id: str, error: str,
|
|
352
|
+
observation: bytes | None = None) -> None:
|
|
353
|
+
"""The tool RAN and reported failure. That is a third thing.
|
|
354
|
+
|
|
355
|
+
Not `record_success`: committing a failed effect means a resume treats
|
|
356
|
+
work that never happened as done, and silently skips it. A real run
|
|
357
|
+
recorded five `exit=1` shell commands as COMMITTED (`docs/0031` §10).
|
|
358
|
+
|
|
359
|
+
Not `record_failure` either. `FAILED` means *provably did not land*,
|
|
360
|
+
and a non-zero exit is not proof of that — `echo x > a.txt && bad`
|
|
361
|
+
writes the file and exits 1. Asserting it did not land would be a lie
|
|
362
|
+
in the direction that permits a duplicate.
|
|
363
|
+
|
|
364
|
+
So it splits on whether repeating is safe, which is what the effect
|
|
365
|
+
class already encodes:
|
|
366
|
+
|
|
367
|
+
replay_safe -> FAILED, and the agent may simply try again
|
|
368
|
+
everything -> BLOCKED. Nobody knows whether it landed; that is
|
|
369
|
+
else the definition of the ambiguous case, and
|
|
370
|
+
`docs/0008` §6.5 says effect decisions fail closed.
|
|
371
|
+
"""
|
|
372
|
+
rec = None
|
|
373
|
+
try:
|
|
374
|
+
rec = self.store.lookup(tool_call_id)
|
|
375
|
+
except Exception: # noqa: BLE001
|
|
376
|
+
log.exception("could not look up %s", tool_call_id)
|
|
377
|
+
|
|
378
|
+
if rec is not None and not rec.effect_class.replay_safe:
|
|
379
|
+
self._try_block(
|
|
380
|
+
tool_call_id,
|
|
381
|
+
f"the tool reported failure and this effect cannot be safely "
|
|
382
|
+
f"repeated, so whether it landed is unknown: {error[:300]}")
|
|
383
|
+
return
|
|
384
|
+
self.record_failure_by_id(tool_call_id, error)
|
|
385
|
+
|
|
386
|
+
def _try_block(self, tool_call_id: str, reason: str) -> None:
|
|
387
|
+
try:
|
|
388
|
+
if self.store.lookup(tool_call_id):
|
|
389
|
+
self.store.block(tool_call_id, reason)
|
|
390
|
+
except Exception: # noqa: BLE001
|
|
391
|
+
log.exception("could not block %s", tool_call_id)
|
agentctl/kernel/hook.py
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""Seam A — the LiteLLM request hook. Spec: `docs/0012` §3.4.
|
|
2
|
+
|
|
3
|
+
Runs inside the LiteLLM proxy, before the request reaches a provider. It is
|
|
4
|
+
the only seam that is harness-independent: anything speaking the
|
|
5
|
+
OpenAI-compatible format — OpenHands, Claude Code, Aider, Cline — passes
|
|
6
|
+
through it (`docs/0013` §5).
|
|
7
|
+
|
|
8
|
+
What it does:
|
|
9
|
+
|
|
10
|
+
1. **Turn-atomic routing** — an endpoint may change between turns and never
|
|
11
|
+
within one. A message list ending in unresolved tool calls is pinned to the
|
|
12
|
+
deployment that opened the turn (`docs/0010` §7.3). Model families mint
|
|
13
|
+
`tool_call_id` differently, and the effect ledger is keyed on it, so a
|
|
14
|
+
mid-turn switch could make a committed effect invisible.
|
|
15
|
+
2. **Attribution** — stamps a turn-scoped trace id. The SDK already sends a
|
|
16
|
+
conversation-scoped `x-litellm-session-id` (`docs/0015` §5); this adds the
|
|
17
|
+
turn.
|
|
18
|
+
3. **Telemetry** — records cost, tokens, cache hits and the chosen deployment.
|
|
19
|
+
|
|
20
|
+
Two things it deliberately does *not* do: decide anything about effects (that
|
|
21
|
+
is the gate's job at Seams B and C), and depend on the control plane being
|
|
22
|
+
reachable. It reads local state only.
|
|
23
|
+
|
|
24
|
+
**This module holds only the logic.** The LiteLLM binding lives in
|
|
25
|
+
`agentctl.adapters.litellm.hook` — the kernel must not import a data plane any
|
|
26
|
+
more than it imports a harness (`docs/0008` R6), and `tests/test_boundaries.py`
|
|
27
|
+
enforces that. `docs/0012` §1 originally placed the whole hook here; the split
|
|
28
|
+
is recorded in `docs/0021` §7.
|
|
29
|
+
|
|
30
|
+
**Seam A is best-effort and must be verified live.** LiteLLM has an open bug
|
|
31
|
+
where `async_pre_call_hook` is bypassed on the Anthropic `/v1/messages`
|
|
32
|
+
endpoint (#27518) and never fires for `/mcp/` tool calls (#25011) — silently,
|
|
33
|
+
in both cases. `docs/0021` asserts it actually fires rather than assuming.
|
|
34
|
+
"""
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import json
|
|
38
|
+
import logging
|
|
39
|
+
import time
|
|
40
|
+
from pathlib import Path
|
|
41
|
+
from typing import Any
|
|
42
|
+
|
|
43
|
+
log = logging.getLogger("agentctl.hook")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def ends_with_unresolved_tool_calls(messages: list[dict] | None) -> bool:
|
|
47
|
+
"""Is the conversation mid-turn — awaiting tool results?
|
|
48
|
+
|
|
49
|
+
True when the last assistant message asked for tools and no matching tool
|
|
50
|
+
results follow it. Switching endpoints here is the hazard in `docs/0010`
|
|
51
|
+
§7.3.
|
|
52
|
+
"""
|
|
53
|
+
if not messages:
|
|
54
|
+
return False
|
|
55
|
+
|
|
56
|
+
pending: set[str] = set()
|
|
57
|
+
for m in messages:
|
|
58
|
+
if not isinstance(m, dict):
|
|
59
|
+
continue
|
|
60
|
+
role = m.get("role")
|
|
61
|
+
if role == "assistant":
|
|
62
|
+
calls = m.get("tool_calls") or []
|
|
63
|
+
# A fresh assistant turn supersedes anything left dangling.
|
|
64
|
+
pending = {c.get("id") for c in calls if isinstance(c, dict) and c.get("id")}
|
|
65
|
+
elif role == "tool":
|
|
66
|
+
pending.discard(m.get("tool_call_id"))
|
|
67
|
+
elif role == "user":
|
|
68
|
+
# A user message means the turn was abandoned, not continued.
|
|
69
|
+
pending.clear()
|
|
70
|
+
return bool(pending)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def turn_signature(messages: list[dict] | None) -> str | None:
|
|
74
|
+
"""Identify the turn: the id of its first outstanding tool call."""
|
|
75
|
+
if not messages:
|
|
76
|
+
return None
|
|
77
|
+
for m in reversed(messages):
|
|
78
|
+
if isinstance(m, dict) and m.get("role") == "assistant":
|
|
79
|
+
calls = m.get("tool_calls") or []
|
|
80
|
+
ids = sorted(c.get("id") for c in calls
|
|
81
|
+
if isinstance(c, dict) and c.get("id"))
|
|
82
|
+
return ids[0] if ids else None
|
|
83
|
+
return None
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class TurnAffinity:
|
|
87
|
+
"""Remembers which deployment opened each turn. Local, in-memory.
|
|
88
|
+
|
|
89
|
+
Deliberately not shared state: `docs/0008` R2 says the hot path must not
|
|
90
|
+
depend on anything that can be unreachable. Losing this map costs a pinning
|
|
91
|
+
opportunity, never correctness.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
def __init__(self, ttl_s: float = 900.0):
|
|
95
|
+
self._pins: dict[str, tuple[str, float]] = {}
|
|
96
|
+
self.ttl_s = ttl_s
|
|
97
|
+
|
|
98
|
+
def remember(self, turn: str | None, deployment: str | None) -> None:
|
|
99
|
+
if turn and deployment:
|
|
100
|
+
self._pins[turn] = (deployment, time.time())
|
|
101
|
+
|
|
102
|
+
def pinned(self, turn: str | None) -> str | None:
|
|
103
|
+
if not turn:
|
|
104
|
+
return None
|
|
105
|
+
hit = self._pins.get(turn)
|
|
106
|
+
if not hit:
|
|
107
|
+
return None
|
|
108
|
+
deployment, at = hit
|
|
109
|
+
if time.time() - at > self.ttl_s:
|
|
110
|
+
self._pins.pop(turn, None)
|
|
111
|
+
return None
|
|
112
|
+
return deployment
|
|
113
|
+
|
|
114
|
+
def __len__(self) -> int:
|
|
115
|
+
return len(self._pins)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class RequestHook:
|
|
119
|
+
"""Seam A. Subclasses LiteLLM's CustomLogger when it is importable.
|
|
120
|
+
|
|
121
|
+
Written so the logic is testable without the proxy: every method takes
|
|
122
|
+
plain dicts.
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
def __init__(self, telemetry_path: str | Path | None = None,
|
|
126
|
+
affinity: TurnAffinity | None = None):
|
|
127
|
+
self.affinity = affinity or TurnAffinity()
|
|
128
|
+
self.telemetry_path = Path(telemetry_path) if telemetry_path else None
|
|
129
|
+
self.calls: list[dict] = [] # live-fire evidence for M1
|
|
130
|
+
self.records: list[dict] = []
|
|
131
|
+
|
|
132
|
+
# ── pre-call ───────────────────────────────────────────────────────
|
|
133
|
+
def apply(self, data: dict) -> dict:
|
|
134
|
+
"""Mutate an outbound request. The testable core of the hook."""
|
|
135
|
+
messages = data.get("messages") or []
|
|
136
|
+
meta = data.setdefault("metadata", {})
|
|
137
|
+
|
|
138
|
+
turn = turn_signature(messages)
|
|
139
|
+
mid_turn = ends_with_unresolved_tool_calls(messages)
|
|
140
|
+
|
|
141
|
+
if mid_turn:
|
|
142
|
+
# THE RULE: the turn is the atomic unit of routing.
|
|
143
|
+
if (pin := self.affinity.pinned(turn)):
|
|
144
|
+
data["model"] = pin
|
|
145
|
+
meta["agentctl_pinned"] = pin
|
|
146
|
+
meta["agentctl_mid_turn"] = True
|
|
147
|
+
else:
|
|
148
|
+
self.affinity.remember(turn, data.get("model"))
|
|
149
|
+
meta["agentctl_mid_turn"] = False
|
|
150
|
+
|
|
151
|
+
# Where the proxy ACTUALLY puts it (`docs/0042` I-09, `docs/0048`).
|
|
152
|
+
# OpenHands sends the conversation id as the `x-litellm-session-id`
|
|
153
|
+
# header, and the proxy maps it to `litellm_session_id` and
|
|
154
|
+
# `metadata.session_id` -- captured from a real pre-call `data` dict,
|
|
155
|
+
# not assumed. `extra_headers` is a CLIENT-side kwarg that never
|
|
156
|
+
# reaches the proxy, so reading it alone gave `unknown:` on every live
|
|
157
|
+
# record (23/23, then 10/10). It stays last, for a direct caller.
|
|
158
|
+
conv = (meta.get("conversation_id")
|
|
159
|
+
or data.get("litellm_session_id")
|
|
160
|
+
or meta.get("session_id")
|
|
161
|
+
or (data.get("extra_headers") or {}).get("x-litellm-session-id")
|
|
162
|
+
or "unknown")
|
|
163
|
+
meta["agentctl_trace_id"] = f"{conv}:{turn or 'turn0'}"
|
|
164
|
+
|
|
165
|
+
self.calls.append({
|
|
166
|
+
"model": data.get("model"),
|
|
167
|
+
"n_messages": len(messages),
|
|
168
|
+
"mid_turn": mid_turn,
|
|
169
|
+
"turn": turn,
|
|
170
|
+
"trace_id": meta["agentctl_trace_id"],
|
|
171
|
+
"ts": time.time(),
|
|
172
|
+
})
|
|
173
|
+
self._flush()
|
|
174
|
+
return data
|
|
175
|
+
|
|
176
|
+
# ── post-call ──────────────────────────────────────────────────────
|
|
177
|
+
def record(self, kwargs: dict, response: Any,
|
|
178
|
+
start: float | None = None, end: float | None = None) -> dict:
|
|
179
|
+
"""Extract the telemetry the cost ledger will need (M5)."""
|
|
180
|
+
usage = _get(response, "usage") or {}
|
|
181
|
+
details = _get(usage, "prompt_tokens_details") or {}
|
|
182
|
+
# WHERE THE METADATA ACTUALLY IS (docs/0021 §4). Anything the
|
|
183
|
+
# pre-call hook writes into data["metadata"] arrives on the logging
|
|
184
|
+
# callback under litellm_params.metadata, NOT kwargs["metadata"],
|
|
185
|
+
# which is empty. Reading only the obvious place yields trace_id=None
|
|
186
|
+
# and silently breaks cost attribution.
|
|
187
|
+
meta = kwargs.get("metadata") or {}
|
|
188
|
+
nested = (kwargs.get("litellm_params") or {}).get("metadata") or {}
|
|
189
|
+
trace = (meta.get("agentctl_trace_id")
|
|
190
|
+
or nested.get("agentctl_trace_id")
|
|
191
|
+
or (nested.get("requester_metadata") or {}).get("agentctl_trace_id"))
|
|
192
|
+
info = (kwargs.get("litellm_params") or {}).get("model_info") or {}
|
|
193
|
+
rec = {
|
|
194
|
+
"model": kwargs.get("model"),
|
|
195
|
+
"deployment": info.get("id"),
|
|
196
|
+
# The generated config's own claim (`model_info.free`). The cost
|
|
197
|
+
# ledger's third state reads it (`docs/0042` I-10).
|
|
198
|
+
"free": info.get("free"),
|
|
199
|
+
"trace_id": trace,
|
|
200
|
+
"prompt_tokens": _get(usage, "prompt_tokens"),
|
|
201
|
+
"completion_tokens": _get(usage, "completion_tokens"),
|
|
202
|
+
"cached_tokens": _get(details, "cached_tokens"),
|
|
203
|
+
"cost": kwargs.get("response_cost"),
|
|
204
|
+
"latency_s": (end - start) if (start and end) else None,
|
|
205
|
+
"ts": time.time(),
|
|
206
|
+
}
|
|
207
|
+
self.records.append(rec)
|
|
208
|
+
self._flush()
|
|
209
|
+
return rec
|
|
210
|
+
|
|
211
|
+
def _flush(self) -> None:
|
|
212
|
+
"""Evidence on disk, so a subprocess proxy can be inspected."""
|
|
213
|
+
if self.telemetry_path is None:
|
|
214
|
+
return
|
|
215
|
+
try:
|
|
216
|
+
self.telemetry_path.parent.mkdir(parents=True, exist_ok=True)
|
|
217
|
+
self.telemetry_path.write_text(
|
|
218
|
+
json.dumps({"calls": self.calls, "records": self.records},
|
|
219
|
+
indent=2, default=str), encoding="utf-8")
|
|
220
|
+
except Exception: # noqa: BLE001
|
|
221
|
+
log.exception("could not write hook telemetry")
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _get(obj: Any, key: str, default=None):
|
|
225
|
+
if obj is None:
|
|
226
|
+
return default
|
|
227
|
+
if isinstance(obj, dict):
|
|
228
|
+
return obj.get(key, default)
|
|
229
|
+
return getattr(obj, key, default)
|
|
File without changes
|