handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,596 @@
|
|
|
1
|
+
"""Durable effect ledger. Local, single-writer, fsynced.
|
|
2
|
+
|
|
3
|
+
Spec: `docs/0012` §2.1, §3.1.
|
|
4
|
+
|
|
5
|
+
The one rule that matters: **a write returns only when it is durable.**
|
|
6
|
+
`synchronous = FULL` is deliberate — `NORMAL` can lose the last commit on power
|
|
7
|
+
failure, and that is precisely the record whose absence causes a double effect.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import sqlite3
|
|
12
|
+
import time
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from .models import (
|
|
16
|
+
TRANSITIONS,
|
|
17
|
+
EffectClass,
|
|
18
|
+
EffectRecord,
|
|
19
|
+
EffectState,
|
|
20
|
+
IllegalTransition,
|
|
21
|
+
ToolCall,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
SCHEMA = (Path(__file__).parent / "schema.sql").read_text(encoding="utf-8")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class LeaseHeld(RuntimeError):
|
|
28
|
+
"""Another live holder owns this conversation. Refuse to act."""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class StaleFence(RuntimeError):
|
|
32
|
+
"""A superseded holder tried to write. Its lease was taken over."""
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class AmbiguousPrefix(LookupError):
|
|
36
|
+
"""More than one effect starts with this. Carries the candidates."""
|
|
37
|
+
|
|
38
|
+
def __init__(self, prefix: str, matches: list[str]):
|
|
39
|
+
self.prefix, self.matches = prefix, matches
|
|
40
|
+
super().__init__(f"{prefix!r} matches {len(matches)} effects")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
#: Enough of a `tool_call_id` to show a human.
|
|
44
|
+
#:
|
|
45
|
+
#: The id is minted by the *model*, and Gemini smuggles a **thought signature**
|
|
46
|
+
#: through it (`research/phase-10-3` V4). A real one from a live run is over
|
|
47
|
+
#: 300 characters of base64. The ledger keys on that id, so `agentctl resolve`
|
|
48
|
+
#: was asking a human to retype 300 characters to answer a blocked effect --
|
|
49
|
+
#: which makes the human half of fail-closed unusable at exactly the moment it
|
|
50
|
+
#: is needed (`docs/0013` §3, Panel 3).
|
|
51
|
+
#:
|
|
52
|
+
#: So ids print short and are accepted by prefix, the way git takes a SHA. The
|
|
53
|
+
#: distinguishing part is early (`call_2524396__thought__...`), and an
|
|
54
|
+
#: ambiguous prefix is reported rather than guessed at.
|
|
55
|
+
SHORT_ID = 12
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def short_id(tool_call_id: str) -> str:
|
|
59
|
+
"""The displayable form, marked when it is not the whole id."""
|
|
60
|
+
if len(tool_call_id) <= SHORT_ID:
|
|
61
|
+
return tool_call_id
|
|
62
|
+
return tool_call_id[:SHORT_ID] + "..."
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
#: Opening a ledger right after its previous holder was KILLED can fail on
|
|
66
|
+
#: Windows with "disk I/O error": the dead process's locks on the WAL's -shm
|
|
67
|
+
#: file are released asynchronously, and SQLite's recovery trips over them.
|
|
68
|
+
#: Measured (`docs/0046`): 5 of 8 kill-then-open trials failed on the first
|
|
69
|
+
#: attempt and every one succeeded ~50 ms later. That is the crash -> resume
|
|
70
|
+
#: path, so it is retried, briefly, and never for any other error.
|
|
71
|
+
_TRANSIENT = ("disk i/o error", "database is locked")
|
|
72
|
+
_OPEN_DEADLINE_S = 3.0
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _open(path: Path) -> sqlite3.Connection:
|
|
76
|
+
import logging
|
|
77
|
+
|
|
78
|
+
deadline, delay, attempt = time.monotonic() + _OPEN_DEADLINE_S, 0.02, 0
|
|
79
|
+
while True:
|
|
80
|
+
attempt += 1
|
|
81
|
+
db = sqlite3.connect(str(path), isolation_level=None)
|
|
82
|
+
try:
|
|
83
|
+
db.row_factory = sqlite3.Row
|
|
84
|
+
db.executescript(SCHEMA)
|
|
85
|
+
if attempt > 1:
|
|
86
|
+
logging.getLogger("agentctl.ledger").info(
|
|
87
|
+
"opened %s on attempt %d", path, attempt)
|
|
88
|
+
return db
|
|
89
|
+
except sqlite3.OperationalError as e:
|
|
90
|
+
db.close()
|
|
91
|
+
if (not any(t in str(e).lower() for t in _TRANSIENT)
|
|
92
|
+
or time.monotonic() + delay > deadline):
|
|
93
|
+
raise
|
|
94
|
+
time.sleep(delay)
|
|
95
|
+
delay = min(delay * 2, 0.5)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class LedgerStore:
|
|
99
|
+
"""SQLite-backed effect ledger.
|
|
100
|
+
|
|
101
|
+
Not thread-safe by design: one conversation, one writer, one lease.
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
def __init__(self, path: str | Path, holder: str = "local"):
|
|
105
|
+
self.path = Path(path)
|
|
106
|
+
self.holder = holder
|
|
107
|
+
self.path.parent.mkdir(parents=True, exist_ok=True)
|
|
108
|
+
self._db = _open(self.path)
|
|
109
|
+
self._fences: dict[str, int] = {} # conversation_id -> our fence token
|
|
110
|
+
|
|
111
|
+
def close(self) -> None:
|
|
112
|
+
self._db.close()
|
|
113
|
+
|
|
114
|
+
def __enter__(self) -> "LedgerStore":
|
|
115
|
+
return self
|
|
116
|
+
|
|
117
|
+
def __exit__(self, *_) -> None:
|
|
118
|
+
self.close()
|
|
119
|
+
|
|
120
|
+
# ── lease / fencing ────────────────────────────────────────────────
|
|
121
|
+
def acquire(
|
|
122
|
+
self,
|
|
123
|
+
conversation_id: str,
|
|
124
|
+
ttl_s: float = 60.0,
|
|
125
|
+
takeover: bool = False,
|
|
126
|
+
) -> int:
|
|
127
|
+
"""Take the writer lease. Returns a monotonic fence token.
|
|
128
|
+
|
|
129
|
+
An expired lease is taken silently. A *live* one raises `LeaseHeld`,
|
|
130
|
+
because two processes must never act on the same pending effect.
|
|
131
|
+
|
|
132
|
+
`takeover=True` steals a live lease. That is safe only because every
|
|
133
|
+
write is fenced (see `_assert_fence`): the superseded holder — a
|
|
134
|
+
process that crashed, or hung and is about to wake up — is rejected on
|
|
135
|
+
its next write rather than corrupting the ledger.
|
|
136
|
+
|
|
137
|
+
A crashed process cannot release its own lease, so recovery needs
|
|
138
|
+
either a short TTL or an explicit takeover. Keep `ttl_s` short and
|
|
139
|
+
renew it while working.
|
|
140
|
+
"""
|
|
141
|
+
now = time.time()
|
|
142
|
+
row = self._db.execute(
|
|
143
|
+
"SELECT holder, fence_token, expires_at FROM lease WHERE conversation_id=?",
|
|
144
|
+
(conversation_id,),
|
|
145
|
+
).fetchone()
|
|
146
|
+
|
|
147
|
+
if (row and row["expires_at"] > now and row["holder"] != self.holder
|
|
148
|
+
and not takeover):
|
|
149
|
+
raise LeaseHeld(
|
|
150
|
+
f"conversation {conversation_id} held by {row['holder']} "
|
|
151
|
+
f"for another {row['expires_at'] - now:.1f}s "
|
|
152
|
+
f"(pass takeover=True to steal it; fencing makes that safe)"
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
token = (row["fence_token"] + 1) if row else 1
|
|
156
|
+
self._db.execute(
|
|
157
|
+
"INSERT INTO lease(conversation_id, holder, fence_token, expires_at) "
|
|
158
|
+
"VALUES(?,?,?,?) ON CONFLICT(conversation_id) DO UPDATE SET "
|
|
159
|
+
"holder=excluded.holder, fence_token=excluded.fence_token, "
|
|
160
|
+
"expires_at=excluded.expires_at",
|
|
161
|
+
(conversation_id, self.holder, token, now + ttl_s),
|
|
162
|
+
)
|
|
163
|
+
self._fences[conversation_id] = token
|
|
164
|
+
return token
|
|
165
|
+
|
|
166
|
+
def renew(self, conversation_id: str, ttl_s: float = 60.0) -> None:
|
|
167
|
+
"""Extend our lease. Call periodically while working."""
|
|
168
|
+
self._db.execute(
|
|
169
|
+
"UPDATE lease SET expires_at=? WHERE conversation_id=? AND holder=?",
|
|
170
|
+
(time.time() + ttl_s, conversation_id, self.holder),
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
def release(self, conversation_id: str) -> None:
|
|
174
|
+
self._db.execute(
|
|
175
|
+
"UPDATE lease SET expires_at=0 WHERE conversation_id=? AND holder=?",
|
|
176
|
+
(conversation_id, self.holder),
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
def lease(self, conversation_id: str) -> dict | None:
|
|
180
|
+
"""The lease as it stands: holder, fence_token, expires_at."""
|
|
181
|
+
row = self._db.execute(
|
|
182
|
+
"SELECT holder, fence_token, expires_at FROM lease "
|
|
183
|
+
"WHERE conversation_id=?", (conversation_id,)).fetchone()
|
|
184
|
+
return dict(row) if row else None
|
|
185
|
+
|
|
186
|
+
# ── reads ──────────────────────────────────────────────────────────
|
|
187
|
+
def lookup(self, tool_call_id: str) -> EffectRecord | None:
|
|
188
|
+
row = self._db.execute(
|
|
189
|
+
"SELECT * FROM effect_record WHERE tool_call_id=?", (tool_call_id,)
|
|
190
|
+
).fetchone()
|
|
191
|
+
return _to_record(row) if row else None
|
|
192
|
+
|
|
193
|
+
def resolve_id(self, prefix: str) -> str | None:
|
|
194
|
+
"""An exact id, or the one effect whose id starts with `prefix`.
|
|
195
|
+
|
|
196
|
+
Exact match wins outright — a full id must never be re-interpreted as
|
|
197
|
+
a prefix, or a short id that happens to prefix a longer one would
|
|
198
|
+
resolve to the wrong effect.
|
|
199
|
+
|
|
200
|
+
Raises `AmbiguousPrefix` rather than picking. This is the command that
|
|
201
|
+
decides whether an effect is recorded as having happened; guessing
|
|
202
|
+
between two candidates is the one thing it must not do.
|
|
203
|
+
"""
|
|
204
|
+
if self.lookup(prefix) is not None:
|
|
205
|
+
return prefix
|
|
206
|
+
rows = self._db.execute(
|
|
207
|
+
"SELECT tool_call_id FROM effect_record WHERE tool_call_id LIKE ? "
|
|
208
|
+
"ESCAPE '\\' ORDER BY started_at",
|
|
209
|
+
(prefix.replace("\\", "\\\\").replace("%", "\\%")
|
|
210
|
+
.replace("_", "\\_") + "%",),
|
|
211
|
+
).fetchall()
|
|
212
|
+
ids = [r["tool_call_id"] for r in rows]
|
|
213
|
+
if not ids:
|
|
214
|
+
return None
|
|
215
|
+
if len(ids) > 1:
|
|
216
|
+
raise AmbiguousPrefix(prefix, ids)
|
|
217
|
+
return ids[0]
|
|
218
|
+
|
|
219
|
+
def find_by_intent(self, conversation_id: str, intent_hash: str,
|
|
220
|
+
exclude_tool_call_id: str | None = None) -> EffectRecord | None:
|
|
221
|
+
"""The same logical effect under a different `tool_call_id`.
|
|
222
|
+
|
|
223
|
+
WHY THIS EXISTS (`docs/0023` §4). The ledger is keyed on
|
|
224
|
+
`tool_call_id`, which the *model* mints. Ask a different model the same
|
|
225
|
+
question -- as any multi-model pool will on a retry or resume -- and it
|
|
226
|
+
returns a different id for the identical call. Keyed lookup misses, the
|
|
227
|
+
gate sees a first sighting, and the effect repeats.
|
|
228
|
+
|
|
229
|
+
`intent_hash` is ours: tool name plus canonicalised arguments. Same
|
|
230
|
+
logical effect, same hash, whoever asked.
|
|
231
|
+
|
|
232
|
+
Only the MOST RECENT identical call is a candidate, and not when it is
|
|
233
|
+
OBSERVED (`docs/0045`). Aliasing exists for a call the model re-mints
|
|
234
|
+
because it never saw the first one's result: a crash before the
|
|
235
|
+
observation was persisted. Once the result is in the history the model
|
|
236
|
+
reads -- success or reported failure -- an identical call is the model
|
|
237
|
+
choosing to repeat it, which is the ordinary edit -> test -> re-test
|
|
238
|
+
loop. Matching that as "the same effect" handed back stale output or
|
|
239
|
+
blocked a fixed retry, in the first two tasks a new user ran
|
|
240
|
+
(`docs/0044` N15). The crash cases are untouched: an INTENT twin, or a
|
|
241
|
+
COMMITTED one whose observation was never delivered, still matches.
|
|
242
|
+
"""
|
|
243
|
+
rows = self._db.execute(
|
|
244
|
+
"SELECT * FROM effect_record WHERE conversation_id=? AND intent_hash=? "
|
|
245
|
+
"ORDER BY started_at DESC", (conversation_id, intent_hash)).fetchall()
|
|
246
|
+
for r in rows:
|
|
247
|
+
if exclude_tool_call_id and r["tool_call_id"] == exclude_tool_call_id:
|
|
248
|
+
continue
|
|
249
|
+
if r["state"] == EffectState.OBSERVED.value:
|
|
250
|
+
return None
|
|
251
|
+
return _to_record(r)
|
|
252
|
+
return None
|
|
253
|
+
|
|
254
|
+
def pending(self, conversation_id: str) -> list[EffectRecord]:
|
|
255
|
+
"""Records stuck in INTENT — the ambiguous set after a crash."""
|
|
256
|
+
rows = self._db.execute(
|
|
257
|
+
"SELECT * FROM effect_record WHERE conversation_id=? AND state=? "
|
|
258
|
+
"ORDER BY started_at",
|
|
259
|
+
(conversation_id, EffectState.INTENT.value),
|
|
260
|
+
).fetchall()
|
|
261
|
+
return [_to_record(r) for r in rows]
|
|
262
|
+
|
|
263
|
+
def committed_since(self, conversation_id: str, since: float,
|
|
264
|
+
exclude_tool_call_id: str) -> list[EffectRecord]:
|
|
265
|
+
"""Effects in this conversation that landed at or after `since`.
|
|
266
|
+
|
|
267
|
+
WHY THIS EXISTS (`docs/0038` §3). A probe that reasons from world state
|
|
268
|
+
— HEAD moved, therefore my commit landed — is only sound while this
|
|
269
|
+
call is the sole writer. A sibling call that executed after this one
|
|
270
|
+
was fingerprinted breaks that, and the probe cannot see siblings: it is
|
|
271
|
+
handed one record and asked about the world.
|
|
272
|
+
|
|
273
|
+
The gate can see them. This is the query that lets it check the
|
|
274
|
+
assumption before trusting a verdict built on it.
|
|
275
|
+
|
|
276
|
+
The comparison is against `committed_at`, not `started_at`: what
|
|
277
|
+
threatens a fingerprint is a sibling whose effect *landed* after it was
|
|
278
|
+
taken, not one that was merely decided earlier in the same batch. A
|
|
279
|
+
sibling that committed *before* this call was fingerprinted is already
|
|
280
|
+
part of the world that fingerprint describes, and is no threat at all.
|
|
281
|
+
|
|
282
|
+
`>=` rather than `>`: two records can share a timestamp, and counting a
|
|
283
|
+
simultaneous sibling as concurrent costs a human decision, while
|
|
284
|
+
missing one costs a dropped effect.
|
|
285
|
+
|
|
286
|
+
OBSERVED counts as landed here. It is where a delivered effect now
|
|
287
|
+
ends up, so leaving it out would hide every ordinary sibling from the
|
|
288
|
+
sole-writer check (`docs/0045`). A delivered *reported failure* is
|
|
289
|
+
OBSERVED too and counts as well: whether it touched the world is
|
|
290
|
+
unknown, and unknown is treated as "it might have".
|
|
291
|
+
"""
|
|
292
|
+
rows = self._db.execute(
|
|
293
|
+
"SELECT * FROM effect_record WHERE conversation_id=? "
|
|
294
|
+
"AND state IN (?, ?) "
|
|
295
|
+
"AND COALESCE(committed_at, started_at)>=? AND tool_call_id!=? "
|
|
296
|
+
"ORDER BY committed_at",
|
|
297
|
+
(conversation_id, EffectState.COMMITTED.value,
|
|
298
|
+
EffectState.OBSERVED.value, since, exclude_tool_call_id),
|
|
299
|
+
).fetchall()
|
|
300
|
+
return [_to_record(r) for r in rows]
|
|
301
|
+
|
|
302
|
+
def blocked(self, conversation_id: str | None = None) -> list[EffectRecord]:
|
|
303
|
+
"""Effects awaiting a human decision (`docs/0013` §3, Panel 3)."""
|
|
304
|
+
if conversation_id:
|
|
305
|
+
rows = self._db.execute(
|
|
306
|
+
"SELECT * FROM effect_record WHERE state=? AND conversation_id=?",
|
|
307
|
+
(EffectState.BLOCKED.value, conversation_id),
|
|
308
|
+
).fetchall()
|
|
309
|
+
else:
|
|
310
|
+
rows = self._db.execute(
|
|
311
|
+
"SELECT * FROM effect_record WHERE state=?",
|
|
312
|
+
(EffectState.BLOCKED.value,),
|
|
313
|
+
).fetchall()
|
|
314
|
+
return [_to_record(r) for r in rows]
|
|
315
|
+
|
|
316
|
+
# ── the write-ahead protocol ───────────────────────────────────────
|
|
317
|
+
def _assert_fence(self, conversation_id: str, record_fence: int | None) -> None:
|
|
318
|
+
"""Reject a write from a superseded holder.
|
|
319
|
+
|
|
320
|
+
This is what makes lease takeover safe: a zombie process that wakes up
|
|
321
|
+
after its lease was stolen holds an older token and is refused here.
|
|
322
|
+
"""
|
|
323
|
+
ours = self._fences.get(conversation_id)
|
|
324
|
+
if ours is None or record_fence is None:
|
|
325
|
+
return # no lease taken; single-process use
|
|
326
|
+
if record_fence > ours:
|
|
327
|
+
raise StaleFence(
|
|
328
|
+
f"fence {ours} superseded by {record_fence} for {conversation_id}"
|
|
329
|
+
)
|
|
330
|
+
|
|
331
|
+
def _assert_current_lease(self, conversation_id: str) -> None:
|
|
332
|
+
"""Refuse to START an effect if another holder has taken the lease."""
|
|
333
|
+
ours = self._fences.get(conversation_id)
|
|
334
|
+
if ours is None:
|
|
335
|
+
return # no lease taken; single-process use
|
|
336
|
+
row = self.lease(conversation_id)
|
|
337
|
+
if row and row["fence_token"] > ours:
|
|
338
|
+
raise StaleFence(
|
|
339
|
+
f"fence {ours} superseded by {row['fence_token']} "
|
|
340
|
+
f"({row['holder']}) for {conversation_id}")
|
|
341
|
+
|
|
342
|
+
def write_intent(
|
|
343
|
+
self, call: ToolCall, cls: EffectClass, fence: int,
|
|
344
|
+
pre_state: str | None = None,
|
|
345
|
+
) -> EffectRecord:
|
|
346
|
+
"""Record intent *before* the tool runs. Durable on return.
|
|
347
|
+
|
|
348
|
+
Two fence checks, because they catch different zombies. The record's
|
|
349
|
+
own fence catches a superseded holder touching an effect the new one
|
|
350
|
+
has re-claimed. The LEASE's fence catches the one that check cannot
|
|
351
|
+
see: a superseded holder starting a *new* effect, under an id nobody
|
|
352
|
+
has recorded yet. Without it a zombie's fresh call went straight
|
|
353
|
+
through and committed (`docs/0042` I-02, probe P2).
|
|
354
|
+
"""
|
|
355
|
+
self._assert_current_lease(call.conversation_id)
|
|
356
|
+
prev = self.lookup(call.tool_call_id)
|
|
357
|
+
if prev is not None:
|
|
358
|
+
self._assert_fence(call.conversation_id, prev.fence_token)
|
|
359
|
+
_check(prev.state if prev else None, EffectState.INTENT)
|
|
360
|
+
attempt = (prev.attempt + 1) if prev else 1
|
|
361
|
+
now = time.time()
|
|
362
|
+
|
|
363
|
+
self._db.execute(
|
|
364
|
+
"INSERT INTO effect_record(tool_call_id, conversation_id, turn_id, "
|
|
365
|
+
"action_event_id, tool_name, intent_hash, effect_class, state, "
|
|
366
|
+
"fence_token, attempt, started_at, pre_state) "
|
|
367
|
+
"VALUES(?,?,?,?,?,?,?,?,?,?,?,?) "
|
|
368
|
+
"ON CONFLICT(tool_call_id) DO UPDATE SET state=excluded.state, "
|
|
369
|
+
"attempt=excluded.attempt, started_at=excluded.started_at, "
|
|
370
|
+
"fence_token=excluded.fence_token, error=NULL, "
|
|
371
|
+
"pre_state=COALESCE(excluded.pre_state, effect_record.pre_state)",
|
|
372
|
+
(call.tool_call_id, call.conversation_id, call.turn_id, None,
|
|
373
|
+
call.tool_name, call.intent_hash(), cls.value,
|
|
374
|
+
EffectState.INTENT.value, fence, attempt, now, pre_state),
|
|
375
|
+
)
|
|
376
|
+
rec = self.lookup(call.tool_call_id)
|
|
377
|
+
assert rec is not None
|
|
378
|
+
return rec
|
|
379
|
+
|
|
380
|
+
def refresh_pre_state(self, tool_call_id: str, pre_state: str | None) -> None:
|
|
381
|
+
"""Re-fingerprint the world immediately before this call executes.
|
|
382
|
+
|
|
383
|
+
WHY THIS EXISTS (`docs/0038` §3). The gate decides at Seam B, which the
|
|
384
|
+
harness runs for *every* call in an assistant message before it
|
|
385
|
+
executes *any* of them. So a batch of calls all carry a fingerprint of
|
|
386
|
+
the world as it was before the first one ran — and the git probe's
|
|
387
|
+
LANDED rule ("HEAD moved, and there is a single writer, so it was our
|
|
388
|
+
commit") is then wrong about every call but the first.
|
|
389
|
+
|
|
390
|
+
Seam C is the only place that sees a call at its own execution moment.
|
|
391
|
+
Refreshing here makes each fingerprint describe the world that call
|
|
392
|
+
actually acts on.
|
|
393
|
+
|
|
394
|
+
Deliberately narrow: `pre_state` and `started_at` only. The state
|
|
395
|
+
machine is untouched and `attempt` is not incremented, because this is
|
|
396
|
+
not a new attempt — it is the same attempt, finally about to run.
|
|
397
|
+
A record that has left INTENT is not refreshed: it has already
|
|
398
|
+
executed, and its fingerprint is evidence rather than a prediction.
|
|
399
|
+
"""
|
|
400
|
+
if pre_state is None:
|
|
401
|
+
return
|
|
402
|
+
self._db.execute(
|
|
403
|
+
"UPDATE effect_record SET pre_state=?, started_at=? "
|
|
404
|
+
"WHERE tool_call_id=? AND state=?",
|
|
405
|
+
(pre_state, time.time(), tool_call_id, EffectState.INTENT.value),
|
|
406
|
+
)
|
|
407
|
+
|
|
408
|
+
def commit(self, tool_call_id: str, observation: bytes | None = None) -> None:
|
|
409
|
+
"""The effect landed. Durable on return."""
|
|
410
|
+
self._transition(
|
|
411
|
+
tool_call_id, EffectState.COMMITTED,
|
|
412
|
+
extra={"committed_at": time.time(), "observation": observation},
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
def fail(self, tool_call_id: str, error: str) -> None:
|
|
416
|
+
"""The effect provably did not land."""
|
|
417
|
+
self._transition(tool_call_id, EffectState.FAILED, extra={"error": error[:2000]})
|
|
418
|
+
|
|
419
|
+
def block(self, tool_call_id: str, reason: str) -> None:
|
|
420
|
+
"""Fail closed. A human must resolve this."""
|
|
421
|
+
self._transition(tool_call_id, EffectState.BLOCKED, extra={"error": reason[:2000]})
|
|
422
|
+
|
|
423
|
+
def observed(self, tool_call_id: str) -> None:
|
|
424
|
+
"""A COMMITTED effect's recorded result has now reached the model."""
|
|
425
|
+
self._transition(tool_call_id, EffectState.OBSERVED)
|
|
426
|
+
|
|
427
|
+
def deliver(self, tool_call_id: str, observation: bytes | None = None,
|
|
428
|
+
error: str | None = None) -> None:
|
|
429
|
+
"""INTENT -> OBSERVED in ONE write: the tool returned, and its result
|
|
430
|
+
is already persisted in the history the model reads. Durable on return.
|
|
431
|
+
|
|
432
|
+
`error` is set when the tool reported failure. Whether such an effect
|
|
433
|
+
touched the world is unknown, and the record says so; what IS known is
|
|
434
|
+
that the model saw the failure, so a repeat is its decision.
|
|
435
|
+
"""
|
|
436
|
+
self._transition(tool_call_id, EffectState.OBSERVED, extra={
|
|
437
|
+
"committed_at": time.time(), "observation": observation,
|
|
438
|
+
"error": error[:2000] if error else None,
|
|
439
|
+
})
|
|
440
|
+
|
|
441
|
+
# ── approvals (`docs/0042` I-06) ──────────────────────────────────────
|
|
442
|
+
def decide(self, rec: EffectRecord, decision: str, summary: str = "") -> None:
|
|
443
|
+
"""A human's answer to an action that asked first.
|
|
444
|
+
|
|
445
|
+
The record leaves BLOCKED for FAILED -- "did not run" is the truth for
|
|
446
|
+
an action that was queued rather than executed -- and the answer is
|
|
447
|
+
kept by intent hash, so the identical call on resume is recognised.
|
|
448
|
+
"""
|
|
449
|
+
if decision not in ("approve", "deny"):
|
|
450
|
+
raise ValueError(decision)
|
|
451
|
+
self._db.execute(
|
|
452
|
+
"INSERT INTO approval(conversation_id, intent_hash, tool_call_id, "
|
|
453
|
+
"summary, decision, decided_at) VALUES(?,?,?,?,?,?) "
|
|
454
|
+
"ON CONFLICT(conversation_id, intent_hash) DO UPDATE SET "
|
|
455
|
+
"decision=excluded.decision, decided_at=excluded.decided_at, "
|
|
456
|
+
"summary=excluded.summary, told_at=NULL, used_at=NULL",
|
|
457
|
+
(rec.conversation_id, rec.intent_hash, rec.tool_call_id, summary,
|
|
458
|
+
decision, time.time()))
|
|
459
|
+
self.fail(rec.tool_call_id,
|
|
460
|
+
"approved by the operator; not yet run" if decision == "approve"
|
|
461
|
+
else "denied by the operator; not run")
|
|
462
|
+
|
|
463
|
+
def approval(self, conversation_id: str, intent_hash: str) -> str | None:
|
|
464
|
+
"""'approve' | 'deny' for an unused decision on this action, or None."""
|
|
465
|
+
row = self._db.execute(
|
|
466
|
+
"SELECT decision FROM approval WHERE conversation_id=? AND "
|
|
467
|
+
"intent_hash=? AND used_at IS NULL", (conversation_id, intent_hash)
|
|
468
|
+
).fetchone()
|
|
469
|
+
return row["decision"] if row else None
|
|
470
|
+
|
|
471
|
+
def use_approval(self, conversation_id: str, intent_hash: str) -> None:
|
|
472
|
+
self._db.execute("UPDATE approval SET used_at=? WHERE conversation_id=? "
|
|
473
|
+
"AND intent_hash=?", (time.time(), conversation_id,
|
|
474
|
+
intent_hash))
|
|
475
|
+
|
|
476
|
+
def untold(self, conversation_id: str) -> list[dict]:
|
|
477
|
+
"""Decisions the agent has not been told about yet, oldest first."""
|
|
478
|
+
rows = self._db.execute(
|
|
479
|
+
"SELECT * FROM approval WHERE conversation_id=? AND told_at IS NULL "
|
|
480
|
+
"ORDER BY decided_at", (conversation_id,)).fetchall()
|
|
481
|
+
return [dict(r) for r in rows]
|
|
482
|
+
|
|
483
|
+
def told(self, conversation_id: str) -> None:
|
|
484
|
+
self._db.execute("UPDATE approval SET told_at=? WHERE conversation_id=? "
|
|
485
|
+
"AND told_at IS NULL", (time.time(), conversation_id))
|
|
486
|
+
|
|
487
|
+
def reconcile(self, tool_call_id: str, verdict: str, landed: bool) -> None:
|
|
488
|
+
"""Resolve an ambiguous INTENT using out-of-band evidence."""
|
|
489
|
+
target = EffectState.COMMITTED if landed else EffectState.FAILED
|
|
490
|
+
self._transition(tool_call_id, target, extra={
|
|
491
|
+
"probe_verdict": verdict,
|
|
492
|
+
**({"committed_at": time.time()} if landed else {}),
|
|
493
|
+
})
|
|
494
|
+
|
|
495
|
+
# ── internals ──────────────────────────────────────────────────────
|
|
496
|
+
def _transition(self, tool_call_id: str, target: EffectState, extra: dict | None = None) -> None:
|
|
497
|
+
rec = self.lookup(tool_call_id)
|
|
498
|
+
if rec is None:
|
|
499
|
+
raise IllegalTransition(f"no record for {tool_call_id}")
|
|
500
|
+
self._assert_fence(rec.conversation_id, rec.fence_token)
|
|
501
|
+
_check(rec.state, target)
|
|
502
|
+
|
|
503
|
+
cols, vals = ["state=?"], [target.value]
|
|
504
|
+
for k, v in (extra or {}).items():
|
|
505
|
+
cols.append(f"{k}=?")
|
|
506
|
+
vals.append(v)
|
|
507
|
+
vals.append(tool_call_id)
|
|
508
|
+
self._db.execute(
|
|
509
|
+
f"UPDATE effect_record SET {', '.join(cols)} WHERE tool_call_id=?", vals
|
|
510
|
+
)
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
class LeaseHeartbeat:
|
|
514
|
+
"""Keep a lease live for as long as the process holding it is.
|
|
515
|
+
|
|
516
|
+
`renew()` had no callers, and the TTL is 60 s (`docs/0042` I-02). Any tool
|
|
517
|
+
call longer than that -- a test suite, a build -- let the lease lapse, and
|
|
518
|
+
the next `acquire` from another terminal took it silently.
|
|
519
|
+
|
|
520
|
+
A thread rather than a renew-per-decision: nothing decides while a long
|
|
521
|
+
tool runs, which is exactly when the lease runs out. It holds its own
|
|
522
|
+
connection, because `LedgerStore` is single-threaded by design.
|
|
523
|
+
|
|
524
|
+
A crash stops the heartbeat with the process, so the lease expires within
|
|
525
|
+
one TTL -- which is what lets a resume recover without being told to. It
|
|
526
|
+
renews only while the holder still matches: after a takeover it renews
|
|
527
|
+
nothing, so it cannot resurrect a lease it lost.
|
|
528
|
+
"""
|
|
529
|
+
|
|
530
|
+
def __init__(self, path: str | Path, holder: str, conversation_id: str,
|
|
531
|
+
ttl_s: float = 60.0, interval_s: float | None = None):
|
|
532
|
+
import threading
|
|
533
|
+
|
|
534
|
+
self._args = (Path(path), holder, conversation_id, ttl_s)
|
|
535
|
+
self._interval = interval_s if interval_s is not None else ttl_s / 4
|
|
536
|
+
self._stop = threading.Event()
|
|
537
|
+
self._thread = threading.Thread(target=self._run, daemon=True,
|
|
538
|
+
name=f"lease-heartbeat-{conversation_id}")
|
|
539
|
+
|
|
540
|
+
def start(self) -> "LeaseHeartbeat":
|
|
541
|
+
self._thread.start()
|
|
542
|
+
return self
|
|
543
|
+
|
|
544
|
+
def stop(self) -> None:
|
|
545
|
+
self._stop.set()
|
|
546
|
+
if self._thread.is_alive():
|
|
547
|
+
self._thread.join(timeout=self._interval + 5)
|
|
548
|
+
|
|
549
|
+
def _run(self) -> None:
|
|
550
|
+
import logging
|
|
551
|
+
|
|
552
|
+
path, holder, cid, ttl = self._args
|
|
553
|
+
try:
|
|
554
|
+
store = LedgerStore(path, holder=holder)
|
|
555
|
+
except Exception: # noqa: BLE001
|
|
556
|
+
logging.getLogger("agentctl.lease").exception("heartbeat could not open %s", path)
|
|
557
|
+
return
|
|
558
|
+
try:
|
|
559
|
+
while not self._stop.wait(self._interval):
|
|
560
|
+
try:
|
|
561
|
+
store.renew(cid, ttl_s=ttl)
|
|
562
|
+
except Exception: # noqa: BLE001
|
|
563
|
+
# A missed beat costs at most a lapsed lease, which the
|
|
564
|
+
# fence turns into a refused write -- never a duplicate.
|
|
565
|
+
logging.getLogger("agentctl.lease").exception("renew failed")
|
|
566
|
+
finally:
|
|
567
|
+
store.close()
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def _check(current: EffectState | None, target: EffectState) -> None:
|
|
571
|
+
allowed = TRANSITIONS.get(current, set())
|
|
572
|
+
if target not in allowed:
|
|
573
|
+
raise IllegalTransition(
|
|
574
|
+
f"{current.value if current else 'None'} -> {target.value} is not legal"
|
|
575
|
+
)
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def _to_record(row: sqlite3.Row) -> EffectRecord:
|
|
579
|
+
return EffectRecord(
|
|
580
|
+
tool_call_id=row["tool_call_id"],
|
|
581
|
+
conversation_id=row["conversation_id"],
|
|
582
|
+
turn_id=row["turn_id"],
|
|
583
|
+
tool_name=row["tool_name"],
|
|
584
|
+
intent_hash=row["intent_hash"],
|
|
585
|
+
effect_class=EffectClass(row["effect_class"]),
|
|
586
|
+
state=EffectState(row["state"]),
|
|
587
|
+
fence_token=row["fence_token"],
|
|
588
|
+
attempt=row["attempt"],
|
|
589
|
+
started_at=row["started_at"],
|
|
590
|
+
committed_at=row["committed_at"],
|
|
591
|
+
observation=row["observation"],
|
|
592
|
+
probe_verdict=row["probe_verdict"],
|
|
593
|
+
pre_state=row["pre_state"],
|
|
594
|
+
error=row["error"],
|
|
595
|
+
action_event_id=row["action_event_id"],
|
|
596
|
+
)
|