handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,596 @@
1
+ """Durable effect ledger. Local, single-writer, fsynced.
2
+
3
+ Spec: `docs/0012` §2.1, §3.1.
4
+
5
+ The one rule that matters: **a write returns only when it is durable.**
6
+ `synchronous = FULL` is deliberate — `NORMAL` can lose the last commit on power
7
+ failure, and that is precisely the record whose absence causes a double effect.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import sqlite3
12
+ import time
13
+ from pathlib import Path
14
+
15
+ from .models import (
16
+ TRANSITIONS,
17
+ EffectClass,
18
+ EffectRecord,
19
+ EffectState,
20
+ IllegalTransition,
21
+ ToolCall,
22
+ )
23
+
24
+ SCHEMA = (Path(__file__).parent / "schema.sql").read_text(encoding="utf-8")
25
+
26
+
27
+ class LeaseHeld(RuntimeError):
28
+ """Another live holder owns this conversation. Refuse to act."""
29
+
30
+
31
+ class StaleFence(RuntimeError):
32
+ """A superseded holder tried to write. Its lease was taken over."""
33
+
34
+
35
+ class AmbiguousPrefix(LookupError):
36
+ """More than one effect starts with this. Carries the candidates."""
37
+
38
+ def __init__(self, prefix: str, matches: list[str]):
39
+ self.prefix, self.matches = prefix, matches
40
+ super().__init__(f"{prefix!r} matches {len(matches)} effects")
41
+
42
+
43
+ #: Enough of a `tool_call_id` to show a human.
44
+ #:
45
+ #: The id is minted by the *model*, and Gemini smuggles a **thought signature**
46
+ #: through it (`research/phase-10-3` V4). A real one from a live run is over
47
+ #: 300 characters of base64. The ledger keys on that id, so `agentctl resolve`
48
+ #: was asking a human to retype 300 characters to answer a blocked effect --
49
+ #: which makes the human half of fail-closed unusable at exactly the moment it
50
+ #: is needed (`docs/0013` §3, Panel 3).
51
+ #:
52
+ #: So ids print short and are accepted by prefix, the way git takes a SHA. The
53
+ #: distinguishing part is early (`call_2524396__thought__...`), and an
54
+ #: ambiguous prefix is reported rather than guessed at.
55
+ SHORT_ID = 12
56
+
57
+
58
+ def short_id(tool_call_id: str) -> str:
59
+ """The displayable form, marked when it is not the whole id."""
60
+ if len(tool_call_id) <= SHORT_ID:
61
+ return tool_call_id
62
+ return tool_call_id[:SHORT_ID] + "..."
63
+
64
+
65
+ #: Opening a ledger right after its previous holder was KILLED can fail on
66
+ #: Windows with "disk I/O error": the dead process's locks on the WAL's -shm
67
+ #: file are released asynchronously, and SQLite's recovery trips over them.
68
+ #: Measured (`docs/0046`): 5 of 8 kill-then-open trials failed on the first
69
+ #: attempt and every one succeeded ~50 ms later. That is the crash -> resume
70
+ #: path, so it is retried, briefly, and never for any other error.
71
+ _TRANSIENT = ("disk i/o error", "database is locked")
72
+ _OPEN_DEADLINE_S = 3.0
73
+
74
+
75
+ def _open(path: Path) -> sqlite3.Connection:
76
+ import logging
77
+
78
+ deadline, delay, attempt = time.monotonic() + _OPEN_DEADLINE_S, 0.02, 0
79
+ while True:
80
+ attempt += 1
81
+ db = sqlite3.connect(str(path), isolation_level=None)
82
+ try:
83
+ db.row_factory = sqlite3.Row
84
+ db.executescript(SCHEMA)
85
+ if attempt > 1:
86
+ logging.getLogger("agentctl.ledger").info(
87
+ "opened %s on attempt %d", path, attempt)
88
+ return db
89
+ except sqlite3.OperationalError as e:
90
+ db.close()
91
+ if (not any(t in str(e).lower() for t in _TRANSIENT)
92
+ or time.monotonic() + delay > deadline):
93
+ raise
94
+ time.sleep(delay)
95
+ delay = min(delay * 2, 0.5)
96
+
97
+
98
+ class LedgerStore:
99
+ """SQLite-backed effect ledger.
100
+
101
+ Not thread-safe by design: one conversation, one writer, one lease.
102
+ """
103
+
104
+ def __init__(self, path: str | Path, holder: str = "local"):
105
+ self.path = Path(path)
106
+ self.holder = holder
107
+ self.path.parent.mkdir(parents=True, exist_ok=True)
108
+ self._db = _open(self.path)
109
+ self._fences: dict[str, int] = {} # conversation_id -> our fence token
110
+
111
+ def close(self) -> None:
112
+ self._db.close()
113
+
114
+ def __enter__(self) -> "LedgerStore":
115
+ return self
116
+
117
+ def __exit__(self, *_) -> None:
118
+ self.close()
119
+
120
+ # ── lease / fencing ────────────────────────────────────────────────
121
+ def acquire(
122
+ self,
123
+ conversation_id: str,
124
+ ttl_s: float = 60.0,
125
+ takeover: bool = False,
126
+ ) -> int:
127
+ """Take the writer lease. Returns a monotonic fence token.
128
+
129
+ An expired lease is taken silently. A *live* one raises `LeaseHeld`,
130
+ because two processes must never act on the same pending effect.
131
+
132
+ `takeover=True` steals a live lease. That is safe only because every
133
+ write is fenced (see `_assert_fence`): the superseded holder — a
134
+ process that crashed, or hung and is about to wake up — is rejected on
135
+ its next write rather than corrupting the ledger.
136
+
137
+ A crashed process cannot release its own lease, so recovery needs
138
+ either a short TTL or an explicit takeover. Keep `ttl_s` short and
139
+ renew it while working.
140
+ """
141
+ now = time.time()
142
+ row = self._db.execute(
143
+ "SELECT holder, fence_token, expires_at FROM lease WHERE conversation_id=?",
144
+ (conversation_id,),
145
+ ).fetchone()
146
+
147
+ if (row and row["expires_at"] > now and row["holder"] != self.holder
148
+ and not takeover):
149
+ raise LeaseHeld(
150
+ f"conversation {conversation_id} held by {row['holder']} "
151
+ f"for another {row['expires_at'] - now:.1f}s "
152
+ f"(pass takeover=True to steal it; fencing makes that safe)"
153
+ )
154
+
155
+ token = (row["fence_token"] + 1) if row else 1
156
+ self._db.execute(
157
+ "INSERT INTO lease(conversation_id, holder, fence_token, expires_at) "
158
+ "VALUES(?,?,?,?) ON CONFLICT(conversation_id) DO UPDATE SET "
159
+ "holder=excluded.holder, fence_token=excluded.fence_token, "
160
+ "expires_at=excluded.expires_at",
161
+ (conversation_id, self.holder, token, now + ttl_s),
162
+ )
163
+ self._fences[conversation_id] = token
164
+ return token
165
+
166
+ def renew(self, conversation_id: str, ttl_s: float = 60.0) -> None:
167
+ """Extend our lease. Call periodically while working."""
168
+ self._db.execute(
169
+ "UPDATE lease SET expires_at=? WHERE conversation_id=? AND holder=?",
170
+ (time.time() + ttl_s, conversation_id, self.holder),
171
+ )
172
+
173
+ def release(self, conversation_id: str) -> None:
174
+ self._db.execute(
175
+ "UPDATE lease SET expires_at=0 WHERE conversation_id=? AND holder=?",
176
+ (conversation_id, self.holder),
177
+ )
178
+
179
+ def lease(self, conversation_id: str) -> dict | None:
180
+ """The lease as it stands: holder, fence_token, expires_at."""
181
+ row = self._db.execute(
182
+ "SELECT holder, fence_token, expires_at FROM lease "
183
+ "WHERE conversation_id=?", (conversation_id,)).fetchone()
184
+ return dict(row) if row else None
185
+
186
+ # ── reads ──────────────────────────────────────────────────────────
187
+ def lookup(self, tool_call_id: str) -> EffectRecord | None:
188
+ row = self._db.execute(
189
+ "SELECT * FROM effect_record WHERE tool_call_id=?", (tool_call_id,)
190
+ ).fetchone()
191
+ return _to_record(row) if row else None
192
+
193
+ def resolve_id(self, prefix: str) -> str | None:
194
+ """An exact id, or the one effect whose id starts with `prefix`.
195
+
196
+ Exact match wins outright — a full id must never be re-interpreted as
197
+ a prefix, or a short id that happens to prefix a longer one would
198
+ resolve to the wrong effect.
199
+
200
+ Raises `AmbiguousPrefix` rather than picking. This is the command that
201
+ decides whether an effect is recorded as having happened; guessing
202
+ between two candidates is the one thing it must not do.
203
+ """
204
+ if self.lookup(prefix) is not None:
205
+ return prefix
206
+ rows = self._db.execute(
207
+ "SELECT tool_call_id FROM effect_record WHERE tool_call_id LIKE ? "
208
+ "ESCAPE '\\' ORDER BY started_at",
209
+ (prefix.replace("\\", "\\\\").replace("%", "\\%")
210
+ .replace("_", "\\_") + "%",),
211
+ ).fetchall()
212
+ ids = [r["tool_call_id"] for r in rows]
213
+ if not ids:
214
+ return None
215
+ if len(ids) > 1:
216
+ raise AmbiguousPrefix(prefix, ids)
217
+ return ids[0]
218
+
219
+ def find_by_intent(self, conversation_id: str, intent_hash: str,
220
+ exclude_tool_call_id: str | None = None) -> EffectRecord | None:
221
+ """The same logical effect under a different `tool_call_id`.
222
+
223
+ WHY THIS EXISTS (`docs/0023` §4). The ledger is keyed on
224
+ `tool_call_id`, which the *model* mints. Ask a different model the same
225
+ question -- as any multi-model pool will on a retry or resume -- and it
226
+ returns a different id for the identical call. Keyed lookup misses, the
227
+ gate sees a first sighting, and the effect repeats.
228
+
229
+ `intent_hash` is ours: tool name plus canonicalised arguments. Same
230
+ logical effect, same hash, whoever asked.
231
+
232
+ Only the MOST RECENT identical call is a candidate, and not when it is
233
+ OBSERVED (`docs/0045`). Aliasing exists for a call the model re-mints
234
+ because it never saw the first one's result: a crash before the
235
+ observation was persisted. Once the result is in the history the model
236
+ reads -- success or reported failure -- an identical call is the model
237
+ choosing to repeat it, which is the ordinary edit -> test -> re-test
238
+ loop. Matching that as "the same effect" handed back stale output or
239
+ blocked a fixed retry, in the first two tasks a new user ran
240
+ (`docs/0044` N15). The crash cases are untouched: an INTENT twin, or a
241
+ COMMITTED one whose observation was never delivered, still matches.
242
+ """
243
+ rows = self._db.execute(
244
+ "SELECT * FROM effect_record WHERE conversation_id=? AND intent_hash=? "
245
+ "ORDER BY started_at DESC", (conversation_id, intent_hash)).fetchall()
246
+ for r in rows:
247
+ if exclude_tool_call_id and r["tool_call_id"] == exclude_tool_call_id:
248
+ continue
249
+ if r["state"] == EffectState.OBSERVED.value:
250
+ return None
251
+ return _to_record(r)
252
+ return None
253
+
254
+ def pending(self, conversation_id: str) -> list[EffectRecord]:
255
+ """Records stuck in INTENT — the ambiguous set after a crash."""
256
+ rows = self._db.execute(
257
+ "SELECT * FROM effect_record WHERE conversation_id=? AND state=? "
258
+ "ORDER BY started_at",
259
+ (conversation_id, EffectState.INTENT.value),
260
+ ).fetchall()
261
+ return [_to_record(r) for r in rows]
262
+
263
+ def committed_since(self, conversation_id: str, since: float,
264
+ exclude_tool_call_id: str) -> list[EffectRecord]:
265
+ """Effects in this conversation that landed at or after `since`.
266
+
267
+ WHY THIS EXISTS (`docs/0038` §3). A probe that reasons from world state
268
+ — HEAD moved, therefore my commit landed — is only sound while this
269
+ call is the sole writer. A sibling call that executed after this one
270
+ was fingerprinted breaks that, and the probe cannot see siblings: it is
271
+ handed one record and asked about the world.
272
+
273
+ The gate can see them. This is the query that lets it check the
274
+ assumption before trusting a verdict built on it.
275
+
276
+ The comparison is against `committed_at`, not `started_at`: what
277
+ threatens a fingerprint is a sibling whose effect *landed* after it was
278
+ taken, not one that was merely decided earlier in the same batch. A
279
+ sibling that committed *before* this call was fingerprinted is already
280
+ part of the world that fingerprint describes, and is no threat at all.
281
+
282
+ `>=` rather than `>`: two records can share a timestamp, and counting a
283
+ simultaneous sibling as concurrent costs a human decision, while
284
+ missing one costs a dropped effect.
285
+
286
+ OBSERVED counts as landed here. It is where a delivered effect now
287
+ ends up, so leaving it out would hide every ordinary sibling from the
288
+ sole-writer check (`docs/0045`). A delivered *reported failure* is
289
+ OBSERVED too and counts as well: whether it touched the world is
290
+ unknown, and unknown is treated as "it might have".
291
+ """
292
+ rows = self._db.execute(
293
+ "SELECT * FROM effect_record WHERE conversation_id=? "
294
+ "AND state IN (?, ?) "
295
+ "AND COALESCE(committed_at, started_at)>=? AND tool_call_id!=? "
296
+ "ORDER BY committed_at",
297
+ (conversation_id, EffectState.COMMITTED.value,
298
+ EffectState.OBSERVED.value, since, exclude_tool_call_id),
299
+ ).fetchall()
300
+ return [_to_record(r) for r in rows]
301
+
302
+ def blocked(self, conversation_id: str | None = None) -> list[EffectRecord]:
303
+ """Effects awaiting a human decision (`docs/0013` §3, Panel 3)."""
304
+ if conversation_id:
305
+ rows = self._db.execute(
306
+ "SELECT * FROM effect_record WHERE state=? AND conversation_id=?",
307
+ (EffectState.BLOCKED.value, conversation_id),
308
+ ).fetchall()
309
+ else:
310
+ rows = self._db.execute(
311
+ "SELECT * FROM effect_record WHERE state=?",
312
+ (EffectState.BLOCKED.value,),
313
+ ).fetchall()
314
+ return [_to_record(r) for r in rows]
315
+
316
+ # ── the write-ahead protocol ───────────────────────────────────────
317
+ def _assert_fence(self, conversation_id: str, record_fence: int | None) -> None:
318
+ """Reject a write from a superseded holder.
319
+
320
+ This is what makes lease takeover safe: a zombie process that wakes up
321
+ after its lease was stolen holds an older token and is refused here.
322
+ """
323
+ ours = self._fences.get(conversation_id)
324
+ if ours is None or record_fence is None:
325
+ return # no lease taken; single-process use
326
+ if record_fence > ours:
327
+ raise StaleFence(
328
+ f"fence {ours} superseded by {record_fence} for {conversation_id}"
329
+ )
330
+
331
+ def _assert_current_lease(self, conversation_id: str) -> None:
332
+ """Refuse to START an effect if another holder has taken the lease."""
333
+ ours = self._fences.get(conversation_id)
334
+ if ours is None:
335
+ return # no lease taken; single-process use
336
+ row = self.lease(conversation_id)
337
+ if row and row["fence_token"] > ours:
338
+ raise StaleFence(
339
+ f"fence {ours} superseded by {row['fence_token']} "
340
+ f"({row['holder']}) for {conversation_id}")
341
+
342
+ def write_intent(
343
+ self, call: ToolCall, cls: EffectClass, fence: int,
344
+ pre_state: str | None = None,
345
+ ) -> EffectRecord:
346
+ """Record intent *before* the tool runs. Durable on return.
347
+
348
+ Two fence checks, because they catch different zombies. The record's
349
+ own fence catches a superseded holder touching an effect the new one
350
+ has re-claimed. The LEASE's fence catches the one that check cannot
351
+ see: a superseded holder starting a *new* effect, under an id nobody
352
+ has recorded yet. Without it a zombie's fresh call went straight
353
+ through and committed (`docs/0042` I-02, probe P2).
354
+ """
355
+ self._assert_current_lease(call.conversation_id)
356
+ prev = self.lookup(call.tool_call_id)
357
+ if prev is not None:
358
+ self._assert_fence(call.conversation_id, prev.fence_token)
359
+ _check(prev.state if prev else None, EffectState.INTENT)
360
+ attempt = (prev.attempt + 1) if prev else 1
361
+ now = time.time()
362
+
363
+ self._db.execute(
364
+ "INSERT INTO effect_record(tool_call_id, conversation_id, turn_id, "
365
+ "action_event_id, tool_name, intent_hash, effect_class, state, "
366
+ "fence_token, attempt, started_at, pre_state) "
367
+ "VALUES(?,?,?,?,?,?,?,?,?,?,?,?) "
368
+ "ON CONFLICT(tool_call_id) DO UPDATE SET state=excluded.state, "
369
+ "attempt=excluded.attempt, started_at=excluded.started_at, "
370
+ "fence_token=excluded.fence_token, error=NULL, "
371
+ "pre_state=COALESCE(excluded.pre_state, effect_record.pre_state)",
372
+ (call.tool_call_id, call.conversation_id, call.turn_id, None,
373
+ call.tool_name, call.intent_hash(), cls.value,
374
+ EffectState.INTENT.value, fence, attempt, now, pre_state),
375
+ )
376
+ rec = self.lookup(call.tool_call_id)
377
+ assert rec is not None
378
+ return rec
379
+
380
+ def refresh_pre_state(self, tool_call_id: str, pre_state: str | None) -> None:
381
+ """Re-fingerprint the world immediately before this call executes.
382
+
383
+ WHY THIS EXISTS (`docs/0038` §3). The gate decides at Seam B, which the
384
+ harness runs for *every* call in an assistant message before it
385
+ executes *any* of them. So a batch of calls all carry a fingerprint of
386
+ the world as it was before the first one ran — and the git probe's
387
+ LANDED rule ("HEAD moved, and there is a single writer, so it was our
388
+ commit") is then wrong about every call but the first.
389
+
390
+ Seam C is the only place that sees a call at its own execution moment.
391
+ Refreshing here makes each fingerprint describe the world that call
392
+ actually acts on.
393
+
394
+ Deliberately narrow: `pre_state` and `started_at` only. The state
395
+ machine is untouched and `attempt` is not incremented, because this is
396
+ not a new attempt — it is the same attempt, finally about to run.
397
+ A record that has left INTENT is not refreshed: it has already
398
+ executed, and its fingerprint is evidence rather than a prediction.
399
+ """
400
+ if pre_state is None:
401
+ return
402
+ self._db.execute(
403
+ "UPDATE effect_record SET pre_state=?, started_at=? "
404
+ "WHERE tool_call_id=? AND state=?",
405
+ (pre_state, time.time(), tool_call_id, EffectState.INTENT.value),
406
+ )
407
+
408
+ def commit(self, tool_call_id: str, observation: bytes | None = None) -> None:
409
+ """The effect landed. Durable on return."""
410
+ self._transition(
411
+ tool_call_id, EffectState.COMMITTED,
412
+ extra={"committed_at": time.time(), "observation": observation},
413
+ )
414
+
415
+ def fail(self, tool_call_id: str, error: str) -> None:
416
+ """The effect provably did not land."""
417
+ self._transition(tool_call_id, EffectState.FAILED, extra={"error": error[:2000]})
418
+
419
+ def block(self, tool_call_id: str, reason: str) -> None:
420
+ """Fail closed. A human must resolve this."""
421
+ self._transition(tool_call_id, EffectState.BLOCKED, extra={"error": reason[:2000]})
422
+
423
+ def observed(self, tool_call_id: str) -> None:
424
+ """A COMMITTED effect's recorded result has now reached the model."""
425
+ self._transition(tool_call_id, EffectState.OBSERVED)
426
+
427
+ def deliver(self, tool_call_id: str, observation: bytes | None = None,
428
+ error: str | None = None) -> None:
429
+ """INTENT -> OBSERVED in ONE write: the tool returned, and its result
430
+ is already persisted in the history the model reads. Durable on return.
431
+
432
+ `error` is set when the tool reported failure. Whether such an effect
433
+ touched the world is unknown, and the record says so; what IS known is
434
+ that the model saw the failure, so a repeat is its decision.
435
+ """
436
+ self._transition(tool_call_id, EffectState.OBSERVED, extra={
437
+ "committed_at": time.time(), "observation": observation,
438
+ "error": error[:2000] if error else None,
439
+ })
440
+
441
+ # ── approvals (`docs/0042` I-06) ──────────────────────────────────────
442
+ def decide(self, rec: EffectRecord, decision: str, summary: str = "") -> None:
443
+ """A human's answer to an action that asked first.
444
+
445
+ The record leaves BLOCKED for FAILED -- "did not run" is the truth for
446
+ an action that was queued rather than executed -- and the answer is
447
+ kept by intent hash, so the identical call on resume is recognised.
448
+ """
449
+ if decision not in ("approve", "deny"):
450
+ raise ValueError(decision)
451
+ self._db.execute(
452
+ "INSERT INTO approval(conversation_id, intent_hash, tool_call_id, "
453
+ "summary, decision, decided_at) VALUES(?,?,?,?,?,?) "
454
+ "ON CONFLICT(conversation_id, intent_hash) DO UPDATE SET "
455
+ "decision=excluded.decision, decided_at=excluded.decided_at, "
456
+ "summary=excluded.summary, told_at=NULL, used_at=NULL",
457
+ (rec.conversation_id, rec.intent_hash, rec.tool_call_id, summary,
458
+ decision, time.time()))
459
+ self.fail(rec.tool_call_id,
460
+ "approved by the operator; not yet run" if decision == "approve"
461
+ else "denied by the operator; not run")
462
+
463
+ def approval(self, conversation_id: str, intent_hash: str) -> str | None:
464
+ """'approve' | 'deny' for an unused decision on this action, or None."""
465
+ row = self._db.execute(
466
+ "SELECT decision FROM approval WHERE conversation_id=? AND "
467
+ "intent_hash=? AND used_at IS NULL", (conversation_id, intent_hash)
468
+ ).fetchone()
469
+ return row["decision"] if row else None
470
+
471
+ def use_approval(self, conversation_id: str, intent_hash: str) -> None:
472
+ self._db.execute("UPDATE approval SET used_at=? WHERE conversation_id=? "
473
+ "AND intent_hash=?", (time.time(), conversation_id,
474
+ intent_hash))
475
+
476
+ def untold(self, conversation_id: str) -> list[dict]:
477
+ """Decisions the agent has not been told about yet, oldest first."""
478
+ rows = self._db.execute(
479
+ "SELECT * FROM approval WHERE conversation_id=? AND told_at IS NULL "
480
+ "ORDER BY decided_at", (conversation_id,)).fetchall()
481
+ return [dict(r) for r in rows]
482
+
483
+ def told(self, conversation_id: str) -> None:
484
+ self._db.execute("UPDATE approval SET told_at=? WHERE conversation_id=? "
485
+ "AND told_at IS NULL", (time.time(), conversation_id))
486
+
487
+ def reconcile(self, tool_call_id: str, verdict: str, landed: bool) -> None:
488
+ """Resolve an ambiguous INTENT using out-of-band evidence."""
489
+ target = EffectState.COMMITTED if landed else EffectState.FAILED
490
+ self._transition(tool_call_id, target, extra={
491
+ "probe_verdict": verdict,
492
+ **({"committed_at": time.time()} if landed else {}),
493
+ })
494
+
495
+ # ── internals ──────────────────────────────────────────────────────
496
+ def _transition(self, tool_call_id: str, target: EffectState, extra: dict | None = None) -> None:
497
+ rec = self.lookup(tool_call_id)
498
+ if rec is None:
499
+ raise IllegalTransition(f"no record for {tool_call_id}")
500
+ self._assert_fence(rec.conversation_id, rec.fence_token)
501
+ _check(rec.state, target)
502
+
503
+ cols, vals = ["state=?"], [target.value]
504
+ for k, v in (extra or {}).items():
505
+ cols.append(f"{k}=?")
506
+ vals.append(v)
507
+ vals.append(tool_call_id)
508
+ self._db.execute(
509
+ f"UPDATE effect_record SET {', '.join(cols)} WHERE tool_call_id=?", vals
510
+ )
511
+
512
+
513
+ class LeaseHeartbeat:
514
+ """Keep a lease live for as long as the process holding it is.
515
+
516
+ `renew()` had no callers, and the TTL is 60 s (`docs/0042` I-02). Any tool
517
+ call longer than that -- a test suite, a build -- let the lease lapse, and
518
+ the next `acquire` from another terminal took it silently.
519
+
520
+ A thread rather than a renew-per-decision: nothing decides while a long
521
+ tool runs, which is exactly when the lease runs out. It holds its own
522
+ connection, because `LedgerStore` is single-threaded by design.
523
+
524
+ A crash stops the heartbeat with the process, so the lease expires within
525
+ one TTL -- which is what lets a resume recover without being told to. It
526
+ renews only while the holder still matches: after a takeover it renews
527
+ nothing, so it cannot resurrect a lease it lost.
528
+ """
529
+
530
+ def __init__(self, path: str | Path, holder: str, conversation_id: str,
531
+ ttl_s: float = 60.0, interval_s: float | None = None):
532
+ import threading
533
+
534
+ self._args = (Path(path), holder, conversation_id, ttl_s)
535
+ self._interval = interval_s if interval_s is not None else ttl_s / 4
536
+ self._stop = threading.Event()
537
+ self._thread = threading.Thread(target=self._run, daemon=True,
538
+ name=f"lease-heartbeat-{conversation_id}")
539
+
540
+ def start(self) -> "LeaseHeartbeat":
541
+ self._thread.start()
542
+ return self
543
+
544
+ def stop(self) -> None:
545
+ self._stop.set()
546
+ if self._thread.is_alive():
547
+ self._thread.join(timeout=self._interval + 5)
548
+
549
+ def _run(self) -> None:
550
+ import logging
551
+
552
+ path, holder, cid, ttl = self._args
553
+ try:
554
+ store = LedgerStore(path, holder=holder)
555
+ except Exception: # noqa: BLE001
556
+ logging.getLogger("agentctl.lease").exception("heartbeat could not open %s", path)
557
+ return
558
+ try:
559
+ while not self._stop.wait(self._interval):
560
+ try:
561
+ store.renew(cid, ttl_s=ttl)
562
+ except Exception: # noqa: BLE001
563
+ # A missed beat costs at most a lapsed lease, which the
564
+ # fence turns into a refused write -- never a duplicate.
565
+ logging.getLogger("agentctl.lease").exception("renew failed")
566
+ finally:
567
+ store.close()
568
+
569
+
570
+ def _check(current: EffectState | None, target: EffectState) -> None:
571
+ allowed = TRANSITIONS.get(current, set())
572
+ if target not in allowed:
573
+ raise IllegalTransition(
574
+ f"{current.value if current else 'None'} -> {target.value} is not legal"
575
+ )
576
+
577
+
578
+ def _to_record(row: sqlite3.Row) -> EffectRecord:
579
+ return EffectRecord(
580
+ tool_call_id=row["tool_call_id"],
581
+ conversation_id=row["conversation_id"],
582
+ turn_id=row["turn_id"],
583
+ tool_name=row["tool_name"],
584
+ intent_hash=row["intent_hash"],
585
+ effect_class=EffectClass(row["effect_class"]),
586
+ state=EffectState(row["state"]),
587
+ fence_token=row["fence_token"],
588
+ attempt=row["attempt"],
589
+ started_at=row["started_at"],
590
+ committed_at=row["committed_at"],
591
+ observation=row["observation"],
592
+ probe_verdict=row["probe_verdict"],
593
+ pre_state=row["pre_state"],
594
+ error=row["error"],
595
+ action_event_id=row["action_event_id"],
596
+ )