handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,309 @@
1
+ """A LiteLLM proxy that agentctl installs, starts and stops for you.
2
+
3
+ `docs/0043` Phase 2, part B. Before this, the pool took three commands and two
4
+ flags (`proxy --out`, `start.sh` or `start.ps1`, then `--model openai/pool
5
+ --base-url ...`), plus a two-step install, because `litellm[proxy]` declares
6
+ `mcp<2` and the README said the SDK needed `mcp>=2` (`docs/0028`).
7
+
8
+ The proxy is a separate PROCESS already, so it can have a separate
9
+ ENVIRONMENT, and then the conflict cannot exist:
10
+
11
+ ~/.agentctl/proxy-env/ a venv holding litellm[proxy] and a COPY of
12
+ agentctl's pure-Python package (the hook needs
13
+ its kernel), and never the OpenHands SDK
14
+ ~/.agentctl/proxy/ config, hook, pid file, log
15
+
16
+ The user's own install never contains `litellm[proxy]`. Nothing is put on the
17
+ proxy's PYTHONPATH: the generated start scripts used to point it at the main
18
+ environment's site-packages, which would bring every package -- and the
19
+ conflict -- back in (`docs/0047`).
20
+
21
+ agentctl proxy up create the env if needed, verify, start, wait
22
+ agentctl proxy status is it running, and does it answer
23
+ agentctl proxy down stop it
24
+ agentctl run --pool `up` if needed, then route through it
25
+ """
26
+ from __future__ import annotations
27
+
28
+ import json
29
+ import os
30
+ import shutil
31
+ import subprocess
32
+ import sys
33
+ import time
34
+ import urllib.request
35
+ from pathlib import Path
36
+
37
+ LITELLM_SPEC = "litellm[proxy]>=1.100.0"
38
+ DEFAULT_PORT = 4000
39
+
40
+
41
+ def home() -> Path:
42
+ return Path(os.environ.get("AGENTCTL_HOME") or Path.home() / ".agentctl")
43
+
44
+
45
+ def env_dir() -> Path:
46
+ """`~/.agentctl/proxy-env`, or `AGENTCTL_PROXY_ENV`.
47
+
48
+ The override is for the container image (`docs/0051` Stage 2), which
49
+ builds the environment in at `/opt/handcode/proxy-env`. Under
50
+ `~/.agentctl` it would be hidden by the volume that keeps `status` and
51
+ `resume` working across containers.
52
+ """
53
+ if (p := os.environ.get("AGENTCTL_PROXY_ENV")):
54
+ return Path(p)
55
+ return home() / "proxy-env"
56
+
57
+
58
+ def run_dir() -> Path:
59
+ return home() / "proxy"
60
+
61
+
62
+ def _env_python() -> Path:
63
+ d = env_dir()
64
+ return d / ("Scripts/python.exe" if sys.platform == "win32" else "bin/python")
65
+
66
+
67
+ def _env_litellm() -> Path:
68
+ d = env_dir()
69
+ return d / ("Scripts/litellm.exe" if sys.platform == "win32" else "bin/litellm")
70
+
71
+
72
+ def _say(msg: str) -> None:
73
+ print(f" {msg}", flush=True)
74
+
75
+
76
+ # ── the environment ────────────────────────────────────────────────────
77
+ def ensure_env(log=_say) -> Path:
78
+ """Create the proxy's own venv, install litellm[proxy], copy agentctl in.
79
+
80
+ The first call downloads litellm and its proxy extras, which takes
81
+ minutes; later calls only refresh the agentctl copy, which takes well
82
+ under a second and keeps the hook in step with the CLI that started it.
83
+ """
84
+ py = _env_python()
85
+ if not py.exists():
86
+ log(f"creating the proxy environment at {env_dir()} (once)")
87
+ r = subprocess.run([sys.executable, "-m", "venv", str(env_dir())],
88
+ capture_output=True, text=True)
89
+ if r.returncode != 0:
90
+ raise SystemExit(
91
+ f"could not create a venv for the proxy:\n{r.stderr.strip()[-600:]}\n"
92
+ f" on Ubuntu: sudo apt install python3-venv")
93
+ if not _env_litellm().exists():
94
+ log(f"installing {LITELLM_SPEC} into it -- a few minutes, once")
95
+ r = subprocess.run([str(py), "-m", "pip", "install", "-q",
96
+ "--disable-pip-version-check", LITELLM_SPEC],
97
+ capture_output=True, text=True)
98
+ if r.returncode != 0 or not _env_litellm().exists():
99
+ raise SystemExit(f"installing {LITELLM_SPEC} failed:\n"
100
+ f"{(r.stdout + r.stderr).strip()[-1200:]}")
101
+ _copy_agentctl(py)
102
+ return py
103
+
104
+
105
+ def _copy_agentctl(py: Path) -> Path:
106
+ """Put this exact agentctl package into the proxy env's site-packages.
107
+
108
+ A copy, not a PYTHONPATH entry and not a pip install: the hook must be the
109
+ same code as the CLI that generated its config, it must come without the
110
+ main environment's other packages, and it must work whether agentctl was
111
+ installed from PyPI, from a wheel, or editable from a clone.
112
+ """
113
+ import agentctl
114
+
115
+ src = Path(agentctl.__file__).resolve().parent
116
+ purelib = subprocess.run(
117
+ [str(py), "-c", "import sysconfig; print(sysconfig.get_paths()['purelib'])"],
118
+ capture_output=True, text=True, check=True).stdout.strip()
119
+ dst = Path(purelib) / "agentctl"
120
+ if dst.exists():
121
+ shutil.rmtree(dst)
122
+ shutil.copytree(src, dst, ignore=shutil.ignore_patterns("__pycache__", "*.pyc"))
123
+ return dst
124
+
125
+
126
+ # ── the process ────────────────────────────────────────────────────────
127
+ def _pidfile() -> Path:
128
+ return run_dir() / "proxy.json"
129
+
130
+
131
+ def _read_pid() -> dict | None:
132
+ try:
133
+ return json.loads(_pidfile().read_text(encoding="utf-8"))
134
+ except Exception: # noqa: BLE001
135
+ return None
136
+
137
+
138
+ def url(port: int = DEFAULT_PORT) -> str:
139
+ return f"http://127.0.0.1:{port}"
140
+
141
+
142
+ def answers(port: int, timeout: float = 2.0) -> bool:
143
+ """Does a proxy on this port answer its unauthenticated liveness check?"""
144
+ try:
145
+ with urllib.request.urlopen(f"{url(port)}/health/liveliness",
146
+ timeout=timeout) as r:
147
+ return r.status == 200
148
+ except Exception: # noqa: BLE001
149
+ return False
150
+
151
+
152
+ def status() -> dict:
153
+ """{'state': 'running' | 'stopped' | 'dead' | 'foreign', ...}
154
+
155
+ `foreign` is something answering on the port that agentctl did not start
156
+ (a proxy you ran by hand, say). It is reported, never stopped.
157
+ """
158
+ from agentctl.runtime.lease import pid_alive
159
+
160
+ rec = _read_pid()
161
+ port = (rec or {}).get("port", DEFAULT_PORT)
162
+ alive = bool(rec) and pid_alive(int(rec["pid"]))
163
+ up = answers(port)
164
+ if rec and alive:
165
+ state = "running" if up else "starting"
166
+ elif rec:
167
+ state = "dead"
168
+ else:
169
+ state = "foreign" if up else "stopped"
170
+ return {"state": state, "port": port, "pid": (rec or {}).get("pid"),
171
+ "answers": up, "log": str(run_dir() / "proxy.log"),
172
+ "env": str(env_dir())}
173
+
174
+
175
+ def up(port: int = DEFAULT_PORT, verify: bool = True, log=_say,
176
+ wait_s: float = 120.0) -> dict:
177
+ """Start the managed proxy, or reuse it if it is already running."""
178
+ s = status()
179
+ if s["state"] in ("running", "starting"):
180
+ if s["port"] == port:
181
+ log(f"proxy already running (pid {s['pid']}) at {url(port)}")
182
+ return s
183
+ raise SystemExit(f"a managed proxy is running on port {s['port']}, "
184
+ f"not {port}. `agentctl proxy down` first.")
185
+ # Only the port being asked for matters. `status()` with no pid file
186
+ # probes the DEFAULT port, and refusing port N because something answers
187
+ # on 4000 was a bug a live proxy on 4000 exposed in the tests.
188
+ if answers(port):
189
+ raise SystemExit(f"something not started by agentctl answers on port "
190
+ f"{port}. Stop it, or use --port.")
191
+
192
+ ensure_env(log)
193
+
194
+ from agentctl.control.proxy import available, write
195
+
196
+ only, drop = None, set()
197
+ if verify:
198
+ from agentctl.control.providers import BY_NAME
199
+ from agentctl.control.proxy import verified_providers
200
+
201
+ log("verifying providers (one completion per model id; paid skipped) ...")
202
+ only, report = verified_providers()
203
+ for name, r in sorted(report.items()):
204
+ log(f" {'ok' if name in only else '--'} {name:<11} {r.status}")
205
+ drop = {f"{BY_NAME[n].prefix}{m}" for n, r in report.items() for m in r.gone}
206
+ for m in sorted(drop):
207
+ log(f" left out {m} (the provider no longer serves it)")
208
+ entries = [e for e in available()
209
+ if (only is None or _provider_name(e[0]) in only)
210
+ and e[1] not in drop]
211
+ if not entries:
212
+ raise SystemExit("no provider can serve right now -- nothing to pool.\n"
213
+ " `agentctl keys --check` says why.")
214
+ log(f"{len(entries)} deployment(s) in the pool")
215
+
216
+ d = run_dir()
217
+ d.mkdir(parents=True, exist_ok=True)
218
+ cfg, _hook = write(d, only=only, drop_models=drop)
219
+
220
+ env = {**os.environ, "PYTHONUTF8": "1", "PYTHONIOENCODING": "utf-8",
221
+ "AGENTCTL_TELEMETRY": str(d / "hook_telemetry.json")}
222
+ env.pop("PYTHONPATH", None) # never the main env's packages
223
+ logf = open(d / "proxy.log", "ab")
224
+ kw: dict = {"stdout": logf, "stderr": subprocess.STDOUT, "env": env,
225
+ "cwd": str(d), "stdin": subprocess.DEVNULL}
226
+ # The env's own python, not the `litellm.exe` console-script wrapper:
227
+ # launched DETACHED, the wrapper exited during startup with 0xC000013A
228
+ # (STATUS_CONTROL_C_EXIT) and an empty log (`docs/0047`).
229
+ argv = [str(_env_python()), "-c", "from litellm import run_server; run_server()",
230
+ "--config", str(cfg), "--port", str(port)]
231
+ if sys.platform == "win32":
232
+ # No window, its own group, and OUT of the launcher's job object, so
233
+ # it outlives the `agentctl` process that started it. A job that does
234
+ # not allow breakaway refuses the flag; then it runs without it.
235
+ base = subprocess.CREATE_NO_WINDOW | subprocess.CREATE_NEW_PROCESS_GROUP
236
+ try:
237
+ p = subprocess.Popen(argv, creationflags=base | 0x01000000, **kw)
238
+ except OSError:
239
+ p = subprocess.Popen(argv, creationflags=base, **kw)
240
+ else:
241
+ p = subprocess.Popen(argv, start_new_session=True, **kw)
242
+ logf.close()
243
+ _pidfile().write_text(json.dumps({"pid": p.pid, "port": port,
244
+ "started": time.time()}), encoding="utf-8")
245
+
246
+ log(f"starting litellm (pid {p.pid}) on {url(port)} ...")
247
+ deadline = time.time() + wait_s
248
+ while time.time() < deadline:
249
+ if p.poll() is not None:
250
+ _pidfile().unlink(missing_ok=True)
251
+ raise SystemExit(f"the proxy exited during startup (code {p.returncode}). "
252
+ f"Last lines of {d / 'proxy.log'}:\n"
253
+ + _tail(d / "proxy.log"))
254
+ if answers(port):
255
+ log(f"proxy up at {url(port)}")
256
+ return status()
257
+ time.sleep(1)
258
+ raise SystemExit(f"the proxy did not answer within {wait_s:.0f}s; it is "
259
+ f"still running (pid {p.pid}). See {d / 'proxy.log'}")
260
+
261
+
262
+ def _provider_name(env_var: str) -> str:
263
+ """`OPENROUTER_API_KEY_2` -> `openrouter`."""
264
+ from agentctl.control.providers import BY_KEY
265
+ return BY_KEY[env_var.split("_API_KEY")[0] + "_API_KEY"].name
266
+
267
+
268
+ def down(log=_say) -> bool:
269
+ """Stop the managed proxy. True if something was stopped."""
270
+ from agentctl.runtime.lease import pid_alive
271
+
272
+ rec = _read_pid()
273
+ if not rec:
274
+ log("no managed proxy is running")
275
+ return False
276
+ pid = int(rec["pid"])
277
+ if pid_alive(pid):
278
+ if sys.platform == "win32":
279
+ subprocess.run(["taskkill", "/PID", str(pid), "/T", "/F"],
280
+ capture_output=True)
281
+ else:
282
+ import signal
283
+ # The group only when it is the proxy's OWN. `up` starts it in a
284
+ # new session, but a pid file can name anything, and signalling a
285
+ # group we belong to kills the caller: it SIGTERMed pytest, and the
286
+ # CI step with it, on every Linux job (`docs/0047`).
287
+ try:
288
+ pgid = os.getpgid(pid)
289
+ if pgid != os.getpgid(0):
290
+ os.killpg(pgid, signal.SIGTERM)
291
+ else:
292
+ os.kill(pid, signal.SIGTERM)
293
+ except ProcessLookupError:
294
+ pass
295
+ for _ in range(20):
296
+ if not pid_alive(pid):
297
+ break
298
+ time.sleep(0.25)
299
+ _pidfile().unlink(missing_ok=True)
300
+ log(f"proxy stopped (pid {pid})")
301
+ return True
302
+
303
+
304
+ def _tail(p: Path, n: int = 15) -> str:
305
+ try:
306
+ lines = p.read_text(encoding="utf-8", errors="replace").splitlines()
307
+ return "\n".join(" " + l for l in lines[-n:])
308
+ except Exception: # noqa: BLE001
309
+ return " (no log)"
@@ -0,0 +1,14 @@
1
+ """Record and replay real LLM sessions offline, at zero cost. M6.
2
+
3
+ from agentctl.control.replay import Cassette, ReplayServer
4
+
5
+ `cassette.py` is pure and has no dependencies; `server.py` is stdlib only.
6
+ The recorder lives in `adapters/litellm/` because it is vendor-specific,
7
+ which is the same split the rest of the package uses.
8
+ """
9
+ from .cassette import (Cassette, Miss, Turn, current_env, fingerprint,
10
+ incompatible, summarise)
11
+ from .server import ReplayServer
12
+
13
+ __all__ = ["Cassette", "Miss", "Turn", "ReplayServer", "current_env",
14
+ "fingerprint", "incompatible", "summarise"]
@@ -0,0 +1,281 @@
1
+ r"""A recorded LLM session, and the rule for matching a request to a response.
2
+
3
+ The point of M6 (`docs/0012` §7): **re-run a real session offline at zero
4
+ cost.** No API key, no network, no tokens, and — the part that matters —
5
+ no sampling. The same code produces the same run every time.
6
+
7
+ ## The matching rule is the whole design
8
+
9
+ A cassette is keyed by a fingerprint of what actually determines the response:
10
+
11
+ model + messages + tool schemas
12
+
13
+ and deliberately *not* by `temperature`, `max_tokens`, `api_key`, `base_url`,
14
+ request id, or timestamp. Those either do not change the mapping or change on
15
+ every run, and folding them in would mean a cassette that never hits twice.
16
+
17
+ A **miss is the product, not a failure.** Replaying a recorded session against
18
+ changed code and getting a miss means the agent asked something different —
19
+ which is precisely the regression signal M6 exists to produce. So a miss is
20
+ reported with the turn number and a diff of what changed, rather than being
21
+ papered over.
22
+
23
+ ## What determinism hides
24
+
25
+ Replay pins `tool_call_id`, because the recorded response carries the one the
26
+ model minted. That is convenient and it is also a trap: `docs/0023` found that
27
+ `tool_call_id` is **not** stable across a real model pool, and a suite that
28
+ only ever ran under replay would never have found it. Replay is for testing
29
+ *our* logic against fixed inputs. It cannot test how the world varies.
30
+
31
+ ## Cassettes contain prompts
32
+
33
+ Every recorded request holds the full message history — source code, file
34
+ contents, whatever the agent was working on. Credentials are never recorded
35
+ (they live in headers, which this never sees), but a cassette is as sensitive
36
+ as the workspace it was recorded in. Treat it like a log, not like a fixture.
37
+ """
38
+ from __future__ import annotations
39
+
40
+ import hashlib
41
+ import json
42
+ import sys
43
+ from dataclasses import dataclass, field
44
+ from pathlib import Path
45
+ from typing import Any, Iterator
46
+
47
+ # THE list. It decides both what the recorder captures and what the
48
+ # fingerprint is built from, and it is one list on purpose.
49
+ #
50
+ # It began as two: an allow-list in the recorder and a deny-list here. They
51
+ # drifted, and a cassette recorded from a Python callback could never match the
52
+ # same request arriving as HTTP, because the wire carries fields the callback
53
+ # never saw -- `prompt_cache_key` (the conversation id, different every run by
54
+ # definition) and `usage: {include: true}`. Every replay missed on turn 0.
55
+ #
56
+ # A deny-list cannot work here. The two sides are different representations of
57
+ # one call, and any provider may add a field to the wire form at any time; a
58
+ # deny-list would have to predict them all. An allow-list is closed, and the
59
+ # cost of being closed is explicit: **a field outside this list is asserted not
60
+ # to determine the response.** Adding a sampling parameter that does would
61
+ # produce a wrong match rather than a miss, so the list is short and additions
62
+ # belong in review.
63
+ SIGNIFICANT_FIELDS = ("model", "messages", "tools", "tool_choice",
64
+ "response_format", "functions", "function_call")
65
+
66
+
67
+ def fingerprint(request: dict) -> str:
68
+ """Stable hash of the parts of a request that decide the response."""
69
+ return hashlib.sha256(canonical(request).encode("utf-8")).hexdigest()
70
+
71
+
72
+ def canonical(request: dict) -> str:
73
+ """The exact text that gets hashed. Exposed so a miss can be diffed."""
74
+ kept = {k: request[k] for k in SIGNIFICANT_FIELDS
75
+ if request.get(k) is not None}
76
+ return json.dumps(kept, sort_keys=True, separators=(",", ":"), default=str)
77
+
78
+
79
+ @dataclass
80
+ class Turn:
81
+ """One request and the response it produced."""
82
+ index: int
83
+ fingerprint: str
84
+ request: dict
85
+ response: dict
86
+ model: str = ""
87
+ usage: dict = field(default_factory=dict)
88
+ # What the CALLER configured, e.g. "openrouter/vendor/model:free".
89
+ # `model` is what litellm passed on after stripping the provider prefix,
90
+ # and that stripped form is what the request carries and what the
91
+ # fingerprint is built from. Replay needs the routable name to reconfigure
92
+ # the run; matching needs the stripped one. They are not the same string
93
+ # and using one for both breaks replay (`docs/0029` §3).
94
+ provider_model: str = ""
95
+ # The environment the recording was made in. A cassette is NOT portable:
96
+ # the SDK builds a platform-dependent system prompt -- a Windows recording
97
+ # says "powershell" in message 0 -- so a cassette recorded on one OS misses
98
+ # on turn 0 everywhere else. Recording this is what lets a replay say so
99
+ # instead of looking like a behaviour change (`docs/0029` §6).
100
+ env: dict = field(default_factory=dict)
101
+
102
+ def to_json(self) -> str:
103
+ return json.dumps({
104
+ "index": self.index, "fingerprint": self.fingerprint,
105
+ "model": self.model, "provider_model": self.provider_model,
106
+ "usage": self.usage, "env": self.env,
107
+ "request": self.request, "response": self.response,
108
+ }, default=str)
109
+
110
+ @classmethod
111
+ def from_json(cls, line: str) -> "Turn":
112
+ d = json.loads(line)
113
+ return cls(index=d["index"], fingerprint=d["fingerprint"],
114
+ request=d["request"], response=d["response"],
115
+ model=d.get("model", ""), usage=d.get("usage") or {},
116
+ provider_model=d.get("provider_model", ""),
117
+ env=d.get("env") or {})
118
+
119
+
120
+ @dataclass
121
+ class Miss:
122
+ """A replayed request that the cassette has no recording for."""
123
+ turn: int
124
+ request: dict
125
+ nearest: Turn | None = None
126
+ recorded_turns: int = 0
127
+
128
+ def describe(self) -> str:
129
+ """Say what diverged, not merely that something did."""
130
+ if self.nearest is None:
131
+ return (f"turn {self.turn}: the recording ended after "
132
+ f"{self.recorded_turns} turns — this run wanted more")
133
+ want = _messages(self.nearest.request)
134
+ got = _messages(self.request)
135
+ if len(want) != len(got):
136
+ return (f"turn {self.turn}: {len(got)} messages, recorded run had "
137
+ f"{len(want)} — the conversation took a different shape")
138
+ for i, (a, b) in enumerate(zip(want, got)):
139
+ if a != b:
140
+ return (f"turn {self.turn}: message {i} ({b.get('role')}) "
141
+ f"differs from the recording")
142
+ return f"turn {self.turn}: same messages, different tools or model"
143
+
144
+
145
+ def current_env() -> dict:
146
+ """What this machine is, for the parts a recording depends on."""
147
+ import platform
148
+ try:
149
+ from importlib.metadata import version
150
+ sdk = version("openhands-sdk")
151
+ except Exception: # noqa: BLE001
152
+ sdk = "?"
153
+ return {"platform": sys.platform, "python": platform.python_version(),
154
+ "openhands_sdk": sdk}
155
+
156
+
157
+ def incompatible(cassette: "Cassette") -> str | None:
158
+ """Why this cassette cannot replay here, or None if it can.
159
+
160
+ Only `platform` is fatal, and it is fatal for a concrete reason rather than
161
+ caution: the SDK writes the shell name into the system prompt, so message 0
162
+ differs and every turn misses. The SDK version is reported when it differs
163
+ but not refused -- a prompt change would show up as an honest miss.
164
+ """
165
+ if not cassette.turns:
166
+ return "the cassette is empty"
167
+ rec = cassette.turns[0].env or {}
168
+ here = current_env()
169
+ if rec.get("platform") and rec["platform"] != here["platform"]:
170
+ return (f"recorded on {rec['platform']}, replaying on "
171
+ f"{here['platform']} -- the SDK puts the shell name in the "
172
+ f"system prompt, so turn 0 cannot match")
173
+ return None
174
+
175
+
176
+ def _messages(request: dict) -> list[dict]:
177
+ m = request.get("messages")
178
+ return m if isinstance(m, list) else []
179
+
180
+
181
+ class Cassette:
182
+ """Recorded turns, addressed by fingerprint.
183
+
184
+ Duplicate fingerprints are kept in order and served in order: an agent that
185
+ genuinely asks the same question twice must get both recorded answers, not
186
+ the first one twice.
187
+ """
188
+
189
+ def __init__(self, turns: list[Turn] | None = None):
190
+ self.turns: list[Turn] = list(turns or [])
191
+ self.misses: list[Miss] = []
192
+ self._served: set[int] = set()
193
+ self._index: dict[str, list[int]] = {}
194
+ for i, t in enumerate(self.turns):
195
+ self._index.setdefault(t.fingerprint, []).append(i)
196
+
197
+ # ── recording ──────────────────────────────────────────────────────
198
+ def append(self, request: dict, response: dict, *, model: str = "",
199
+ usage: dict | None = None, provider_model: str = "",
200
+ env: dict | None = None) -> Turn:
201
+ t = Turn(index=len(self.turns), fingerprint=fingerprint(request),
202
+ request=request, response=response, model=model,
203
+ usage=usage or {}, provider_model=provider_model,
204
+ env=env if env is not None else current_env())
205
+ self.turns.append(t)
206
+ self._index.setdefault(t.fingerprint, []).append(t.index)
207
+ return t
208
+
209
+ # ── replaying ──────────────────────────────────────────────────────
210
+ def match(self, request: dict) -> Turn | None:
211
+ """The recorded response for this request, or None on a divergence."""
212
+ fp = fingerprint(request)
213
+ for i in self._index.get(fp, ()):
214
+ if i not in self._served:
215
+ self._served.add(i)
216
+ return self.turns[i]
217
+
218
+ self.misses.append(Miss(turn=len(self._served), request=request,
219
+ nearest=self._nearest(),
220
+ recorded_turns=len(self.turns)))
221
+ return None
222
+
223
+ def _nearest(self) -> Turn | None:
224
+ """The turn the recorded run would have been at by now.
225
+
226
+ Not a similarity search -- just position. It is what makes a miss
227
+ legible: 'at this point the recording asked X, you asked Y'.
228
+ """
229
+ n = len(self._served)
230
+ return self.turns[n] if n < len(self.turns) else None
231
+
232
+ @property
233
+ def exhausted(self) -> bool:
234
+ return len(self._served) >= len(self.turns)
235
+
236
+ def unplayed(self) -> list[Turn]:
237
+ """Recorded turns never reached. A short run is a divergence too."""
238
+ return [t for i, t in enumerate(self.turns) if i not in self._served]
239
+
240
+ # ── storage ────────────────────────────────────────────────────────
241
+ def save(self, path: str | Path) -> Path:
242
+ p = Path(path)
243
+ p.parent.mkdir(parents=True, exist_ok=True)
244
+ with p.open("w", encoding="utf-8", newline="\n") as fh:
245
+ for t in self.turns:
246
+ fh.write(t.to_json() + "\n")
247
+ return p
248
+
249
+ @classmethod
250
+ def load(cls, path: str | Path) -> "Cassette":
251
+ p = Path(path)
252
+ if not p.exists():
253
+ raise FileNotFoundError(f"no cassette at {p}")
254
+ turns = [Turn.from_json(line) for line in _lines(p)]
255
+ return cls(turns)
256
+
257
+ def __len__(self) -> int:
258
+ return len(self.turns)
259
+
260
+ def __iter__(self) -> Iterator[Turn]:
261
+ return iter(self.turns)
262
+
263
+
264
+ def _lines(p: Path) -> Iterator[str]:
265
+ with p.open(encoding="utf-8") as fh:
266
+ for line in fh:
267
+ if line.strip():
268
+ yield line
269
+
270
+
271
+ def summarise(cassette: Cassette) -> dict[str, Any]:
272
+ """What a replay run actually did. The evaluator's output."""
273
+ return {
274
+ "turns_recorded": len(cassette.turns),
275
+ "turns_replayed": len(cassette.turns) - len(cassette.unplayed()),
276
+ "misses": len(cassette.misses),
277
+ "unplayed": len(cassette.unplayed()),
278
+ "diverged": bool(cassette.misses) or bool(cassette.unplayed()),
279
+ "first_divergence": (cassette.misses[0].describe()
280
+ if cassette.misses else None),
281
+ }