handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
agentctl/kernel/paths.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
r"""Which files would this call write, and do any of them escape the workspace?
|
|
2
|
+
|
|
3
|
+
Separate from `classify.py` on purpose. The classifier answers *what kind of
|
|
4
|
+
effect is this* -- the question the ledger needs to decide whether replaying it
|
|
5
|
+
is safe. This module answers *where does it land*, which is a question about
|
|
6
|
+
authority, not about replay.
|
|
7
|
+
|
|
8
|
+
`docs/0025` §4 draws that line:
|
|
9
|
+
|
|
10
|
+
the gate stops an effect happening TWICE
|
|
11
|
+
confirmation stops an effect happening AT ALL without a human
|
|
12
|
+
|
|
13
|
+
Escalating `write_file(path="~/.ssh/authorized_keys")` to `DESTRUCTIVE` would
|
|
14
|
+
have collapsed the two -- the ledger would then treat an ordinary idempotent
|
|
15
|
+
write as unrecoverable, and a perfectly safe replay would start failing closed.
|
|
16
|
+
`cp a.txt /tmp/b.txt` is not destructive. It is merely none of the agent's
|
|
17
|
+
business. So the effect class stays honest and the *authorization* layer asks.
|
|
18
|
+
|
|
19
|
+
## What counts as a write target
|
|
20
|
+
|
|
21
|
+
Only things that are unambiguously write destinations:
|
|
22
|
+
|
|
23
|
+
* the declared path argument of a structured tool (`write_file`, `edit`, ...)
|
|
24
|
+
* a shell redirect target -- `> f`, `>> f`, `tee f`
|
|
25
|
+
* arguments to commands whose entire purpose is to mutate a named file
|
|
26
|
+
|
|
27
|
+
Read paths are deliberately NOT collected. `grep secret /etc/shadow` reads
|
|
28
|
+
outside the workspace, which is a real concern and a different one
|
|
29
|
+
(exfiltration, not authority over the machine); pretending this module covers
|
|
30
|
+
it would be worse than not covering it.
|
|
31
|
+
"""
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import os
|
|
35
|
+
import re
|
|
36
|
+
from pathlib import Path, PurePath
|
|
37
|
+
|
|
38
|
+
from .ledger.models import ToolCall
|
|
39
|
+
|
|
40
|
+
# Arguments that name a file, for structured (non-shell) tools.
|
|
41
|
+
PATH_ARGS = ("path", "file_path", "filename", "file", "dest", "destination")
|
|
42
|
+
|
|
43
|
+
# `> f`, `>> f`, `2> f` -- but not `2>&1`, which writes no file.
|
|
44
|
+
_REDIRECT = re.compile(r"(?<![>&\-])>{1,2}(?![>&])\s*([^\s;&|<>]+)")
|
|
45
|
+
|
|
46
|
+
# Commands whose named arguments ARE the thing being changed.
|
|
47
|
+
_FILE_MUTATORS = re.compile(
|
|
48
|
+
r"^\s*(?:sudo\s+)?(rm|mv|cp|touch|chmod|chown|shred|truncate|tee|ln|mkdir"
|
|
49
|
+
r"|rmdir|install)\b(.*)$"
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
# `cp`, `mv` and friends take sources then ONE destination, and only the
|
|
53
|
+
# destination is written. Collecting every operand would flag
|
|
54
|
+
# `cp /etc/hosts ./local` -- a read from outside, copied in -- as an
|
|
55
|
+
# unauthorised write, and a prompt that is wrong is worse than no prompt (§5).
|
|
56
|
+
_LAST_OPERAND_ONLY = {"cp", "mv", "ln", "install"}
|
|
57
|
+
|
|
58
|
+
# Not paths: flags, operators, and the shell's own punctuation.
|
|
59
|
+
_NOT_A_PATH = re.compile(r"^(-|\d+$|&|\||;|<|>)")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def write_targets(call: ToolCall, *, arg_key: str = "command") -> list[str]:
|
|
63
|
+
"""Every path this call would write to. Best effort, biased to over-report.
|
|
64
|
+
|
|
65
|
+
A path reported here that is not really written costs one confirmation
|
|
66
|
+
prompt. A path missed here is an unauthorised write, so the bias is
|
|
67
|
+
deliberate -- but see the module docstring for what is out of scope.
|
|
68
|
+
"""
|
|
69
|
+
out: list[str] = []
|
|
70
|
+
|
|
71
|
+
for name in PATH_ARGS:
|
|
72
|
+
if (v := call.args.get(name)) and isinstance(v, str):
|
|
73
|
+
out.append(v)
|
|
74
|
+
|
|
75
|
+
if isinstance(command := call.args.get(arg_key), str):
|
|
76
|
+
out.extend(_shell_write_targets(command))
|
|
77
|
+
|
|
78
|
+
seen: set[str] = set()
|
|
79
|
+
return [p for p in out if not (p in seen or seen.add(p))]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _shell_write_targets(command: str) -> list[str]:
|
|
83
|
+
"""Redirect destinations and the operands of file-mutating commands."""
|
|
84
|
+
from .classify import Classifier # same layer, no cycle at import
|
|
85
|
+
|
|
86
|
+
out: list[str] = []
|
|
87
|
+
for segment in Classifier._segments(command):
|
|
88
|
+
out.extend(_REDIRECT.findall(segment))
|
|
89
|
+
if m := _FILE_MUTATORS.match(segment):
|
|
90
|
+
operands = [
|
|
91
|
+
t for token in m.group(2).split()
|
|
92
|
+
if (t := token.strip("'\"")) and not _NOT_A_PATH.match(t)
|
|
93
|
+
]
|
|
94
|
+
if m.group(1) in _LAST_OPERAND_ONLY:
|
|
95
|
+
operands = operands[-1:]
|
|
96
|
+
out.extend(operands)
|
|
97
|
+
return out
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def escapes(target: str, root: str | Path) -> bool:
|
|
101
|
+
"""Does `target` resolve outside `root`?
|
|
102
|
+
|
|
103
|
+
Resolution is lexical, and that is a deliberate choice: `Path.resolve()`
|
|
104
|
+
follows symlinks, so a link planted inside the workspace would resolve
|
|
105
|
+
outside it and be reported -- correct -- but a link planted OUTSIDE could
|
|
106
|
+
resolve back in and be waved through. Lexical normalisation cannot be
|
|
107
|
+
tricked that way, at the cost of treating a benign symlink as an escape.
|
|
108
|
+
|
|
109
|
+
`~` is expanded first. Without that, `~/.bashrc` is a relative path and
|
|
110
|
+
looks like it sits safely inside the workspace, which is precisely the case
|
|
111
|
+
that prompted this module.
|
|
112
|
+
"""
|
|
113
|
+
root_p = PurePath(os.path.abspath(str(root)))
|
|
114
|
+
expanded = os.path.expanduser(os.path.expandvars(target))
|
|
115
|
+
|
|
116
|
+
if not os.path.isabs(expanded):
|
|
117
|
+
expanded = os.path.join(str(root_p), expanded)
|
|
118
|
+
|
|
119
|
+
resolved = PurePath(os.path.normpath(expanded))
|
|
120
|
+
try:
|
|
121
|
+
resolved.relative_to(root_p)
|
|
122
|
+
return False
|
|
123
|
+
except ValueError:
|
|
124
|
+
return True
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
#: Places a redirect can point that are not files anyone owns. `2>/dev/null`
|
|
128
|
+
#: asked for confirmation as "writes outside the workspace" in a live pooled
|
|
129
|
+
#: run, and with no terminal the prompt read EOF and refused it (`docs/0047`).
|
|
130
|
+
#: Exact names only: `/dev/sda` is a device, not a sink, and stays reported.
|
|
131
|
+
_SINKS = frozenset({"/dev/null", "/dev/stdout", "/dev/stderr", "/dev/tty",
|
|
132
|
+
"nul", "nul:", "con"})
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def escaping_writes(call: ToolCall, root: str | Path | None) -> list[str]:
|
|
136
|
+
"""Write targets that land outside `root`. Empty when `root` is None."""
|
|
137
|
+
if root is None:
|
|
138
|
+
return []
|
|
139
|
+
return [t for t in write_targets(call)
|
|
140
|
+
if t.lower() not in _SINKS and escapes(t, root)]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# ── installs into the user's environment (docs/0047 §5, docs/0052) ─────
|
|
144
|
+
#
|
|
145
|
+
# A package install writes no file this module can name -- it writes into an
|
|
146
|
+
# interpreter's site-packages, a global node_modules, /usr/bin -- so neither
|
|
147
|
+
# `escaping_writes` nor the effect class caught it. A live run showed the
|
|
148
|
+
# agent running `pip install pytest` on the host, unasked. It is not
|
|
149
|
+
# destructive, and it is not the agent's business either: the same
|
|
150
|
+
# authorization question as a write outside the workspace.
|
|
151
|
+
_INSTALLERS = re.compile(
|
|
152
|
+
r"^\s*(?:sudo\s+)?(?:"
|
|
153
|
+
r"(?P<pip>(?:\S*[/\\])?(?:pip3?(?:\.\d+)?|python3?(?:\.\d+)?(?:\.exe)?\s+-m\s+pip))"
|
|
154
|
+
r"\s+install\b"
|
|
155
|
+
r"|uv\s+(?:pip\s+install|tool\s+install|add\s+--global)\b"
|
|
156
|
+
r"|pipx\s+(?:install|inject)\b"
|
|
157
|
+
r"|(?:npm|pnpm)\s+(?:install|i|add)\b(?=.*\s(?:-g|--global)\b)"
|
|
158
|
+
r"|yarn\s+global\s+add\b"
|
|
159
|
+
r"|(?:apt|apt-get|dnf|yum|apk|zypper)\s+(?:install|add)\b"
|
|
160
|
+
r"|pacman\s+-S\b"
|
|
161
|
+
r"|(?:brew|cargo|gem|conda|mamba|choco|winget|scoop)\s+install\b"
|
|
162
|
+
r"|go\s+install\b"
|
|
163
|
+
r")")
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def environment_installs(call: ToolCall, root: str | Path | None) -> list[str]:
|
|
167
|
+
"""Shell segments that install software outside the workspace.
|
|
168
|
+
|
|
169
|
+
An install into an environment that lives INSIDE the workspace is not
|
|
170
|
+
flagged: `.venv/bin/pip install pytest`, `uv pip install --python
|
|
171
|
+
.venv/...`, or `pip install --target ./vendor` change only the repo, which
|
|
172
|
+
is the agent's to change. A plain `pip install` goes to whichever
|
|
173
|
+
interpreter is on PATH -- usually the user's own -- and is flagged.
|
|
174
|
+
"""
|
|
175
|
+
command = call.args.get("command")
|
|
176
|
+
if not isinstance(command, str) or root is None:
|
|
177
|
+
return []
|
|
178
|
+
from .classify import Classifier # same layer, no cycle at import
|
|
179
|
+
|
|
180
|
+
out = []
|
|
181
|
+
for segment in Classifier._segments(command):
|
|
182
|
+
m = _INSTALLERS.match(segment)
|
|
183
|
+
if not m:
|
|
184
|
+
continue
|
|
185
|
+
if _lands_inside(segment, m, root):
|
|
186
|
+
continue
|
|
187
|
+
out.append(segment.strip())
|
|
188
|
+
return out
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _lands_inside(segment: str, m: re.Match, root) -> bool:
|
|
192
|
+
"""Does this install target an environment inside the workspace?"""
|
|
193
|
+
# The installer itself lives in the workspace: `.venv/bin/pip`,
|
|
194
|
+
# `./venv/Scripts/python -m pip`.
|
|
195
|
+
exe = (m.group("pip") or "").split()[0] if m.group("pip") else ""
|
|
196
|
+
if exe and ("/" in exe or "\\" in exe) and not escapes(exe, root):
|
|
197
|
+
return True
|
|
198
|
+
# An explicit destination inside the workspace.
|
|
199
|
+
for flag in ("--target", "-t", "--prefix", "--root", "--python", "-p"):
|
|
200
|
+
if (d := re.search(rf"(?:^|\s){re.escape(flag)}[=\s]+(\S+)", segment)):
|
|
201
|
+
if not escapes(d.group(1).strip("'\""), root):
|
|
202
|
+
return True
|
|
203
|
+
return False
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
r"""Read a compiled policy. Lookups only — no evaluation in-band.
|
|
2
|
+
|
|
3
|
+
`docs/0012` §5.2: *"compiles to a flat lookup artifact the kernel reads from
|
|
4
|
+
disk -- no evaluation logic in-band."* That sentence is the whole design, and
|
|
5
|
+
it is the same trick as the capability matrix: the control plane does the
|
|
6
|
+
thinking, writes an artifact, and the kernel reads it. A malformed policy fails
|
|
7
|
+
where a human is watching, not in the middle of a run.
|
|
8
|
+
|
|
9
|
+
So there is no YAML here, no schema, no merge, no defaults resolution, and no
|
|
10
|
+
`if` on a pool name. Those live in `control/policy/compile.py`. If this module
|
|
11
|
+
ever needs to *decide* something, the compiler should have decided it.
|
|
12
|
+
|
|
13
|
+
## Budget, and why a cap can lie
|
|
14
|
+
|
|
15
|
+
`docs/0021` §5 found litellm reports cost `0.0` for endpoints it cannot price,
|
|
16
|
+
indistinguishable from a call that was genuinely free. A cap compared against
|
|
17
|
+
measured spend therefore **under-blocks**: the real figure is higher than the
|
|
18
|
+
one being checked, always in the direction of spending more than intended.
|
|
19
|
+
|
|
20
|
+
The guard refuses to hide that. Every verdict carries the pricing coverage it
|
|
21
|
+
was computed from, and what to do about incomplete coverage is a decision the
|
|
22
|
+
policy states out loud (`on_unpriced`) rather than one this module makes
|
|
23
|
+
quietly. The default is `warn`, which is fail-open — consistent with
|
|
24
|
+
`docs/0008` §6.5, where cost and routing fail open and only effect decisions
|
|
25
|
+
fail closed. Choosing `block` is available and is a real trade: it stops work
|
|
26
|
+
when measurement is incomplete, which on a free-tier pool is most of the time.
|
|
27
|
+
"""
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
from dataclasses import dataclass
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Any
|
|
34
|
+
|
|
35
|
+
DEFAULT_POLICY = (
|
|
36
|
+
Path(__file__).resolve().parent.parent
|
|
37
|
+
/ "control" / "policy" / "data" / "policy.compiled.json"
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class BudgetVerdict:
|
|
43
|
+
"""What the budget says, and how much to trust it."""
|
|
44
|
+
allowed: bool
|
|
45
|
+
reason: str
|
|
46
|
+
scope: str = "" # "per_task" | "daily"
|
|
47
|
+
spent_usd: float = 0.0
|
|
48
|
+
limit_usd: float | None = None
|
|
49
|
+
coverage: float = 1.0 # share of calls that could be priced
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def trustworthy(self) -> bool:
|
|
53
|
+
return self.coverage >= 1.0
|
|
54
|
+
|
|
55
|
+
def describe(self) -> str:
|
|
56
|
+
if self.limit_usd is None:
|
|
57
|
+
return "no budget configured"
|
|
58
|
+
base = f"${self.spent_usd:.4f} of ${self.limit_usd:.2f} ({self.scope})"
|
|
59
|
+
if not self.trustworthy:
|
|
60
|
+
# Never print a spend figure without saying what it omits.
|
|
61
|
+
return (f"{base} — but only {self.coverage:.0%} of calls could be "
|
|
62
|
+
f"priced, so the real spend is HIGHER")
|
|
63
|
+
return base
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class Policy:
|
|
67
|
+
"""A compiled policy artifact. Every method is a lookup."""
|
|
68
|
+
|
|
69
|
+
def __init__(self, data: dict | None = None):
|
|
70
|
+
self._d: dict = data or {}
|
|
71
|
+
|
|
72
|
+
@classmethod
|
|
73
|
+
def load(cls, path: str | Path | None = None) -> "Policy":
|
|
74
|
+
p = Path(path or DEFAULT_POLICY)
|
|
75
|
+
if not p.exists():
|
|
76
|
+
raise FileNotFoundError(
|
|
77
|
+
f"no compiled policy at {p}\n"
|
|
78
|
+
f" compile one: agentctl policy <policy.yaml>")
|
|
79
|
+
return cls(json.loads(p.read_text(encoding="utf-8")))
|
|
80
|
+
|
|
81
|
+
@classmethod
|
|
82
|
+
def empty(cls) -> "Policy":
|
|
83
|
+
"""No policy. Every lookup returns nothing, and nothing is enforced."""
|
|
84
|
+
return cls({})
|
|
85
|
+
|
|
86
|
+
def __bool__(self) -> bool:
|
|
87
|
+
return bool(self._d)
|
|
88
|
+
|
|
89
|
+
# ── effects ────────────────────────────────────────────────────────
|
|
90
|
+
def effect_rule(self, effect_class: str) -> str | None:
|
|
91
|
+
"""e.g. DESTRUCTIVE -> "require_human_approval"."""
|
|
92
|
+
return (self._d.get("effects") or {}).get(str(effect_class))
|
|
93
|
+
|
|
94
|
+
def requires_approval(self, effect_class: str) -> bool:
|
|
95
|
+
return self.effect_rule(effect_class) == "require_human_approval"
|
|
96
|
+
|
|
97
|
+
# ── budget ─────────────────────────────────────────────────────────
|
|
98
|
+
def limit(self, scope: str) -> float | None:
|
|
99
|
+
return (self._d.get("budget") or {}).get(f"{scope}_usd")
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def on_exceeded(self) -> str:
|
|
103
|
+
return (self._d.get("budget") or {}).get("on_exceeded", "block")
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def on_unpriced(self) -> str:
|
|
107
|
+
return (self._d.get("budget") or {}).get("on_unpriced", "warn")
|
|
108
|
+
|
|
109
|
+
def check_budget(self, spent_usd: float, scope: str = "per_task",
|
|
110
|
+
coverage: float = 1.0) -> BudgetVerdict:
|
|
111
|
+
"""Is there room left? The spend is SUPPLIED, never fetched.
|
|
112
|
+
|
|
113
|
+
The cost ledger lives in the control plane and the kernel may not
|
|
114
|
+
import it (`docs/0008` R2), which is not an inconvenience here but the
|
|
115
|
+
right shape: a guard that queried a database in-band would fail when
|
|
116
|
+
the control plane is down, and this must not.
|
|
117
|
+
"""
|
|
118
|
+
limit = self.limit(scope)
|
|
119
|
+
if limit is None:
|
|
120
|
+
return BudgetVerdict(True, "no budget configured", scope,
|
|
121
|
+
spent_usd, None, coverage)
|
|
122
|
+
|
|
123
|
+
v = BudgetVerdict(spent_usd < limit, "", scope, spent_usd, limit, coverage)
|
|
124
|
+
|
|
125
|
+
if spent_usd >= limit:
|
|
126
|
+
if self.on_exceeded == "warn":
|
|
127
|
+
return BudgetVerdict(True, f"over budget ({v.describe()}), "
|
|
128
|
+
f"policy says warn", scope, spent_usd,
|
|
129
|
+
limit, coverage)
|
|
130
|
+
return BudgetVerdict(False, f"budget exceeded: {v.describe()}",
|
|
131
|
+
scope, spent_usd, limit, coverage)
|
|
132
|
+
|
|
133
|
+
# Under the cap on paper. If the measurement is incomplete the real
|
|
134
|
+
# figure is higher, and saying nothing would be the wrong silence.
|
|
135
|
+
if coverage < 1.0 and self.on_unpriced == "block":
|
|
136
|
+
return BudgetVerdict(
|
|
137
|
+
False,
|
|
138
|
+
f"only {coverage:.0%} of calls could be priced and policy says "
|
|
139
|
+
f"block; measured {v.describe()}",
|
|
140
|
+
scope, spent_usd, limit, coverage)
|
|
141
|
+
|
|
142
|
+
return BudgetVerdict(True, v.describe(), scope, spent_usd, limit,
|
|
143
|
+
coverage)
|
|
144
|
+
|
|
145
|
+
# ── routing (read-only; the compiler resolved it) ───────────────────
|
|
146
|
+
@property
|
|
147
|
+
def default_pool(self) -> str | None:
|
|
148
|
+
return (self._d.get("routing") or {}).get("default_pool")
|
|
149
|
+
|
|
150
|
+
def pool(self, name: str) -> list[str]:
|
|
151
|
+
return list(((self._d.get("routing") or {}).get("pools") or {}).get(name, []))
|
|
152
|
+
|
|
153
|
+
@property
|
|
154
|
+
def escalation(self) -> dict[str, Any]:
|
|
155
|
+
return dict((self._d.get("routing") or {}).get("escalation") or {})
|
|
156
|
+
|
|
157
|
+
@property
|
|
158
|
+
def source_sha256(self) -> str:
|
|
159
|
+
"""Which policy.yaml this was compiled from. For `agentctl policy show`."""
|
|
160
|
+
return self._d.get("source_sha256", "")
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""Reconciliation probes: ask the world whether an effect already landed."""
|
|
2
|
+
from .base import (
|
|
3
|
+
DID_NOT_LAND,
|
|
4
|
+
INCONCLUSIVE,
|
|
5
|
+
LANDED,
|
|
6
|
+
SAFE_TO_RETRY,
|
|
7
|
+
ProbeRegistry,
|
|
8
|
+
ReconciliationProbe,
|
|
9
|
+
)
|
|
10
|
+
from .external import IdempotencyProbe
|
|
11
|
+
from .filesystem import FileAppendProbe
|
|
12
|
+
from .git import GitProbe
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"DID_NOT_LAND", "INCONCLUSIVE", "LANDED", "SAFE_TO_RETRY",
|
|
16
|
+
"ProbeRegistry", "ReconciliationProbe",
|
|
17
|
+
"FileAppendProbe", "GitProbe", "IdempotencyProbe",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def default_registry(repo_root=None, idempotency_fields=None) -> ProbeRegistry:
|
|
22
|
+
"""The probes worth having on by default.
|
|
23
|
+
|
|
24
|
+
Order matters for auto-selection: the specific probes are tried before the
|
|
25
|
+
idempotency one, which handles any call carrying a key.
|
|
26
|
+
"""
|
|
27
|
+
return ProbeRegistry(
|
|
28
|
+
GitProbe(repo_root),
|
|
29
|
+
FileAppendProbe(),
|
|
30
|
+
IdempotencyProbe(idempotency_fields),
|
|
31
|
+
)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
"""Reconciliation probes. Spec: `docs/0012` §3.3, revised.
|
|
2
|
+
|
|
3
|
+
The problem: after a crash we hold an `INTENT` record and cannot tell whether
|
|
4
|
+
the effect landed. A probe asks the world.
|
|
5
|
+
|
|
6
|
+
**Design change from `docs/0012` §3.3.** That section specified injecting the
|
|
7
|
+
intent hash into the effect itself (a git trailer), which requires mutating the
|
|
8
|
+
command before execution — only possible at Seam C. This uses *world
|
|
9
|
+
fingerprinting* instead:
|
|
10
|
+
|
|
11
|
+
capture() before execution -> record what the world looked like
|
|
12
|
+
probe() after a crash -> compare; did it change?
|
|
13
|
+
|
|
14
|
+
That needs no cooperation from the tool, works at Seam B, and generalises to
|
|
15
|
+
effects that have nowhere to carry a marker.
|
|
16
|
+
|
|
17
|
+
Remote effects cannot be fingerprinted this way — the world in question is
|
|
18
|
+
somebody else's server. They are handled by a different mechanism entirely
|
|
19
|
+
(`external.py`, `docs/0020`): make the retry *safe* rather than trying to
|
|
20
|
+
determine whether it is *necessary*.
|
|
21
|
+
|
|
22
|
+
**The assumption fingerprinting rests on**, stated plainly: the agent's
|
|
23
|
+
workspace has a single writer. If a human commits to the same repo during the crash window, a
|
|
24
|
+
probe can misread that as the agent's effect. Every probe returns
|
|
25
|
+
`INCONCLUSIVE` rather than guessing when it can distinguish the cases, and the
|
|
26
|
+
gate fails closed on `INCONCLUSIVE`.
|
|
27
|
+
"""
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
from typing import Protocol, runtime_checkable
|
|
31
|
+
|
|
32
|
+
LANDED = "LANDED"
|
|
33
|
+
DID_NOT_LAND = "DID_NOT_LAND"
|
|
34
|
+
INCONCLUSIVE = "INCONCLUSIVE"
|
|
35
|
+
|
|
36
|
+
#: The effect may have landed, and it does not matter: the call carries an
|
|
37
|
+
#: idempotency key, so re-sending it produces at most one effect at the remote
|
|
38
|
+
#: end. Answering "did it land?" is impossible for a remote API and also the
|
|
39
|
+
#: wrong question -- see `docs/0020`.
|
|
40
|
+
SAFE_TO_RETRY = "SAFE_TO_RETRY"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@runtime_checkable
|
|
44
|
+
class ReconciliationProbe(Protocol):
|
|
45
|
+
"""Answers 'did this effect already happen?' from outside the ledger."""
|
|
46
|
+
|
|
47
|
+
name: str
|
|
48
|
+
|
|
49
|
+
def handles(self, call) -> bool:
|
|
50
|
+
"""Can this probe say anything about this call?"""
|
|
51
|
+
|
|
52
|
+
def capture(self, call) -> dict | None:
|
|
53
|
+
"""Fingerprint the world BEFORE execution.
|
|
54
|
+
|
|
55
|
+
Returns a JSON-serialisable dict stored on the INTENT record, or None
|
|
56
|
+
if nothing useful can be captured. Must never raise — a probe that
|
|
57
|
+
explodes here would block the call entirely.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
def probe(self, call, record) -> str:
|
|
61
|
+
"""LANDED | DID_NOT_LAND | SAFE_TO_RETRY | INCONCLUSIVE.
|
|
62
|
+
|
|
63
|
+
Uses `record.pre_state`. Must never raise. When in doubt, return
|
|
64
|
+
INCONCLUSIVE and let the gate fail closed.
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class ProbeRegistry:
|
|
69
|
+
"""Holds the probes and picks one per call."""
|
|
70
|
+
|
|
71
|
+
def __init__(self, *probes: ReconciliationProbe):
|
|
72
|
+
self._probes = list(probes)
|
|
73
|
+
self._by_name = {p.name: p for p in probes}
|
|
74
|
+
|
|
75
|
+
def add(self, probe: ReconciliationProbe) -> "ProbeRegistry":
|
|
76
|
+
self._probes.append(probe)
|
|
77
|
+
self._by_name[probe.name] = probe
|
|
78
|
+
return self
|
|
79
|
+
|
|
80
|
+
def get(self, name: str | None):
|
|
81
|
+
return self._by_name.get(name) if name else None
|
|
82
|
+
|
|
83
|
+
def for_call(self, call, name: str | None = None):
|
|
84
|
+
"""Prefer the probe the capability matrix named; else the first match.
|
|
85
|
+
|
|
86
|
+
A *named* probe is trusted without consulting `handles()`. The matrix is
|
|
87
|
+
an operator declaration -- "this tool commits to git" -- and it knows
|
|
88
|
+
things a heuristic cannot: a dedicated `commit` tool carries no "git
|
|
89
|
+
commit" string in its arguments to sniff for.
|
|
90
|
+
"""
|
|
91
|
+
if name and (p := self._by_name.get(name)) is not None:
|
|
92
|
+
return p
|
|
93
|
+
for p in self._probes:
|
|
94
|
+
if _safe_handles(p, call):
|
|
95
|
+
return p
|
|
96
|
+
return None
|
|
97
|
+
|
|
98
|
+
def __len__(self) -> int:
|
|
99
|
+
return len(self._probes)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _safe_handles(probe, call) -> bool:
|
|
103
|
+
try:
|
|
104
|
+
return bool(probe.handles(call))
|
|
105
|
+
except Exception: # noqa: BLE001
|
|
106
|
+
return False
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Idempotency-key probe for EXTERNAL effects. `docs/0020`.
|
|
2
|
+
|
|
3
|
+
Git and filesystem effects are fingerprintable: we look at the world before and
|
|
4
|
+
after. A remote API is not — the world in question is somebody else's server,
|
|
5
|
+
and we cannot see it without its cooperation.
|
|
6
|
+
|
|
7
|
+
So this probe does not answer *"did it land?"*. It answers a different and
|
|
8
|
+
better question:
|
|
9
|
+
|
|
10
|
+
Is it SAFE to send this again?
|
|
11
|
+
|
|
12
|
+
If the call carries a stable idempotency key, the answer is yes regardless of
|
|
13
|
+
what happened, because a compliant server collapses the retry into the original
|
|
14
|
+
effect. That is the mechanism Stripe standardised, and it is the only general
|
|
15
|
+
way to make a remote non-idempotent effect replay-safe.
|
|
16
|
+
|
|
17
|
+
**This probe only works because the same key goes out twice.** Seam C stamps a
|
|
18
|
+
key derived from the call's arguments before execution, so a crash and resume
|
|
19
|
+
produce the same value. A key the model supplied itself is equally good. What
|
|
20
|
+
the probe refuses is a request that carried *no* key, or a different one —
|
|
21
|
+
there the remote has nothing to deduplicate against and the retry is simply a
|
|
22
|
+
second effect.
|
|
23
|
+
|
|
24
|
+
What this does *not* give you: the original response. A deduped retry returns
|
|
25
|
+
the server's recorded response, which is usually what you want — but the agent
|
|
26
|
+
sees the retry's return value, not the pre-crash one.
|
|
27
|
+
"""
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
|
|
32
|
+
from .base import INCONCLUSIVE, SAFE_TO_RETRY
|
|
33
|
+
|
|
34
|
+
#: Arguments a tool might already use for an idempotency key. Checked before
|
|
35
|
+
#: falling back to whatever the capability matrix declares.
|
|
36
|
+
_CONVENTIONAL_KEYS = (
|
|
37
|
+
"idempotency_key", "idempotencyKey", "request_id", "requestId",
|
|
38
|
+
"client_token", "clientToken", "dedupe_key", "dedupeKey",
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class IdempotencyProbe:
|
|
43
|
+
"""Declares a retry safe when the call carries a stable idempotency key.
|
|
44
|
+
|
|
45
|
+
Selection is deliberately conservative: it handles a call only when a key
|
|
46
|
+
field is present or the capability matrix names one. An EXTERNAL effect
|
|
47
|
+
with no key mechanism is left to fail closed, which is correct — there is
|
|
48
|
+
genuinely no safe way to retry it.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
name = "idempotency"
|
|
52
|
+
|
|
53
|
+
def __init__(self, key_fields: dict[str, str] | None = None):
|
|
54
|
+
#: {tool_name: argument holding the key}, from the capability matrix.
|
|
55
|
+
self.key_fields = key_fields or {}
|
|
56
|
+
|
|
57
|
+
# ── selection ──────────────────────────────────────────────────────
|
|
58
|
+
def handles(self, call) -> bool:
|
|
59
|
+
return self.key_field(call) is not None
|
|
60
|
+
|
|
61
|
+
def key_field(self, call) -> str | None:
|
|
62
|
+
"""Which argument carries the idempotency key for this call."""
|
|
63
|
+
declared = self.key_fields.get(call.tool_name)
|
|
64
|
+
if declared:
|
|
65
|
+
return declared
|
|
66
|
+
for k in _CONVENTIONAL_KEYS:
|
|
67
|
+
if k in call.args:
|
|
68
|
+
return k
|
|
69
|
+
return None
|
|
70
|
+
|
|
71
|
+
def key_for(self, call) -> str:
|
|
72
|
+
"""The key we stamp when the call has none. Stable across a crash.
|
|
73
|
+
|
|
74
|
+
Derived from the arguments *excluding the key field itself* — deriving
|
|
75
|
+
it from arguments that contain it would be circular: stamping the key
|
|
76
|
+
changes the args, which changes the key.
|
|
77
|
+
"""
|
|
78
|
+
import hashlib
|
|
79
|
+
import json as _json
|
|
80
|
+
|
|
81
|
+
field = self.key_field(call)
|
|
82
|
+
args = {k: v for k, v in call.args.items() if k != field}
|
|
83
|
+
canonical = _json.dumps({"t": call.tool_name, "a": args},
|
|
84
|
+
sort_keys=True, separators=(",", ":"), default=str)
|
|
85
|
+
return f"agentctl-{hashlib.sha256(canonical.encode()).hexdigest()[:32]}"
|
|
86
|
+
|
|
87
|
+
# ── before ─────────────────────────────────────────────────────────
|
|
88
|
+
def capture(self, call) -> dict | None:
|
|
89
|
+
"""Record the key that is actually going out with this request.
|
|
90
|
+
|
|
91
|
+
Not the key we would choose — the one on the call. A model-supplied
|
|
92
|
+
key is just as good, provided the same one is sent again.
|
|
93
|
+
"""
|
|
94
|
+
field = self.key_field(call)
|
|
95
|
+
if field is None:
|
|
96
|
+
return None
|
|
97
|
+
present = str(call.args.get(field) or "")
|
|
98
|
+
return {
|
|
99
|
+
"probe": self.name,
|
|
100
|
+
"field": field,
|
|
101
|
+
# Empty means no key went out, so a retry would be a second effect.
|
|
102
|
+
# That happens on a Seam-B-only deployment, where nothing stamps.
|
|
103
|
+
"key": present,
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
# ── after a crash ──────────────────────────────────────────────────
|
|
107
|
+
def probe(self, call, record) -> str:
|
|
108
|
+
pre = _pre(record)
|
|
109
|
+
if not pre or pre.get("probe") != self.name:
|
|
110
|
+
return INCONCLUSIVE
|
|
111
|
+
|
|
112
|
+
field, sent = pre.get("field"), pre.get("key")
|
|
113
|
+
if not field:
|
|
114
|
+
return INCONCLUSIVE
|
|
115
|
+
|
|
116
|
+
# No key went out, so the remote has nothing to deduplicate against
|
|
117
|
+
# and a retry would be a second effect.
|
|
118
|
+
if not sent:
|
|
119
|
+
return INCONCLUSIVE
|
|
120
|
+
|
|
121
|
+
# The same key must go out again. Anything else -- a regenerated key, a
|
|
122
|
+
# changed argument that feeds it -- makes the retry a fresh request.
|
|
123
|
+
now = str(call.args.get(field) or "") or self.key_for(call)
|
|
124
|
+
if now != sent:
|
|
125
|
+
return INCONCLUSIVE
|
|
126
|
+
|
|
127
|
+
return SAFE_TO_RETRY
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _pre(record) -> dict | None:
|
|
131
|
+
raw = getattr(record, "pre_state", None)
|
|
132
|
+
if not raw:
|
|
133
|
+
return None
|
|
134
|
+
try:
|
|
135
|
+
return json.loads(raw)
|
|
136
|
+
except Exception: # noqa: BLE001
|
|
137
|
+
return None
|