veract 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_runtime/__init__.py +3 -0
- agent_runtime/capabilities.py +221 -0
- agent_runtime/checkpoint.py +34 -0
- agent_runtime/cli.py +132 -0
- agent_runtime/compiler.py +82 -0
- agent_runtime/config.py +105 -0
- agent_runtime/contract.py +153 -0
- agent_runtime/engines.py +614 -0
- agent_runtime/executor.py +77 -0
- agent_runtime/guard.py +95 -0
- agent_runtime/llm.py +126 -0
- agent_runtime/mcp.py +147 -0
- agent_runtime/memory.py +78 -0
- agent_runtime/models.py +110 -0
- agent_runtime/planner.py +111 -0
- agent_runtime/recovery.py +39 -0
- agent_runtime/runtime.py +339 -0
- agent_runtime/scaffold.py +45 -0
- agent_runtime/tools.py +459 -0
- agent_runtime/verifier.py +247 -0
- veract-0.3.0.dist-info/METADATA +384 -0
- veract-0.3.0.dist-info/RECORD +26 -0
- veract-0.3.0.dist-info/WHEEL +5 -0
- veract-0.3.0.dist-info/entry_points.txt +3 -0
- veract-0.3.0.dist-info/licenses/LICENSE +9 -0
- veract-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""Capability broker: every action is checked against policy and written to an inspectable audit log.
|
|
2
|
+
|
|
3
|
+
Design rules (each one closes a hole seen in the v0.1 benchmark run):
|
|
4
|
+
* Network reads are http(s) only. ``file://`` and other schemes are denied (no local-file exfil via urlopen).
|
|
5
|
+
* File reads/writes must resolve inside the allowed roots; the runtime's own state (``.agent/``) and ``.git/``
|
|
6
|
+
are never writable by the agent, so it cannot rewrite its audit log, checkpoints or history.
|
|
7
|
+
* Commands are matched on argv, not on a shell glob: a small set of programs and sub-commands, every
|
|
8
|
+
path-like argument must stay inside the workspace, ``python -c`` / pytest plugins are refused, and code the
|
|
9
|
+
agent created during the mission cannot be executed without a human approving it.
|
|
10
|
+
* Every decision — allowed or denied — is appended to the mission's JSONL audit log.
|
|
11
|
+
"""
|
|
12
|
+
import fnmatch
|
|
13
|
+
import ipaddress
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import re
|
|
17
|
+
import shlex
|
|
18
|
+
import socket
|
|
19
|
+
import threading
|
|
20
|
+
import time
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from urllib.parse import urlparse
|
|
23
|
+
|
|
24
|
+
DEFAULT_POLICY = {
|
|
25
|
+
"fs_read": ["{workspace}"],
|
|
26
|
+
"fs_write": ["{workspace}"],
|
|
27
|
+
"fs_protected": [".agent", ".git"], # never writable by the agent (relative to workspace)
|
|
28
|
+
"net_domains": ["*"],
|
|
29
|
+
"block_private_net": True,
|
|
30
|
+
"private_allow": [], # hosts exempt from the private-net block (e.g. self-hosted SearXNG)
|
|
31
|
+
"mcp_allow": [], # "server.tool" patterns callable without approval
|
|
32
|
+
"exec_allow": [], # extra user globs; still subject to the argv/path rules
|
|
33
|
+
"exec_new_code": "approve", # approve|deny|allow: running code the agent itself created
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
# program -> allowed first arguments (None = any args, still path-checked)
|
|
37
|
+
EXEC_RULES = {
|
|
38
|
+
"python": {"-m": {"pytest", "unittest"}},
|
|
39
|
+
"python3": {"-m": {"pytest", "unittest"}},
|
|
40
|
+
"py": {"-m": {"pytest", "unittest"}},
|
|
41
|
+
"pytest": None,
|
|
42
|
+
"git": {"status", "diff", "log", "show"},
|
|
43
|
+
"ls": None,
|
|
44
|
+
"dir": None,
|
|
45
|
+
}
|
|
46
|
+
BANNED_ARGS = {"-c", "--rootdir", "-p", "--pyargs", "--confcutdir", "--import-mode", "-o", "--override-ini"}
|
|
47
|
+
NET_SCHEMES = ("http", "https")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def plausible_test_command(cmd):
|
|
51
|
+
"""Static pre-check (no workspace needed): python -m pytest|unittest or pytest, no banned/absolute args."""
|
|
52
|
+
try:
|
|
53
|
+
argv = shlex.split(cmd)
|
|
54
|
+
except ValueError:
|
|
55
|
+
return False
|
|
56
|
+
if not argv:
|
|
57
|
+
return False
|
|
58
|
+
prog = Path(argv[0]).name.lower().removesuffix(".exe")
|
|
59
|
+
if prog in ("python", "python3", "py"):
|
|
60
|
+
if argv[1:3] not in (["-m", "pytest"], ["-m", "unittest"]):
|
|
61
|
+
return False
|
|
62
|
+
elif prog != "pytest":
|
|
63
|
+
return False
|
|
64
|
+
return not any(a.split("=", 1)[0] in BANNED_ARGS or re.match(r"^(?:[A-Za-z]:[\\/]|[\\/]|~|\.\.)", a.split("=", 1)[-1])
|
|
65
|
+
for a in argv[1:])
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class PermissionDenied(Exception):
|
|
69
|
+
pass
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _is_pathlike(tok):
|
|
73
|
+
return bool(re.match(r"^(?:[A-Za-z]:[\\/]|[\\/]|~|\.\.(?:[\\/]|$))", tok)) or "/" in tok or "\\" in tok
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class CapabilityBroker:
|
|
77
|
+
def __init__(self, workspace, audit_path, policy=None, approver=None):
|
|
78
|
+
self.workspace = Path(workspace).resolve()
|
|
79
|
+
self.audit_path = Path(audit_path)
|
|
80
|
+
self.audit_path.parent.mkdir(parents=True, exist_ok=True)
|
|
81
|
+
p = dict(DEFAULT_POLICY)
|
|
82
|
+
p.update(policy or {})
|
|
83
|
+
self.policy = p
|
|
84
|
+
self.approver = approver # callable(request_dict) -> bool, for interactive approval
|
|
85
|
+
self.grants = [] # (action, pattern, expires)
|
|
86
|
+
self.created = set() # files the agent created during this mission (resolved paths)
|
|
87
|
+
self._lock = threading.Lock()
|
|
88
|
+
|
|
89
|
+
def grant(self, action, pattern, ttl=3600):
|
|
90
|
+
self.grants.append((action, pattern, time.time() + ttl))
|
|
91
|
+
|
|
92
|
+
def _roots(self, key):
|
|
93
|
+
return [Path(r.replace("{workspace}", str(self.workspace))).expanduser().resolve() for r in self.policy[key]]
|
|
94
|
+
|
|
95
|
+
def path(self, target):
|
|
96
|
+
t = str(target)
|
|
97
|
+
if os.name != "nt":
|
|
98
|
+
if re.match(r"^[A-Za-z]:[\\/]", t): # a Windows drive path is never inside a POSIX workspace
|
|
99
|
+
return Path("/__windows_drive__") / t[0]
|
|
100
|
+
t = t.replace("\\", "/") # "..\\secret" must not become a harmless filename
|
|
101
|
+
p = Path(t).expanduser()
|
|
102
|
+
return (p if p.is_absolute() else self.workspace / p).resolve()
|
|
103
|
+
|
|
104
|
+
def inside(self, target, key="fs_read"):
|
|
105
|
+
try:
|
|
106
|
+
rp = self.path(target)
|
|
107
|
+
except (OSError, ValueError):
|
|
108
|
+
return False
|
|
109
|
+
return any(rp.is_relative_to(r) for r in self._roots(key))
|
|
110
|
+
|
|
111
|
+
def _protected(self, rp):
|
|
112
|
+
return any(rp.is_relative_to((self.workspace / d).resolve()) for d in self.policy["fs_protected"])
|
|
113
|
+
|
|
114
|
+
# ------------------------------------------------------------------ exec
|
|
115
|
+
def _eval_exec(self, cmd):
|
|
116
|
+
try:
|
|
117
|
+
argv = shlex.split(cmd, posix=os.name != "nt")
|
|
118
|
+
except ValueError as e:
|
|
119
|
+
return False, f"unparseable command: {e}"
|
|
120
|
+
if not argv:
|
|
121
|
+
return False, "empty command"
|
|
122
|
+
prog = Path(argv[0]).name.lower()
|
|
123
|
+
prog = prog[:-4] if prog.endswith(".exe") else prog
|
|
124
|
+
user_ok = any(fnmatch.fnmatch(cmd, pat) for pat in self.policy["exec_allow"])
|
|
125
|
+
if prog not in EXEC_RULES and not user_ok:
|
|
126
|
+
return False, f"program '{prog}' not allow-listed"
|
|
127
|
+
rule = EXEC_RULES.get(prog)
|
|
128
|
+
args = argv[1:]
|
|
129
|
+
if isinstance(rule, dict): # python -m pytest|unittest
|
|
130
|
+
if len(args) < 2 or args[0] not in rule or args[1] not in rule[args[0]]:
|
|
131
|
+
if not user_ok:
|
|
132
|
+
return False, f"only '{prog} -m pytest|unittest' allowed"
|
|
133
|
+
elif isinstance(rule, set): # git sub-commands
|
|
134
|
+
if not args or args[0] not in rule:
|
|
135
|
+
if not user_ok:
|
|
136
|
+
return False, f"'{prog} {args[0] if args else ''}' not allowed"
|
|
137
|
+
for a in args:
|
|
138
|
+
key = a.split("=", 1)[0]
|
|
139
|
+
if key in BANNED_ARGS:
|
|
140
|
+
return False, f"argument '{key}' not allowed"
|
|
141
|
+
if _is_pathlike(a) and not self.inside(a.split("=", 1)[-1] if "=" in a else a):
|
|
142
|
+
return False, f"argument '{a}' points outside the workspace"
|
|
143
|
+
new_code = sorted(str(p.relative_to(self.workspace)) for p in self.created
|
|
144
|
+
if p.suffix in (".py", ".pyc", ".pth", ".so", ".dll") and p.exists())
|
|
145
|
+
if new_code:
|
|
146
|
+
mode = self.policy["exec_new_code"]
|
|
147
|
+
if mode == "deny":
|
|
148
|
+
return False, f"workspace contains code created by the agent: {new_code}"
|
|
149
|
+
if mode == "approve":
|
|
150
|
+
return False, f"needs approval: would execute code created by the agent: {new_code}"
|
|
151
|
+
return True, "command allowed (argv rules)"
|
|
152
|
+
|
|
153
|
+
# ------------------------------------------------------------------ evaluate
|
|
154
|
+
def _evaluate(self, action, target):
|
|
155
|
+
for a, pat, exp in self.grants:
|
|
156
|
+
if a == action and exp > time.time() and fnmatch.fnmatch(target, pat):
|
|
157
|
+
return True, "temporary grant", exp
|
|
158
|
+
if action == "read" and "://" in str(target):
|
|
159
|
+
u = urlparse(target)
|
|
160
|
+
if u.scheme.lower() not in NET_SCHEMES:
|
|
161
|
+
return False, f"scheme '{u.scheme}' not allowed (http/https only)", None
|
|
162
|
+
host = u.hostname or ""
|
|
163
|
+
if not host:
|
|
164
|
+
return False, "url without host", None
|
|
165
|
+
if not any(fnmatch.fnmatch(host, d) for d in self.policy["net_domains"]):
|
|
166
|
+
return False, f"domain {host} not allowed", None
|
|
167
|
+
if self.policy["block_private_net"] and host not in self.policy.get("private_allow", []):
|
|
168
|
+
try:
|
|
169
|
+
for info in socket.getaddrinfo(host, None):
|
|
170
|
+
ip = ipaddress.ip_address(info[4][0])
|
|
171
|
+
if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_reserved:
|
|
172
|
+
return False, "private/loopback address blocked", None
|
|
173
|
+
except socket.gaierror:
|
|
174
|
+
pass
|
|
175
|
+
return True, "domain allowed", None
|
|
176
|
+
if action in ("read", "write"):
|
|
177
|
+
if "://" in str(target):
|
|
178
|
+
return False, "urls cannot be written", None
|
|
179
|
+
try:
|
|
180
|
+
rp = self.path(target)
|
|
181
|
+
except (OSError, ValueError) as e:
|
|
182
|
+
return False, f"bad path: {e}", None
|
|
183
|
+
ok = any(rp.is_relative_to(r) for r in self._roots("fs_read" if action == "read" else "fs_write"))
|
|
184
|
+
if not ok:
|
|
185
|
+
return False, "path outside allowed roots", None
|
|
186
|
+
if action == "write" and self._protected(rp):
|
|
187
|
+
return False, "runtime state (.agent/.git) is not writable by the agent", None
|
|
188
|
+
return True, "inside allowed roots", None
|
|
189
|
+
if action == "execute":
|
|
190
|
+
ok, why = self._eval_exec(target)
|
|
191
|
+
return ok, why, None
|
|
192
|
+
if action == "mcp":
|
|
193
|
+
if any(fnmatch.fnmatch(target, p) for p in self.policy.get("mcp_allow", [])):
|
|
194
|
+
return True, "mcp tool allow-listed", None
|
|
195
|
+
return False, "needs approval: mcp tool not in mcp_allow", None
|
|
196
|
+
return False, "unknown action", None
|
|
197
|
+
|
|
198
|
+
def check(self, action, target):
|
|
199
|
+
"""Dry-run evaluation (no audit, no approval prompt)."""
|
|
200
|
+
return self._evaluate(action, target)[:2]
|
|
201
|
+
|
|
202
|
+
def note_created(self, target):
|
|
203
|
+
rp = self.path(target)
|
|
204
|
+
self.created.add(rp)
|
|
205
|
+
|
|
206
|
+
def require(self, who, mission, step, tool, action, target, why=""):
|
|
207
|
+
ok, reason, exp = self._evaluate(action, target)
|
|
208
|
+
if not ok and self.approver and not reason.startswith(("scheme", "runtime state")):
|
|
209
|
+
req = dict(who=who, tool=tool, action=action, target=target, why=why, reason=reason)
|
|
210
|
+
if self.approver(req):
|
|
211
|
+
ok, reason = True, "approved interactively"
|
|
212
|
+
self.audit(who, mission, step, tool, action, target, why, ok, reason, exp)
|
|
213
|
+
if not ok:
|
|
214
|
+
raise PermissionDenied(f"{action} {target!r} denied: {reason}")
|
|
215
|
+
return {"allowed": ok, "reason": reason}
|
|
216
|
+
|
|
217
|
+
def audit(self, who, mission, step, tool, action, target, why, allowed, reason, exp=None):
|
|
218
|
+
rec = {"ts": time.time(), "mission": mission, "step": step, "who": who, "tool": tool, "action": action,
|
|
219
|
+
"target": target, "why": why, "allowed": allowed, "reason": reason, "expires": exp}
|
|
220
|
+
with self._lock, open(self.audit_path, "a", encoding="utf-8") as f:
|
|
221
|
+
f.write(json.dumps(rec) + "\n")
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import os
|
|
3
|
+
import time
|
|
4
|
+
import uuid
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from .models import RunState
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class CheckpointStore:
|
|
10
|
+
def __init__(self, root):
|
|
11
|
+
self.root = Path(root)
|
|
12
|
+
|
|
13
|
+
def dir(self, mission_id):
|
|
14
|
+
d = self.root / mission_id
|
|
15
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
16
|
+
return d
|
|
17
|
+
|
|
18
|
+
def save(self, state: RunState):
|
|
19
|
+
d = self.dir(state.mission.id)
|
|
20
|
+
tmp = d / f"state.json.{uuid.uuid4().hex[:6]}.tmp"
|
|
21
|
+
tmp.write_text(json.dumps(state.to_dict(), indent=1, default=str), encoding="utf-8")
|
|
22
|
+
for attempt in range(8): # atomic; retried because Windows AV/indexers briefly lock files
|
|
23
|
+
try:
|
|
24
|
+
os.replace(tmp, d / "state.json")
|
|
25
|
+
return
|
|
26
|
+
except PermissionError:
|
|
27
|
+
time.sleep(0.05 * 2 ** attempt)
|
|
28
|
+
os.replace(tmp, d / "state.json")
|
|
29
|
+
|
|
30
|
+
def load(self, mission_id) -> RunState:
|
|
31
|
+
return RunState.from_dict(json.loads((self.root / mission_id / "state.json").read_text(encoding="utf-8")))
|
|
32
|
+
|
|
33
|
+
def list(self):
|
|
34
|
+
return sorted(p.parent.name for p in self.root.glob("*/state.json")) if self.root.exists() else []
|
agent_runtime/cli.py
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import json
|
|
3
|
+
import subprocess
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from .runtime import Runtime
|
|
7
|
+
from .scaffold import create_tool, publish_tool
|
|
8
|
+
from .tools import trust_plugin
|
|
9
|
+
from . import config as cfg
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _rt(a, interactive=False):
|
|
13
|
+
_, rt = cfg.load(a.workspace)
|
|
14
|
+
yes = getattr(a, "yes", False)
|
|
15
|
+
if yes:
|
|
16
|
+
print(" ! --yes: every action the policy denies will be approved automatically", file=sys.stderr)
|
|
17
|
+
appr = cfg.make_approver(rt["approve"], interactive=interactive and sys.stdin.isatty(), yes=yes)
|
|
18
|
+
return Runtime(a.workspace, approver=appr, max_replans=getattr(a, "max_replans", 2),
|
|
19
|
+
deadline=getattr(a, "deadline", None), on_event=lambda m: print(" ·", m))
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
EXIT = {"passed": 0, "failed": 1, "partial": 1, "unverified": 1, "refused": 3} # scriptable in CI
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _show_mission(state, store):
|
|
26
|
+
m = state.mission
|
|
27
|
+
print(f"mission {m.id} [{state.status}] kind={m.kind} replans={state.replans}")
|
|
28
|
+
print(f" text: {m.text}")
|
|
29
|
+
print(" contract:")
|
|
30
|
+
res = {r["id"]: r for r in (state.report or {}).get("results", [])}
|
|
31
|
+
for c in m.criteria:
|
|
32
|
+
r = res.get(c.id)
|
|
33
|
+
mark = "·" if r is None else ("✓" if r["passed"] else "✗")
|
|
34
|
+
print(f" [{mark}] {c.id:6} {c.check:16} {c.description}" + (f" — {r['detail'][:90]}" if r else ""))
|
|
35
|
+
print(" plan:")
|
|
36
|
+
for s in state.plan.steps:
|
|
37
|
+
ev = state.evidence.get(s.id, {})
|
|
38
|
+
print(f" {s.id:10} {s.tool:18} {s.status:10} items={len(ev.get('items', []))}"
|
|
39
|
+
+ (f" error: {s.error[:80]}" if s.error else ""))
|
|
40
|
+
print(" log:")
|
|
41
|
+
for e in state.log[-12:]:
|
|
42
|
+
print(f" - {e['msg'][:140]}")
|
|
43
|
+
art = store.dir(m.id) / "artifact.json"
|
|
44
|
+
if art.exists():
|
|
45
|
+
import json as _j
|
|
46
|
+
st = _j.loads(art.read_text(encoding="utf-8")).get("stats", {})
|
|
47
|
+
print(f" stats: {st}")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _show(state):
|
|
51
|
+
print(f"\nmission {state.mission.id}: {state.status.upper()} (replans: {state.replans})")
|
|
52
|
+
for r in (state.report or {}).get("results", []):
|
|
53
|
+
print(f" [{'✓' if r['passed'] else '✗'}] {r['description']} — {r['detail']}")
|
|
54
|
+
if state.status != "passed" and state.log:
|
|
55
|
+
print(" reason:", state.log[-1]["msg"])
|
|
56
|
+
if state.status in ("partial", "unverified"):
|
|
57
|
+
print(" note: NOT a pass — only verified evidence was written")
|
|
58
|
+
print(f" artifact: .agent/missions/{state.mission.id}/artifact.json")
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def main(argv=None):
|
|
62
|
+
p = argparse.ArgumentParser(prog="agent", description="Give your agent a mission, not a prompt.")
|
|
63
|
+
p.add_argument("--workspace", default=".")
|
|
64
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
65
|
+
r = sub.add_parser("run"); r.add_argument("mission"); r.add_argument("--yes", action="store_true", help="auto-approve ungranted actions"); r.add_argument("--max-replans", type=int, default=2)
|
|
66
|
+
r.add_argument("--deadline", type=float, default=None, help="wall-clock budget in seconds; on expiry an honest partial result is written")
|
|
67
|
+
s = sub.add_parser("resume"); s.add_argument("id"); s.add_argument("--yes", action="store_true"); s.add_argument("--deadline", type=float, default=None)
|
|
68
|
+
po = sub.add_parser("policy", help="show the effective policy (or --init to write .agent/policy.json)")
|
|
69
|
+
po.add_argument("--init", action="store_true")
|
|
70
|
+
sh = sub.add_parser("show", help="contract, plan, evidence and log of one mission"); sh.add_argument("id")
|
|
71
|
+
tr = sub.add_parser("trust", help="allow a plugin in ./tools/<name> to load (pins its tool.py hash)"); tr.add_argument("name")
|
|
72
|
+
st = sub.add_parser("status"); st.add_argument("id", nargs="?")
|
|
73
|
+
au = sub.add_parser("audit"); au.add_argument("id")
|
|
74
|
+
me = sub.add_parser("memory"); me.add_argument("query", nargs="?", default="")
|
|
75
|
+
cr = sub.add_parser("create"); cr.add_argument("what", choices=["tool"]); cr.add_argument("name")
|
|
76
|
+
sub.add_parser("test")
|
|
77
|
+
pu = sub.add_parser("publish"); pu.add_argument("name")
|
|
78
|
+
a = p.parse_args(argv)
|
|
79
|
+
try:
|
|
80
|
+
if not (a.cmd == "policy" and a.init):
|
|
81
|
+
cfg.load(a.workspace) # a broken policy.json is reported by every command
|
|
82
|
+
return _dispatch(a)
|
|
83
|
+
except cfg.ConfigError as e:
|
|
84
|
+
print(f"config error: {e}", file=sys.stderr)
|
|
85
|
+
return 2
|
|
86
|
+
except BrokenPipeError: # `agent show <id> | head` — exit quietly
|
|
87
|
+
import os
|
|
88
|
+
os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stdout.fileno())
|
|
89
|
+
return 0
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _dispatch(a):
|
|
93
|
+
|
|
94
|
+
if a.cmd in ("run", "resume"):
|
|
95
|
+
rt = _rt(a, True)
|
|
96
|
+
st = rt.start(a.mission) if a.cmd == "run" else rt.resume(a.id)
|
|
97
|
+
_show(st)
|
|
98
|
+
return EXIT.get(st.status, 1)
|
|
99
|
+
elif a.cmd == "status":
|
|
100
|
+
rt = Runtime(a.workspace, llm=None, judge=None, tools={}, config={})
|
|
101
|
+
for mid in ([a.id] if a.id else rt.store.list()):
|
|
102
|
+
s = rt.store.load(mid)
|
|
103
|
+
print(f"{mid} {s.status:10} {s.mission.text[:70]}")
|
|
104
|
+
elif a.cmd == "audit":
|
|
105
|
+
for line in (Path(a.workspace) / ".agent/missions" / a.id / "audit.jsonl").read_text().splitlines():
|
|
106
|
+
e = json.loads(line)
|
|
107
|
+
print(f"{'ALLOW' if e['allowed'] else 'DENY '} {e['who']:8} {e['tool']:16} {e['action']:8} {e['target'][:60]} ← {e['why']} ({e['reason']})")
|
|
108
|
+
elif a.cmd == "memory":
|
|
109
|
+
for m in Runtime(a.workspace, llm=None, judge=None, tools={}, config={}).memory.recall(a.query):
|
|
110
|
+
print(f"[{m['kind']}/{m['status']} c={m['confidence']}] {m['content']} (src: {m['provenance']})")
|
|
111
|
+
elif a.cmd == "policy":
|
|
112
|
+
if a.init:
|
|
113
|
+
print("wrote", cfg.init(a.workspace))
|
|
114
|
+
else:
|
|
115
|
+
print(json.dumps(cfg.effective(a.workspace), indent=2))
|
|
116
|
+
elif a.cmd == "show":
|
|
117
|
+
rt = Runtime(a.workspace, llm=None, judge=None, tools={}, config={})
|
|
118
|
+
_show_mission(rt.store.load(a.id), rt.store)
|
|
119
|
+
elif a.cmd == "trust":
|
|
120
|
+
name, h = trust_plugin(a.workspace, a.name)
|
|
121
|
+
print(f"trusted {name} (sha256 {h[:12]}…) — re-run trust after editing its tool.py")
|
|
122
|
+
elif a.cmd == "create":
|
|
123
|
+
print("created", create_tool(a.name, Path(a.workspace) / "tools"))
|
|
124
|
+
elif a.cmd == "test":
|
|
125
|
+
sys.exit(max((subprocess.call([sys.executable, "-m", "unittest", "discover", "-s", "tests"], cwd=d)
|
|
126
|
+
for d in (Path(a.workspace) / "tools").glob("*/")), default=0))
|
|
127
|
+
elif a.cmd == "publish":
|
|
128
|
+
print("packed", publish_tool(a.name, Path(a.workspace) / "tools", Path(a.workspace) / "dist"))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
if __name__ == "__main__":
|
|
132
|
+
sys.exit(main())
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Mission compiler: free text -> objective + constraints + machine-checkable success criteria."""
|
|
2
|
+
import json
|
|
3
|
+
import re
|
|
4
|
+
from .llm import extract_json
|
|
5
|
+
from .models import Mission, Criterion, CHECKS
|
|
6
|
+
from . import contract
|
|
7
|
+
|
|
8
|
+
RESEARCH = ("find", "research", "list", "compare", "investigate", "who ", "which ", "companies", "sources")
|
|
9
|
+
CODING = ("implement", "fix", "refactor", "bug", "function", "write code", "unit test", "repo")
|
|
10
|
+
SYSTEM = ("You compile a user's mission into JSON: {\"objective\": str, \"kind\": \"research|coding|general\", "
|
|
11
|
+
"\"constraints\": {}, \"criteria\": [{\"id\": str, \"description\": str, \"check\": one of "
|
|
12
|
+
"min_items|unique_items|all_sourced|file_exists|command_ok|contains|llm, \"params\": {}}]}. "
|
|
13
|
+
"params: min_items {n}, file_exists {path}, command_ok {command}, contains {path,text}, llm {question}. "
|
|
14
|
+
"Criteria must be objectively verifiable. Reply with JSON only.")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _heuristic(text, mid):
|
|
18
|
+
low = text.lower()
|
|
19
|
+
kind = "coding" if any(k in low for k in CODING) else "research" if any(k in low for k in RESEARCH) else "general"
|
|
20
|
+
crit, cid = [], iter(range(1, 99))
|
|
21
|
+
n = re.search(r"\b(\d{1,5})\s+(?:\w+\s+){0,2}?[a-z]{3,}", text)
|
|
22
|
+
if kind == "research":
|
|
23
|
+
crit.append(Criterion(f"c{next(cid)}", f"at least {n.group(1) if n else 1} item(s)", "min_items", {"n": int(n.group(1)) if n else 1}))
|
|
24
|
+
crit.append(Criterion(f"c{next(cid)}", "no duplicate items", "unique_items"))
|
|
25
|
+
crit.append(Criterion(f"c{next(cid)}", "every item cites a source that was actually retrieved", "all_sourced"))
|
|
26
|
+
f = re.search(r"(?:to|into|as|in)\s+([\w./-]+\.(?:md|txt|json|csv|py|html))\b", text)
|
|
27
|
+
if f:
|
|
28
|
+
crit.append(Criterion(f"c{next(cid)}", f"file {f.group(1)} exists", "file_exists", {"path": f.group(1)}))
|
|
29
|
+
if kind == "coding" and "test" in low:
|
|
30
|
+
crit.append(Criterion(f"c{next(cid)}", "test suite passes", "command_ok", {"command": "python -m pytest -q"}))
|
|
31
|
+
return Mission(mid, text, text.strip().rstrip("."), kind, {}, crit)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def valid_criteria(crits):
|
|
35
|
+
"""Reject criteria that can never be satisfied (a URL is not a file path) and drop duplicates."""
|
|
36
|
+
seen, out = set(), []
|
|
37
|
+
for c in crits:
|
|
38
|
+
p = c.params or {}
|
|
39
|
+
if c.check in ("file_exists", "contains"):
|
|
40
|
+
path = str(p.get("path", ""))
|
|
41
|
+
if not path or "://" in path or path.startswith("mailto:"):
|
|
42
|
+
continue
|
|
43
|
+
elif c.check == "min_items":
|
|
44
|
+
try:
|
|
45
|
+
if int(p.get("n", 0)) <= 0:
|
|
46
|
+
continue
|
|
47
|
+
except (TypeError, ValueError):
|
|
48
|
+
continue
|
|
49
|
+
elif c.check == "command_ok" and not str(p.get("command", "")).strip():
|
|
50
|
+
continue
|
|
51
|
+
elif c.check == "llm" and not str(p.get("question", "")).strip() and not c.description:
|
|
52
|
+
continue
|
|
53
|
+
key = (c.check, json.dumps(p, sort_keys=True, default=str))
|
|
54
|
+
if key in seen:
|
|
55
|
+
continue
|
|
56
|
+
seen.add(key)
|
|
57
|
+
out.append(c)
|
|
58
|
+
return out
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def compile_mission(text, llm=None, memory_hints="", workspace=None):
|
|
62
|
+
"""LLM (or rules) proposes criteria; the contract floor then guarantees the minimum the text demands."""
|
|
63
|
+
mid = Mission.new_id()
|
|
64
|
+
base = None
|
|
65
|
+
if llm:
|
|
66
|
+
try:
|
|
67
|
+
d = extract_json(llm.complete(SYSTEM, f"Mission: {text}\n{memory_hints}"))
|
|
68
|
+
crit = valid_criteria([Criterion(str(c["id"]), str(c.get("description", c["check"])), c["check"],
|
|
69
|
+
c.get("params") or {})
|
|
70
|
+
for c in d["criteria"] if isinstance(c, dict) and c.get("check") in CHECKS])
|
|
71
|
+
base = Mission(mid, text, str(d.get("objective") or text), str(d.get("kind") or "general"),
|
|
72
|
+
d.get("constraints") if isinstance(d.get("constraints"), dict) else {}, crit)
|
|
73
|
+
except Exception:
|
|
74
|
+
base = None # fall back to rules
|
|
75
|
+
if base is None:
|
|
76
|
+
base = _heuristic(text, mid)
|
|
77
|
+
kind, fl = contract.floor(text, workspace, base.kind)
|
|
78
|
+
base.kind = kind if kind != "general" else base.kind
|
|
79
|
+
base.criteria = contract.merge(base.criteria, fl)
|
|
80
|
+
base.constraints = dict(base.constraints or {})
|
|
81
|
+
base.constraints["verifiable"] = contract.verifiable(base.criteria)
|
|
82
|
+
return base
|
agent_runtime/config.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""User configuration: `.agent/policy.json` in the workspace (the agent itself can never write `.agent/`).
|
|
2
|
+
|
|
3
|
+
{
|
|
4
|
+
"net_domains": ["*"], # broker keys (see capabilities.DEFAULT_POLICY)
|
|
5
|
+
"exec_new_code": "approve",
|
|
6
|
+
"approve": [{"action": "execute", "pattern": "python -m pytest*"}], # auto-approved when policy denies
|
|
7
|
+
"search": {"provider": "duckduckgo"}, # duckduckgo | searxng {url} | brave {api_key_env}
|
|
8
|
+
"sandbox": {"mode": "limits", "cpu_seconds": 120, "memory_mb": 1024}, # none | limits | docker {image}
|
|
9
|
+
"mcp": {"files": {"command": ["npx", "-y", "@modelcontextprotocol/server-filesystem", "."]}}
|
|
10
|
+
}
|
|
11
|
+
"""
|
|
12
|
+
import fnmatch
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from .capabilities import DEFAULT_POLICY
|
|
17
|
+
|
|
18
|
+
SCHEMA = 2
|
|
19
|
+
RUNTIME_KEYS = {"approve", "search", "sandbox", "mcp"}
|
|
20
|
+
DEFAULTS = {
|
|
21
|
+
"approve": [],
|
|
22
|
+
"search": {"provider": "duckduckgo"},
|
|
23
|
+
"sandbox": {"mode": "limits", "cpu_seconds": 300, "memory_mb": 2048, "image": "python:3.12-slim"},
|
|
24
|
+
"mcp": {},
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class ConfigError(ValueError):
|
|
29
|
+
pass
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def policy_path(workspace):
|
|
33
|
+
return Path(workspace) / ".agent" / "policy.json"
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def load(workspace):
|
|
37
|
+
"""Return (broker_policy, runtime_config). Unknown keys are an error, not silently ignored."""
|
|
38
|
+
f = policy_path(workspace)
|
|
39
|
+
raw = {}
|
|
40
|
+
if f.exists():
|
|
41
|
+
try:
|
|
42
|
+
raw = json.loads(f.read_text(encoding="utf-8"))
|
|
43
|
+
except ValueError as e:
|
|
44
|
+
raise ConfigError(f"{f}: invalid JSON ({e})") from e
|
|
45
|
+
if not isinstance(raw, dict):
|
|
46
|
+
raise ConfigError(f"{f}: must be a JSON object")
|
|
47
|
+
unknown = set(raw) - set(DEFAULT_POLICY) - RUNTIME_KEYS
|
|
48
|
+
if unknown:
|
|
49
|
+
raise ConfigError(f"{f}: unknown key(s) {sorted(unknown)}; allowed: {sorted(set(DEFAULT_POLICY) | RUNTIME_KEYS)}")
|
|
50
|
+
broker = {k: v for k, v in raw.items() if k in DEFAULT_POLICY}
|
|
51
|
+
rt = {k: (dict(DEFAULTS[k], **raw[k]) if isinstance(DEFAULTS[k], dict) and isinstance(raw.get(k), dict)
|
|
52
|
+
else raw.get(k, DEFAULTS[k])) for k in RUNTIME_KEYS}
|
|
53
|
+
if rt["sandbox"].get("mode") not in ("none", "limits", "docker"):
|
|
54
|
+
raise ConfigError("sandbox.mode must be none | limits | docker")
|
|
55
|
+
if rt["search"].get("provider") not in ("duckduckgo", "searxng", "brave"):
|
|
56
|
+
raise ConfigError("search.provider must be duckduckgo | searxng | brave")
|
|
57
|
+
for rule in rt["approve"]:
|
|
58
|
+
if not isinstance(rule, dict) or not {"action", "pattern"} <= set(rule):
|
|
59
|
+
raise ConfigError("approve rules look like {\"action\": \"execute\", \"pattern\": \"python -m pytest*\"}")
|
|
60
|
+
return broker, rt
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def effective(workspace):
|
|
64
|
+
broker, rt = load(workspace)
|
|
65
|
+
out = dict(DEFAULT_POLICY)
|
|
66
|
+
out.update(broker)
|
|
67
|
+
out.update(rt)
|
|
68
|
+
return out
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def init(workspace):
|
|
72
|
+
f = policy_path(workspace)
|
|
73
|
+
if f.exists():
|
|
74
|
+
raise ConfigError(f"{f} already exists")
|
|
75
|
+
f.parent.mkdir(parents=True, exist_ok=True)
|
|
76
|
+
body = {k: v for k, v in DEFAULT_POLICY.items()}
|
|
77
|
+
body.update({k: v for k, v in DEFAULTS.items()})
|
|
78
|
+
f.write_text(json.dumps(body, indent=2), encoding="utf-8")
|
|
79
|
+
return f
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def make_approver(rules=(), interactive=False, yes=False, ask=None, out=sys.stderr):
|
|
83
|
+
"""Approval chain: --yes > policy rules > interactive prompt (y / N / a = always this session) > deny."""
|
|
84
|
+
session = []
|
|
85
|
+
|
|
86
|
+
def approve(req):
|
|
87
|
+
target, action = str(req.get("target", "")), req.get("action")
|
|
88
|
+
if yes:
|
|
89
|
+
return True
|
|
90
|
+
for r in list(rules) + session:
|
|
91
|
+
if r["action"] == action and fnmatch.fnmatch(target, r["pattern"]):
|
|
92
|
+
return True
|
|
93
|
+
if not interactive:
|
|
94
|
+
return False
|
|
95
|
+
prompt = (f"\n[approval needed] {req.get('tool')} wants to {action} {target!r}\n"
|
|
96
|
+
f" why: {req.get('why') or '-'}\n policy said: {req.get('reason', '-')}\n allow? [y]es / [N]o / [a]lways this session: ")
|
|
97
|
+
try:
|
|
98
|
+
ans = (ask or input)(prompt).strip().lower()
|
|
99
|
+
except EOFError:
|
|
100
|
+
return False
|
|
101
|
+
if ans == "a":
|
|
102
|
+
session.append({"action": action, "pattern": target})
|
|
103
|
+
return True
|
|
104
|
+
return ans in ("y", "yes")
|
|
105
|
+
return approve
|