handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""Filesystem reconciliation probe.
|
|
2
|
+
|
|
3
|
+
Covers the second common non-idempotent effect: appending to a file.
|
|
4
|
+
|
|
5
|
+
Method:
|
|
6
|
+
|
|
7
|
+
capture() the file's size and content hash, plus the bytes to be appended
|
|
8
|
+
probe() unchanged -> DID_NOT_LAND
|
|
9
|
+
prefix intact + our bytes -> LANDED (proof, not inference)
|
|
10
|
+
anything else -> INCONCLUSIVE
|
|
11
|
+
|
|
12
|
+
The middle case is the good one. Because the tool's arguments carry the content
|
|
13
|
+
being appended, we can verify *our* bytes are present rather than merely
|
|
14
|
+
noticing the file changed. That is stronger than the git probe, which can only
|
|
15
|
+
infer from "HEAD moved".
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import hashlib
|
|
20
|
+
import json
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from .base import DID_NOT_LAND, INCONCLUSIVE, LANDED
|
|
24
|
+
|
|
25
|
+
_PATH_KEYS = ("path", "file_path", "filename", "file", "target")
|
|
26
|
+
_CONTENT_KEYS = ("content", "text", "data", "payload", "new_str", "append")
|
|
27
|
+
_MAX_BYTES = 32 * 1024 * 1024 # do not hash a huge file on the hot path
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class FileAppendProbe:
|
|
31
|
+
name = "filesystem"
|
|
32
|
+
|
|
33
|
+
def __init__(self, max_bytes: int = _MAX_BYTES):
|
|
34
|
+
self.max_bytes = max_bytes
|
|
35
|
+
|
|
36
|
+
# ── selection ──────────────────────────────────────────────────────
|
|
37
|
+
def handles(self, call) -> bool:
|
|
38
|
+
return self._path(call) is not None
|
|
39
|
+
|
|
40
|
+
# ── before ─────────────────────────────────────────────────────────
|
|
41
|
+
def capture(self, call) -> dict | None:
|
|
42
|
+
try:
|
|
43
|
+
path = self._path(call)
|
|
44
|
+
if path is None:
|
|
45
|
+
return None
|
|
46
|
+
pre = self._fingerprint(path)
|
|
47
|
+
if pre is None:
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
out = {"probe": self.name, "path": str(path), **pre}
|
|
51
|
+
|
|
52
|
+
# If we can see what will be appended, record it so the probe can
|
|
53
|
+
# verify prefix + suffix rather than guessing.
|
|
54
|
+
content = self._content(call)
|
|
55
|
+
if content is not None:
|
|
56
|
+
out["appended"] = content.decode("utf-8", "replace")
|
|
57
|
+
return out
|
|
58
|
+
except Exception: # noqa: BLE001
|
|
59
|
+
return None
|
|
60
|
+
|
|
61
|
+
# ── after a crash ──────────────────────────────────────────────────
|
|
62
|
+
def probe(self, call, record) -> str:
|
|
63
|
+
try:
|
|
64
|
+
pre = _pre(record)
|
|
65
|
+
if not pre or pre.get("probe") != self.name:
|
|
66
|
+
return INCONCLUSIVE
|
|
67
|
+
|
|
68
|
+
path = Path(pre["path"])
|
|
69
|
+
now = self._fingerprint(path)
|
|
70
|
+
if now is None:
|
|
71
|
+
return INCONCLUSIVE
|
|
72
|
+
|
|
73
|
+
if now["sha256"] == pre.get("sha256") and now["size"] == pre.get("size"):
|
|
74
|
+
return DID_NOT_LAND
|
|
75
|
+
|
|
76
|
+
appended = pre.get("appended")
|
|
77
|
+
if appended is not None:
|
|
78
|
+
return self._verify_append(path, pre, appended)
|
|
79
|
+
|
|
80
|
+
# Changed, and we could not predict the post-state. Under the
|
|
81
|
+
# single-writer assumption this was our effect, but it is inference.
|
|
82
|
+
return LANDED
|
|
83
|
+
except Exception: # noqa: BLE001
|
|
84
|
+
return INCONCLUSIVE
|
|
85
|
+
|
|
86
|
+
# ── internals ──────────────────────────────────────────────────────
|
|
87
|
+
@staticmethod
|
|
88
|
+
def _path(call) -> Path | None:
|
|
89
|
+
for k in _PATH_KEYS:
|
|
90
|
+
if (v := call.args.get(k)):
|
|
91
|
+
try:
|
|
92
|
+
p = Path(str(v))
|
|
93
|
+
except Exception: # noqa: BLE001
|
|
94
|
+
continue
|
|
95
|
+
if p.parent.exists():
|
|
96
|
+
return p
|
|
97
|
+
return None
|
|
98
|
+
|
|
99
|
+
@staticmethod
|
|
100
|
+
def _content(call) -> bytes | None:
|
|
101
|
+
for k in _CONTENT_KEYS:
|
|
102
|
+
v = call.args.get(k)
|
|
103
|
+
if isinstance(v, (str, bytes)):
|
|
104
|
+
return v.encode() if isinstance(v, str) else v
|
|
105
|
+
return None
|
|
106
|
+
|
|
107
|
+
def _verify_append(self, path: Path, pre: dict, appended: str) -> str:
|
|
108
|
+
"""Prefix intact + our content at the tail means it landed.
|
|
109
|
+
|
|
110
|
+
Deliberately not a whole-file hash. Text-mode writers translate LF to
|
|
111
|
+
CRLF on Windows, so a predicted hash of the exact bytes almost never
|
|
112
|
+
matches and every append would fail closed. Comparing the unchanged
|
|
113
|
+
prefix against a newline-normalised tail is both more robust and a
|
|
114
|
+
stronger claim: it verifies *our* bytes are there, not merely that the
|
|
115
|
+
file changed.
|
|
116
|
+
"""
|
|
117
|
+
pre_size = int(pre.get("size") or 0)
|
|
118
|
+
try:
|
|
119
|
+
data = path.read_bytes()
|
|
120
|
+
except OSError:
|
|
121
|
+
return INCONCLUSIVE
|
|
122
|
+
|
|
123
|
+
if len(data) < pre_size:
|
|
124
|
+
return INCONCLUSIVE # truncated: not our doing
|
|
125
|
+
if _sha(data[:pre_size]) != pre.get("sha256"):
|
|
126
|
+
return INCONCLUSIVE # prefix rewritten
|
|
127
|
+
|
|
128
|
+
tail = _norm(data[pre_size:])
|
|
129
|
+
ours = _norm(appended.encode())
|
|
130
|
+
if tail == ours or ours in tail:
|
|
131
|
+
return LANDED
|
|
132
|
+
return INCONCLUSIVE # someone else wrote
|
|
133
|
+
|
|
134
|
+
def _fingerprint(self, path: Path) -> dict | None:
|
|
135
|
+
try:
|
|
136
|
+
if not path.exists():
|
|
137
|
+
return {"exists": False, "size": 0, "sha256": _sha(b"")}
|
|
138
|
+
size = path.stat().st_size
|
|
139
|
+
if size > self.max_bytes:
|
|
140
|
+
return {"exists": True, "size": size, "sha256": None}
|
|
141
|
+
return {"exists": True, "size": size, "sha256": _sha(path.read_bytes())}
|
|
142
|
+
except OSError:
|
|
143
|
+
return None
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _sha(b: bytes) -> str:
|
|
147
|
+
return hashlib.sha256(b).hexdigest()
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _norm(b: bytes) -> bytes:
|
|
151
|
+
"""Normalise line endings so CRLF and LF compare equal."""
|
|
152
|
+
return b.replace(b"\r\n", b"\n").replace(b"\r", b"\n")
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _pre(record) -> dict | None:
|
|
156
|
+
raw = getattr(record, "pre_state", None)
|
|
157
|
+
if not raw:
|
|
158
|
+
return None
|
|
159
|
+
try:
|
|
160
|
+
return json.loads(raw)
|
|
161
|
+
except Exception: # noqa: BLE001
|
|
162
|
+
return None
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""Git reconciliation probe.
|
|
2
|
+
|
|
3
|
+
The headline case: an agent runs `git commit`, the process dies before the
|
|
4
|
+
observation is recorded, and on resume we must decide whether to run it again.
|
|
5
|
+
|
|
6
|
+
Method — fingerprint, not marker:
|
|
7
|
+
|
|
8
|
+
capture() HEAD sha + a hash of `git status --porcelain`
|
|
9
|
+
probe() HEAD moved? -> LANDED
|
|
10
|
+
HEAD same, tree same? -> DID_NOT_LAND
|
|
11
|
+
anything else -> INCONCLUSIVE
|
|
12
|
+
|
|
13
|
+
No command mutation, so this works at Seam B (`docs/0016` §4).
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
import shutil
|
|
19
|
+
import subprocess
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
from .base import DID_NOT_LAND, INCONCLUSIVE, LANDED
|
|
23
|
+
|
|
24
|
+
#: Commands that move HEAD. `git push` is deliberately absent — it is EXTERNAL,
|
|
25
|
+
#: and a local probe cannot see whether the remote accepted it.
|
|
26
|
+
_MOVES_HEAD = ("commit", "merge", "cherry-pick", "revert", "am", "rebase")
|
|
27
|
+
|
|
28
|
+
NO_COMMITS = "__NO_COMMITS__"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class GitProbe:
|
|
32
|
+
name = "git"
|
|
33
|
+
|
|
34
|
+
def __init__(self, repo_root: str | Path | None = None, timeout_s: float = 10.0):
|
|
35
|
+
self.repo_root = Path(repo_root) if repo_root else None
|
|
36
|
+
self.timeout_s = timeout_s
|
|
37
|
+
|
|
38
|
+
# ── selection ──────────────────────────────────────────────────────
|
|
39
|
+
def handles(self, call) -> bool:
|
|
40
|
+
if shutil.which("git") is None:
|
|
41
|
+
return False
|
|
42
|
+
# A dedicated tool ("commit", "git_commit") carries no command string
|
|
43
|
+
# to sniff, so match on the tool name too.
|
|
44
|
+
name = (call.tool_name or "").lower()
|
|
45
|
+
if any(verb in name for verb in _MOVES_HEAD):
|
|
46
|
+
return True
|
|
47
|
+
text = self._command_text(call)
|
|
48
|
+
if "git " not in text:
|
|
49
|
+
return False
|
|
50
|
+
return any(f"git {verb}" in text for verb in _MOVES_HEAD)
|
|
51
|
+
|
|
52
|
+
# ── before ─────────────────────────────────────────────────────────
|
|
53
|
+
def capture(self, call) -> dict | None:
|
|
54
|
+
try:
|
|
55
|
+
root = self._root(call)
|
|
56
|
+
if root is None:
|
|
57
|
+
return None
|
|
58
|
+
head = self._head(root)
|
|
59
|
+
if head is None:
|
|
60
|
+
return None # not a git repo
|
|
61
|
+
return {
|
|
62
|
+
"probe": self.name,
|
|
63
|
+
"root": str(root),
|
|
64
|
+
# The repo we actually attached to. A workspace nested inside a
|
|
65
|
+
# larger repo silently resolves to the ANCESTOR, whose HEAD
|
|
66
|
+
# moves for reasons that have nothing to do with this agent.
|
|
67
|
+
"toplevel": self._git(root, "rev-parse", "--show-toplevel"),
|
|
68
|
+
"head": head,
|
|
69
|
+
"tree": self._tree_hash(root),
|
|
70
|
+
}
|
|
71
|
+
except Exception: # noqa: BLE001
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
# ── after a crash ──────────────────────────────────────────────────
|
|
75
|
+
def probe(self, call, record) -> str:
|
|
76
|
+
try:
|
|
77
|
+
pre = _pre(record)
|
|
78
|
+
if not pre or pre.get("probe") != self.name:
|
|
79
|
+
return INCONCLUSIVE
|
|
80
|
+
|
|
81
|
+
root = Path(pre["root"])
|
|
82
|
+
head_now = self._head(root)
|
|
83
|
+
if head_now is None:
|
|
84
|
+
return INCONCLUSIVE
|
|
85
|
+
|
|
86
|
+
# A different repo than the one we fingerprinted: say nothing.
|
|
87
|
+
if self._git(root, "rev-parse", "--show-toplevel") != pre.get("toplevel"):
|
|
88
|
+
return INCONCLUSIVE
|
|
89
|
+
|
|
90
|
+
if head_now != pre.get("head"):
|
|
91
|
+
# HEAD moved during the crash window. With a single writer in
|
|
92
|
+
# the workspace, that was our commit.
|
|
93
|
+
return LANDED
|
|
94
|
+
|
|
95
|
+
tree_now = self._tree_hash(root)
|
|
96
|
+
if tree_now is not None and tree_now == pre.get("tree"):
|
|
97
|
+
# HEAD unchanged AND the working tree is exactly as we left it:
|
|
98
|
+
# nothing was committed and nothing else happened either.
|
|
99
|
+
return DID_NOT_LAND
|
|
100
|
+
|
|
101
|
+
# HEAD unchanged but the tree moved. Something happened that was
|
|
102
|
+
# not a commit; we cannot attribute it. Do not guess.
|
|
103
|
+
return INCONCLUSIVE
|
|
104
|
+
except Exception: # noqa: BLE001
|
|
105
|
+
return INCONCLUSIVE
|
|
106
|
+
|
|
107
|
+
# ── internals ──────────────────────────────────────────────────────
|
|
108
|
+
def _root(self, call) -> Path | None:
|
|
109
|
+
"""Configured root wins over anything the model supplied.
|
|
110
|
+
|
|
111
|
+
A model that can choose the path can choose the blast radius, and the
|
|
112
|
+
probe would then faithfully fingerprint the wrong world. Argument keys
|
|
113
|
+
are a fallback for tools that genuinely take a path, never an override.
|
|
114
|
+
"""
|
|
115
|
+
if self.repo_root is not None:
|
|
116
|
+
return self.repo_root
|
|
117
|
+
for key in ("cwd", "working_dir", "path", "repo"):
|
|
118
|
+
if (v := call.args.get(key)):
|
|
119
|
+
p = Path(str(v))
|
|
120
|
+
if p.is_dir():
|
|
121
|
+
return p
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
def _git(self, root: Path, *args: str) -> str | None:
|
|
125
|
+
try:
|
|
126
|
+
r = subprocess.run(
|
|
127
|
+
["git", *args], cwd=str(root), capture_output=True,
|
|
128
|
+
text=True, timeout=self.timeout_s,
|
|
129
|
+
)
|
|
130
|
+
except Exception: # noqa: BLE001
|
|
131
|
+
return None
|
|
132
|
+
return r.stdout.strip() if r.returncode == 0 else None
|
|
133
|
+
|
|
134
|
+
def _head(self, root: Path) -> str | None:
|
|
135
|
+
if self._git(root, "rev-parse", "--is-inside-work-tree") != "true":
|
|
136
|
+
return None
|
|
137
|
+
head = self._git(root, "rev-parse", "HEAD")
|
|
138
|
+
# A repo with no commits yet has no HEAD; that is a valid state.
|
|
139
|
+
return head if head else NO_COMMITS
|
|
140
|
+
|
|
141
|
+
def _tree_hash(self, root: Path) -> str | None:
|
|
142
|
+
status = self._git(root, "status", "--porcelain=v1", "--untracked-files=all")
|
|
143
|
+
if status is None:
|
|
144
|
+
return None
|
|
145
|
+
return hashlib.sha256(status.encode()).hexdigest()
|
|
146
|
+
|
|
147
|
+
@staticmethod
|
|
148
|
+
def _command_text(call) -> str:
|
|
149
|
+
if "command" in call.args:
|
|
150
|
+
return str(call.args["command"])
|
|
151
|
+
return " ".join(str(v) for v in call.args.values())
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _pre(record) -> dict | None:
|
|
155
|
+
import json
|
|
156
|
+
raw = getattr(record, "pre_state", None)
|
|
157
|
+
if not raw:
|
|
158
|
+
return None
|
|
159
|
+
try:
|
|
160
|
+
return json.loads(raw)
|
|
161
|
+
except Exception: # noqa: BLE001
|
|
162
|
+
return None
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Runtime: real tools and a runner, for doing actual work.
|
|
2
|
+
|
|
3
|
+
`experiments/` proves the system; this uses it.
|
|
4
|
+
|
|
5
|
+
The names below load on first use. Importing them eagerly meant that any
|
|
6
|
+
`agentctl.runtime.*` import -- `lease`, `config`, `init` -- pulled in the tools
|
|
7
|
+
and with them pydantic and the OpenHands SDK. So the control plane's `proxy
|
|
8
|
+
down` could not even check a pid without the agent framework (`docs/0047`).
|
|
9
|
+
"""
|
|
10
|
+
__all__ = ["run", "TOOLS", "register_all"]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def __getattr__(name: str):
|
|
14
|
+
if name == "run":
|
|
15
|
+
from .runner import run
|
|
16
|
+
return run
|
|
17
|
+
if name in ("TOOLS", "register_all"):
|
|
18
|
+
from . import tools
|
|
19
|
+
return getattr(tools, name)
|
|
20
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
r"""Check a scout's citation before believing it.
|
|
2
|
+
|
|
3
|
+
`docs/0039` §7, fifteenth entry. The first live fan-out returned:
|
|
4
|
+
|
|
5
|
+
"The lease TTL is enforced in hook.py ... 900.0 seconds ...
|
|
6
|
+
cited at hook.py:13"
|
|
7
|
+
|
|
8
|
+
The lease TTL is `store.py:63`, 60.0 seconds. `hook.py`'s 900.0 is the
|
|
9
|
+
turn-affinity pin, a different timer — and `hook.py:13` is prose in a
|
|
10
|
+
docstring about `tool_call_id`. Nothing malfunctioned: the scout read eight
|
|
11
|
+
files, found *a* `ttl_s`, and reported it as *the* one, with a citation, in
|
|
12
|
+
the format its prompt demanded.
|
|
13
|
+
|
|
14
|
+
**Citation discipline did not help, because a citation is only as good as the
|
|
15
|
+
check nobody performed.** This is that check.
|
|
16
|
+
|
|
17
|
+
## What can and cannot be verified here
|
|
18
|
+
|
|
19
|
+
Mechanically certain:
|
|
20
|
+
|
|
21
|
+
* the file exists in the workspace,
|
|
22
|
+
* the line number is within it,
|
|
23
|
+
* what that line actually says.
|
|
24
|
+
|
|
25
|
+
Mechanically useful, and what caught the live failure: **does the value the
|
|
26
|
+
claim asserts appear where the claim says it does?** `900.0` is 81 lines from
|
|
27
|
+
`hook.py:13`, and saying so is enough to stop a reader believing it.
|
|
28
|
+
|
|
29
|
+
Not verifiable here, and not attempted: whether the claim is *true*. A line
|
|
30
|
+
can exist, contain the token, and still not support the sentence wrapped
|
|
31
|
+
around it. This narrows what must be read by hand; it does not replace
|
|
32
|
+
reading.
|
|
33
|
+
|
|
34
|
+
**A citation that fails these checks is not a false claim, and a citation
|
|
35
|
+
that passes is not a true one.** The output says which check ran.
|
|
36
|
+
"""
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import re
|
|
40
|
+
from dataclasses import dataclass, field
|
|
41
|
+
from pathlib import Path
|
|
42
|
+
|
|
43
|
+
#: `store.py:63`, `agentctl/kernel/hook.py:94`, with optional backticks.
|
|
44
|
+
_CITE = re.compile(r"`?([\w./\\-]+\.[A-Za-z]\w*):(\d+)`?")
|
|
45
|
+
|
|
46
|
+
#: Tokens worth looking for at the cited line: decimals, backticked spans,
|
|
47
|
+
#: and identifiers distinctive enough that finding one means something.
|
|
48
|
+
#: Bare short words are deliberately excluded — "the" appearing near a line
|
|
49
|
+
#: would corroborate nothing.
|
|
50
|
+
_NUMBER = re.compile(r"\b\d+\.\d+\b|\b\d{2,}\b")
|
|
51
|
+
_BACKTICK = re.compile(r"`([^`\n]{2,40})`")
|
|
52
|
+
_IDENT = re.compile(r"\b([A-Za-z_][A-Za-z0-9_]{4,})\b")
|
|
53
|
+
|
|
54
|
+
#: How far from the cited line a token still counts as "there". Small on
|
|
55
|
+
#: purpose: a citation that is 80 lines out is the failure being caught.
|
|
56
|
+
WINDOW = 3
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class Citation:
|
|
61
|
+
path: str
|
|
62
|
+
line: int
|
|
63
|
+
resolved: Path | None = None
|
|
64
|
+
text: str | None = None
|
|
65
|
+
problem: str | None = None
|
|
66
|
+
#: (token, line where it actually is, or None if nowhere in the file)
|
|
67
|
+
misplaced: list[tuple[str, int | None]] = field(default_factory=list)
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def ok(self) -> bool:
|
|
71
|
+
return self.problem is None and not self.misplaced
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _tokens(context: str) -> list[str]:
|
|
75
|
+
"""Distinctive things the claim asserts, worth finding at the line."""
|
|
76
|
+
out: list[str] = []
|
|
77
|
+
for m in _BACKTICK.findall(context):
|
|
78
|
+
m = m.strip()
|
|
79
|
+
# A backticked path:line is the citation itself, not a claim about it.
|
|
80
|
+
if not _CITE.fullmatch(m) and len(m) >= 2:
|
|
81
|
+
out.append(m)
|
|
82
|
+
out += _NUMBER.findall(context)
|
|
83
|
+
out += [t for t in _IDENT.findall(context)
|
|
84
|
+
if "_" in t or not t.islower()]
|
|
85
|
+
# Stable order, no duplicates, and nothing that is just the filename.
|
|
86
|
+
seen, keep = set(), []
|
|
87
|
+
for t in out:
|
|
88
|
+
low = t.lower()
|
|
89
|
+
if low in seen or low.endswith((".py", ".md", ".yaml")):
|
|
90
|
+
continue
|
|
91
|
+
seen.add(low)
|
|
92
|
+
keep.append(t)
|
|
93
|
+
return keep[:6]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def find(report: str) -> list[tuple[str, int, str]]:
|
|
97
|
+
"""Every `path:line` in the report, with the sentence around it."""
|
|
98
|
+
out = []
|
|
99
|
+
for m in _CITE.finditer(report):
|
|
100
|
+
start = max(0, m.start() - 220)
|
|
101
|
+
end = min(len(report), m.end() + 80)
|
|
102
|
+
out.append((m.group(1), int(m.group(2)), report[start:end]))
|
|
103
|
+
return out
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def check(path: str, line: int, context: str, workspace: str | Path,
|
|
107
|
+
window: int = WINDOW) -> Citation:
|
|
108
|
+
"""Resolve one citation and test the claim's tokens against it."""
|
|
109
|
+
c = Citation(path=path, line=line)
|
|
110
|
+
ws = Path(workspace).resolve()
|
|
111
|
+
|
|
112
|
+
candidate = (ws / path).resolve()
|
|
113
|
+
if not str(candidate).startswith(str(ws)):
|
|
114
|
+
c.problem = "cites a path outside the workspace"
|
|
115
|
+
return c
|
|
116
|
+
if not candidate.is_file():
|
|
117
|
+
hits = list(ws.rglob(Path(path).name))
|
|
118
|
+
c.problem = ("no such file in the workspace"
|
|
119
|
+
+ (f"; closest match {hits[0].relative_to(ws)}"
|
|
120
|
+
if len(hits) == 1 else ""))
|
|
121
|
+
return c
|
|
122
|
+
c.resolved = candidate
|
|
123
|
+
|
|
124
|
+
try:
|
|
125
|
+
lines = candidate.read_text(encoding="utf-8",
|
|
126
|
+
errors="replace").splitlines()
|
|
127
|
+
except Exception as e: # noqa: BLE001
|
|
128
|
+
c.problem = f"unreadable: {type(e).__name__}"
|
|
129
|
+
return c
|
|
130
|
+
|
|
131
|
+
if not 1 <= line <= len(lines):
|
|
132
|
+
c.problem = f"line {line} is past the end of the file ({len(lines)})"
|
|
133
|
+
return c
|
|
134
|
+
c.text = lines[line - 1].strip()
|
|
135
|
+
|
|
136
|
+
lo, hi = max(0, line - 1 - window), min(len(lines), line + window)
|
|
137
|
+
near = "\n".join(lines[lo:hi]).lower()
|
|
138
|
+
# Strip the citations themselves first. `store.py:63` otherwise yields the
|
|
139
|
+
# token `63`, and "63 is not at line 63" flags a correct report -- a
|
|
140
|
+
# verifier that cries wolf on good citations trains you to ignore it,
|
|
141
|
+
# which is worse than not checking at all.
|
|
142
|
+
for tok in _tokens(_CITE.sub(" ", context)):
|
|
143
|
+
if tok.lower() in near:
|
|
144
|
+
continue
|
|
145
|
+
where = next((i + 1 for i, l in enumerate(lines)
|
|
146
|
+
if tok.lower() in l.lower()), None)
|
|
147
|
+
c.misplaced.append((tok, where))
|
|
148
|
+
return c
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def verify(report: str, workspace: str | Path,
|
|
152
|
+
window: int = WINDOW) -> list[Citation]:
|
|
153
|
+
"""Every citation in a scout's report, checked."""
|
|
154
|
+
return [check(p, n, ctx, workspace, window) for p, n, ctx in find(report)]
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def describe(cites: list[Citation]) -> str:
|
|
158
|
+
"""What a reader needs to decide whether to believe the report."""
|
|
159
|
+
if not cites:
|
|
160
|
+
return (" no citation to check -- this report asserts things it does "
|
|
161
|
+
"not point at, so none of it has been verified.")
|
|
162
|
+
out = []
|
|
163
|
+
for c in cites:
|
|
164
|
+
if c.problem:
|
|
165
|
+
out.append(f" !! {c.path}:{c.line} {c.problem}")
|
|
166
|
+
continue
|
|
167
|
+
if c.misplaced:
|
|
168
|
+
out.append(f" !! {c.path}:{c.line} says: {(c.text or '')[:56]}")
|
|
169
|
+
for tok, where in c.misplaced:
|
|
170
|
+
loc = f"line {where}" if where else "nowhere in this file"
|
|
171
|
+
out.append(f" {tok!r} is not there -- it is at {loc}")
|
|
172
|
+
continue
|
|
173
|
+
out.append(f" ok {c.path}:{c.line} {(c.text or '')[:60]}")
|
|
174
|
+
bad = sum(1 for c in cites if not c.ok)
|
|
175
|
+
if bad:
|
|
176
|
+
out.append(f" {bad} of {len(cites)} citation(s) did not check "
|
|
177
|
+
f"out. A citation that fails is not a false claim -- it is "
|
|
178
|
+
f"one nobody can take on trust.")
|
|
179
|
+
return "\n".join(out)
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Where `agentctl run` gets its defaults. `docs/0043` Phase 2.
|
|
2
|
+
|
|
3
|
+
Phase 0 (`docs/0044`) needed `--workspace`, `--model`, `--base-url` and
|
|
4
|
+
`--ledger` on almost every command, and the built-in model was one OpenRouter
|
|
5
|
+
id: a user holding only a Gemini or Anthropic key was handed a default that
|
|
6
|
+
could not work.
|
|
7
|
+
|
|
8
|
+
Precedence, highest first:
|
|
9
|
+
|
|
10
|
+
a flag on the command line
|
|
11
|
+
an environment variable AGENTCTL_MODEL, AGENTCTL_BASE_URL
|
|
12
|
+
~/.agentctl/config.toml written by `agentctl init`, editable
|
|
13
|
+
derived the default model of the first provider you
|
|
14
|
+
hold a key for, free tiers first
|
|
15
|
+
|
|
16
|
+
`resolve()` reports which of these each value came from, because a default
|
|
17
|
+
nobody can trace is the confidently-wrong shape `docs/0039` is about.
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import os
|
|
22
|
+
import tomllib
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
#: Keys this file owns. Anything else in config.toml is kept but not read.
|
|
27
|
+
KEYS = ("model", "base_url")
|
|
28
|
+
ENV = {"model": "AGENTCTL_MODEL", "base_url": "AGENTCTL_BASE_URL"}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def path() -> Path:
|
|
32
|
+
"""`~/.agentctl/config.toml`, or `AGENTCTL_CONFIG` when set."""
|
|
33
|
+
if (p := os.environ.get("AGENTCTL_CONFIG")):
|
|
34
|
+
return Path(p)
|
|
35
|
+
return Path.home() / ".agentctl" / "config.toml"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def load(p: Path | None = None) -> dict:
|
|
39
|
+
p = p or path()
|
|
40
|
+
if not p.exists():
|
|
41
|
+
return {}
|
|
42
|
+
try:
|
|
43
|
+
return tomllib.loads(p.read_text(encoding="utf-8"))
|
|
44
|
+
except tomllib.TOMLDecodeError as e:
|
|
45
|
+
# A broken config must not silently fall back to a different model.
|
|
46
|
+
raise SystemExit(f"{p} is not valid TOML: {e}\n"
|
|
47
|
+
f" fix it, or delete it and run `agentctl init`")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def write(values: dict, p: Path | None = None, note: str = "") -> Path:
|
|
51
|
+
"""Write the keys this module owns. Values are plain strings."""
|
|
52
|
+
p = p or path()
|
|
53
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
54
|
+
lines = ["# Written by `agentctl init`. Edit freely; a flag or an",
|
|
55
|
+
"# AGENTCTL_* environment variable still wins over anything here."]
|
|
56
|
+
if note:
|
|
57
|
+
lines += [f"# {line}" for line in note.splitlines()]
|
|
58
|
+
lines.append("")
|
|
59
|
+
for k in KEYS:
|
|
60
|
+
if values.get(k):
|
|
61
|
+
v = str(values[k]).replace("\\", "\\\\").replace('"', '\\"')
|
|
62
|
+
lines.append(f'{k} = "{v}"')
|
|
63
|
+
p.write_text("\n".join(lines) + "\n", encoding="utf-8")
|
|
64
|
+
return p
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def derived_model() -> str | None:
|
|
68
|
+
"""The default model of the first provider holding a key, free first.
|
|
69
|
+
|
|
70
|
+
The registry's first model per provider is the one `init` would try
|
|
71
|
+
first. Without `init` it is unverified -- `resolve` says so.
|
|
72
|
+
"""
|
|
73
|
+
from agentctl.control.providers import configured
|
|
74
|
+
|
|
75
|
+
for p in configured():
|
|
76
|
+
if p.default_model:
|
|
77
|
+
return p.default_model
|
|
78
|
+
return None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(frozen=True)
|
|
82
|
+
class Setting:
|
|
83
|
+
value: str | None
|
|
84
|
+
source: str # "flag" | "env AGENTCTL_MODEL" | "config" | ...
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def resolve(name: str, flag: str | None, cfg: dict | None = None) -> Setting:
|
|
88
|
+
if flag:
|
|
89
|
+
return Setting(flag, "flag")
|
|
90
|
+
if (v := os.environ.get(ENV[name])):
|
|
91
|
+
return Setting(v, f"env {ENV[name]}")
|
|
92
|
+
cfg = load() if cfg is None else cfg
|
|
93
|
+
if cfg.get(name):
|
|
94
|
+
return Setting(str(cfg[name]), f"config {path()}")
|
|
95
|
+
if name == "model" and (m := derived_model()):
|
|
96
|
+
return Setting(m, "derived from your keys (unverified: `agentctl init` checks it)")
|
|
97
|
+
return Setting(None, "unset")
|