handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
r"""Fan a read-only recon job out across sources, by scarcity.
|
|
2
|
+
|
|
3
|
+
`docs/0040` §6.1: **allocate work inversely to quota scarcity, not evenly.**
|
|
4
|
+
That single scheduling choice is the difference between ×1.55 and ×4.33 more
|
|
5
|
+
recon tasks per day on the owner's pool, with no extra quota bought. An even
|
|
6
|
+
split puts the same load on a leg with one allowance as on a leg with six, and
|
|
7
|
+
the one-allowance leg then decides the throughput of the whole job.
|
|
8
|
+
|
|
9
|
+
## Why this allocates by quota count and not by rate limit
|
|
10
|
+
|
|
11
|
+
The obvious allocator divides work by each source's requests-per-day. This one
|
|
12
|
+
cannot, and the reason is a rule rather than an oversight:
|
|
13
|
+
`control/providers.py` deliberately records no rate limits, because they go
|
|
14
|
+
stale within weeks and a confidently wrong number is worse than none.
|
|
15
|
+
|
|
16
|
+
So the allocator uses the one fact the registry *will* vouch for — how many
|
|
17
|
+
**independent quotas** a source has (`providers.quotas_for`) — and splits in
|
|
18
|
+
proportion to that. It is structural, it does not rot, and on the measured
|
|
19
|
+
pool it recovers most of the available gain:
|
|
20
|
+
|
|
21
|
+
Gemini 1 quota : Mistral 6 quotas -> 12 files split 2 / 10
|
|
22
|
+
×3.61 of the ×4.33 ceiling
|
|
23
|
+
|
|
24
|
+
Reaching the last 20% needs each source's real RPD, which is exactly the
|
|
25
|
+
number this project refuses to hardcode. `ratio=` accepts one if the caller
|
|
26
|
+
has measured it today.
|
|
27
|
+
|
|
28
|
+
## Sequential, deliberately
|
|
29
|
+
|
|
30
|
+
`docs/0039` §5 records that subagent concurrency is unproven: the two globals
|
|
31
|
+
that raced are fixed and three further candidates were refuted, but nothing
|
|
32
|
+
exercises the concurrent path. `research/phase-10-5`'s build order puts
|
|
33
|
+
sequential fan-out before concurrent for that reason. Concurrency is a change
|
|
34
|
+
to this file only, once something tests it.
|
|
35
|
+
"""
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
from dataclasses import dataclass, field
|
|
39
|
+
from typing import Any, Callable
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Leg:
|
|
44
|
+
"""One source's share of the job, and what came back."""
|
|
45
|
+
source: str
|
|
46
|
+
quotas: int
|
|
47
|
+
items: list[str] = field(default_factory=list)
|
|
48
|
+
report: str | None = None
|
|
49
|
+
error: str | None = None
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def requests(self) -> int:
|
|
53
|
+
"""One request per item, plus the report it ends with."""
|
|
54
|
+
return len(self.items) + 1 if self.items else 0
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def weights(sources: list[str], ratio: dict[str, int] | None = None,
|
|
58
|
+
env: dict | None = None) -> dict[str, int]:
|
|
59
|
+
"""How much work each source should carry, relatively.
|
|
60
|
+
|
|
61
|
+
Defaults to its independent quota count. `ratio` overrides per source, for
|
|
62
|
+
a caller who has measured real limits and wants to use them — pass the
|
|
63
|
+
numbers, not a promise that they are current.
|
|
64
|
+
"""
|
|
65
|
+
from agentctl.control.providers import BY_NAME, quotas_for
|
|
66
|
+
|
|
67
|
+
out: dict[str, int] = {}
|
|
68
|
+
for name in sources:
|
|
69
|
+
if ratio and name in ratio:
|
|
70
|
+
out[name] = max(int(ratio[name]), 0)
|
|
71
|
+
continue
|
|
72
|
+
p = BY_NAME.get(name)
|
|
73
|
+
out[name] = quotas_for(p, env) if p is not None else 1
|
|
74
|
+
return out
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def allocate(items: list[str], sources: list[str],
|
|
78
|
+
ratio: dict[str, int] | None = None,
|
|
79
|
+
env: dict | None = None) -> list[Leg]:
|
|
80
|
+
"""Split `items` across `sources` in proportion to their quotas.
|
|
81
|
+
|
|
82
|
+
Largest-remainder, so the split is exact rather than drifting by rounding,
|
|
83
|
+
and deterministic — the same inputs always give the same plan, which is
|
|
84
|
+
what makes a fan-out reproducible enough to compare two runs.
|
|
85
|
+
|
|
86
|
+
A source with no quota gets no work. A source that *has* quota gets at
|
|
87
|
+
least one item, because a leg with zero items still costs a report request
|
|
88
|
+
and returns nothing: paying one request to learn nothing is strictly worse
|
|
89
|
+
than not dispatching it, so it is dropped instead.
|
|
90
|
+
"""
|
|
91
|
+
w = weights(sources, ratio, env)
|
|
92
|
+
live = [s for s in sources if w.get(s, 0) > 0]
|
|
93
|
+
if not items or not live:
|
|
94
|
+
return [Leg(s, w.get(s, 0)) for s in sources]
|
|
95
|
+
|
|
96
|
+
# More legs than items would leave some carrying nothing. Keep the
|
|
97
|
+
# highest-weighted, so a scarce leg is dropped before a plentiful one.
|
|
98
|
+
if len(live) > len(items):
|
|
99
|
+
live = sorted(live, key=lambda s: (-w[s], s))[:len(items)]
|
|
100
|
+
|
|
101
|
+
total = sum(w[s] for s in live)
|
|
102
|
+
exact = {s: len(items) * w[s] / total for s in live}
|
|
103
|
+
base = {s: max(int(exact[s]), 1) for s in live}
|
|
104
|
+
|
|
105
|
+
# Largest remainder, then trim from the most plentiful leg if the
|
|
106
|
+
# one-item floor overshot.
|
|
107
|
+
short = len(items) - sum(base.values())
|
|
108
|
+
order = sorted(live, key=lambda s: (-(exact[s] - int(exact[s])), s))
|
|
109
|
+
i = 0
|
|
110
|
+
while short > 0:
|
|
111
|
+
base[order[i % len(order)]] += 1
|
|
112
|
+
short -= 1
|
|
113
|
+
i += 1
|
|
114
|
+
while short < 0:
|
|
115
|
+
victim = max((s for s in live if base[s] > 1),
|
|
116
|
+
key=lambda s: (w[s], s), default=None)
|
|
117
|
+
if victim is None:
|
|
118
|
+
break
|
|
119
|
+
base[victim] -= 1
|
|
120
|
+
short += 1
|
|
121
|
+
|
|
122
|
+
legs, cut = [], 0
|
|
123
|
+
for s in sources:
|
|
124
|
+
n = base.get(s, 0)
|
|
125
|
+
legs.append(Leg(s, w.get(s, 0), items[cut:cut + n]))
|
|
126
|
+
cut += n
|
|
127
|
+
return legs
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def describe(legs: list[Leg]) -> str:
|
|
131
|
+
"""The plan, and why it is uneven. Printed before anything is spent."""
|
|
132
|
+
live = [lg for lg in legs if lg.items]
|
|
133
|
+
if not live:
|
|
134
|
+
return " nothing to dispatch"
|
|
135
|
+
out = []
|
|
136
|
+
for lg in legs:
|
|
137
|
+
if not lg.items:
|
|
138
|
+
why = "no quota" if not lg.quotas else "no share at this size"
|
|
139
|
+
out.append(f" -- {lg.source:<14} {why}")
|
|
140
|
+
continue
|
|
141
|
+
out.append(f" -> {lg.source:<14} {len(lg.items):>2} item(s), "
|
|
142
|
+
f"{lg.requests:>2} request(s) "
|
|
143
|
+
f"{lg.quotas} quota(s)")
|
|
144
|
+
scarce = min(live, key=lambda lg: lg.quotas)
|
|
145
|
+
rich = max(live, key=lambda lg: lg.quotas)
|
|
146
|
+
if scarce.quotas != rich.quotas:
|
|
147
|
+
# ASCII only in printed output. Non-ASCII in a rendered string has
|
|
148
|
+
# broken this project on a cp437 console five times; the dashboard
|
|
149
|
+
# carries a test for exactly this.
|
|
150
|
+
out.append(f" {scarce.source} carries less because it has "
|
|
151
|
+
f"{scarce.quotas} quota(s) to {rich.source}'s {rich.quotas} "
|
|
152
|
+
f"-- an even split would let it set the pace for all of "
|
|
153
|
+
f"them (docs/0040 sec 6.1).")
|
|
154
|
+
return "\n".join(out)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def fan_out(definition: Any, question: str, legs: list[Leg], *,
|
|
158
|
+
workspace: str = ".", base_url: str | None = None,
|
|
159
|
+
runner: Callable[..., str] | None = None,
|
|
160
|
+
on_leg: Callable[[Leg], None] | None = None) -> list[Leg]:
|
|
161
|
+
"""Run each leg's scout in turn, filling in `report` or `error`.
|
|
162
|
+
|
|
163
|
+
Sequential. One leg failing does not stop the others — a recon job that
|
|
164
|
+
loses a scout should return what the rest found, clearly short, rather
|
|
165
|
+
than nothing at all. The caller sees which legs are missing because
|
|
166
|
+
`error` is set, not because the report is quietly thinner.
|
|
167
|
+
"""
|
|
168
|
+
from agentctl.control.proxy import SOURCE_PREFIX
|
|
169
|
+
|
|
170
|
+
from .subagent import run as run_subagent
|
|
171
|
+
|
|
172
|
+
run = runner or run_subagent
|
|
173
|
+
for leg in legs:
|
|
174
|
+
if not leg.items:
|
|
175
|
+
continue
|
|
176
|
+
task = (f"{question}\n\nLook only at these, and report on them:\n"
|
|
177
|
+
+ "\n".join(f" - {i}" for i in leg.items))
|
|
178
|
+
try:
|
|
179
|
+
leg.report = run(
|
|
180
|
+
definition, task, workspace=workspace,
|
|
181
|
+
model=f"openai/{SOURCE_PREFIX}{leg.source}" if base_url else None,
|
|
182
|
+
base_url=base_url)
|
|
183
|
+
except Exception as e: # noqa: BLE001
|
|
184
|
+
leg.error = f"{type(e).__name__}: {e}"
|
|
185
|
+
if on_leg is not None:
|
|
186
|
+
on_leg(leg)
|
|
187
|
+
return legs
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
r"""Claude Code plugins, admitted one capability at a time.
|
|
2
|
+
|
|
3
|
+
`docs/0010` §9.1 gave plugins the only **Partial BUILD** in an otherwise
|
|
4
|
+
SKIP-heavy table, and §9.2 said exactly which slice: packaging is
|
|
5
|
+
harness-specific and not our job, but *"policy over what may be installed and
|
|
6
|
+
reached is not."* This is that slice, at its smallest useful size.
|
|
7
|
+
|
|
8
|
+
A Claude Code plugin directory can contribute six things:
|
|
9
|
+
|
|
10
|
+
agents/ subagent definitions
|
|
11
|
+
hooks/ event handlers
|
|
12
|
+
commands/ slash commands
|
|
13
|
+
skills/ agent skills
|
|
14
|
+
.mcp.json MCP servers
|
|
15
|
+
plugin.json the manifest
|
|
16
|
+
|
|
17
|
+
**Five of those six are refused here, and the refusal is the feature.** A
|
|
18
|
+
plugin is third-party content: `docs/0010` §8.2 records that MCP amplifies
|
|
19
|
+
the effect problem, and hooks and commands are arbitrary behaviour attached
|
|
20
|
+
to an agent loop that this project spends its whole correctness budget
|
|
21
|
+
guarding. Admitting them because they happened to be in the folder would be
|
|
22
|
+
the opposite of a broker.
|
|
23
|
+
|
|
24
|
+
So the rule is **default-deny with an itemised receipt**. Only read-only agent
|
|
25
|
+
definitions are admitted, every one re-validated by
|
|
26
|
+
`subagent.validate` rather than trusted for having arrived in a manifest, and
|
|
27
|
+
everything declined is counted and named. A broker that silently dropped what
|
|
28
|
+
it would not run would leave you believing you had installed something you
|
|
29
|
+
had not.
|
|
30
|
+
|
|
31
|
+
## What is still trusted, and should be said plainly
|
|
32
|
+
|
|
33
|
+
An admitted definition carries a third-party `system_prompt`, and that becomes
|
|
34
|
+
instructions to a model that can read your files. Read-only bounds the damage
|
|
35
|
+
to *reading* — it cannot write, shell out, or reach the network — and the
|
|
36
|
+
workspace is scoped per task, so it reads where you pointed it and nowhere
|
|
37
|
+
else. That is a real bound, not a complete one. Pin plugin versions and do not
|
|
38
|
+
install one you would not read.
|
|
39
|
+
"""
|
|
40
|
+
from __future__ import annotations
|
|
41
|
+
|
|
42
|
+
from pathlib import Path
|
|
43
|
+
from typing import Any
|
|
44
|
+
|
|
45
|
+
#: The one capability a plugin may contribute.
|
|
46
|
+
ADMITTED = "agents"
|
|
47
|
+
|
|
48
|
+
#: Everything else, with why. Named rather than silently skipped — the point of
|
|
49
|
+
#: a broker is that you can audit what it refused.
|
|
50
|
+
REFUSED: dict[str, str] = {
|
|
51
|
+
"mcp_config": "an MCP server is an arbitrary tool surface reached over a "
|
|
52
|
+
"pipe; nothing here can inspect what it would do "
|
|
53
|
+
"(`docs/0010` §8.2)",
|
|
54
|
+
"hooks": "a hook is arbitrary behaviour attached to the agent loop, which "
|
|
55
|
+
"is the thing the gate exists to guard",
|
|
56
|
+
"commands": "slash commands belong to a harness UI, not to a control "
|
|
57
|
+
"plane (`docs/0010` §9.1)",
|
|
58
|
+
"skills": "a skill injects instructions into the context window; the "
|
|
59
|
+
"harness owns that and does it better (`0007`)",
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def load(plugin_dir: str | Path) -> dict[str, Any]:
|
|
64
|
+
"""Inspect a plugin. Reports what would be admitted and what is refused.
|
|
65
|
+
|
|
66
|
+
Loads nothing into a live agent — this is the audit step. `agentctl
|
|
67
|
+
plugins` prints it, and a caller that wants the definitions takes
|
|
68
|
+
`admitted`.
|
|
69
|
+
"""
|
|
70
|
+
from openhands.sdk.plugin.format.claude_code import ClaudeCodePluginFormat
|
|
71
|
+
|
|
72
|
+
from .subagent import NotReadOnly, validate
|
|
73
|
+
|
|
74
|
+
d = Path(plugin_dir)
|
|
75
|
+
if not d.is_dir():
|
|
76
|
+
return {"path": str(d), "error": f"no such directory: {d}"}
|
|
77
|
+
|
|
78
|
+
fmt = ClaudeCodePluginFormat()
|
|
79
|
+
try:
|
|
80
|
+
manifest = fmt.load_manifest(d)
|
|
81
|
+
except Exception as e: # noqa: BLE001
|
|
82
|
+
return {"path": str(d), "error": f"unreadable manifest: {e}"}
|
|
83
|
+
|
|
84
|
+
admitted, rejected = [], []
|
|
85
|
+
try:
|
|
86
|
+
for defn in fmt.load_agents(d):
|
|
87
|
+
try:
|
|
88
|
+
validate(defn)
|
|
89
|
+
admitted.append(defn)
|
|
90
|
+
except NotReadOnly as e:
|
|
91
|
+
rejected.append((getattr(defn, "name", "<unnamed>"), str(e)))
|
|
92
|
+
except Exception as e: # noqa: BLE001
|
|
93
|
+
rejected.append(("<agents/>", f"could not be read: {e}"))
|
|
94
|
+
|
|
95
|
+
return {
|
|
96
|
+
"path": str(d),
|
|
97
|
+
"name": getattr(manifest, "name", d.name),
|
|
98
|
+
"version": getattr(manifest, "version", "?"),
|
|
99
|
+
"description": getattr(manifest, "description", ""),
|
|
100
|
+
"admitted": admitted,
|
|
101
|
+
"rejected": rejected,
|
|
102
|
+
"refused": _refused_counts(fmt, d),
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _refused_counts(fmt: Any, d: Path) -> list[tuple[str, int, str]]:
|
|
107
|
+
"""What the plugin carries that this project will not run.
|
|
108
|
+
|
|
109
|
+
Counted, not just listed. "declines MCP servers" is a policy statement;
|
|
110
|
+
"declines 3 MCP servers" is a fact about the thing in front of you, and
|
|
111
|
+
only the second tells you whether refusing it matters.
|
|
112
|
+
"""
|
|
113
|
+
out = []
|
|
114
|
+
for cap, why in REFUSED.items():
|
|
115
|
+
try:
|
|
116
|
+
loader = {
|
|
117
|
+
"mcp_config": fmt.load_mcp_config,
|
|
118
|
+
"hooks": fmt.load_hooks,
|
|
119
|
+
"commands": fmt.load_commands,
|
|
120
|
+
"skills": lambda p: [], # assembled by the base class' load()
|
|
121
|
+
}[cap]
|
|
122
|
+
got = loader(d)
|
|
123
|
+
except Exception: # noqa: BLE001
|
|
124
|
+
continue
|
|
125
|
+
if got is None:
|
|
126
|
+
continue
|
|
127
|
+
n = len(got) if hasattr(got, "__len__") else 1
|
|
128
|
+
if n:
|
|
129
|
+
out.append((cap, n, why))
|
|
130
|
+
return out
|
|
@@ -0,0 +1,361 @@
|
|
|
1
|
+
"""What a run did, in words a user can act on. `docs/0043` Phase 3.
|
|
2
|
+
|
|
3
|
+
Phase 0 (`docs/0044` F6, F10) ended every run with `decisions {'EXECUTE': 9}`
|
|
4
|
+
-- the gate's vocabulary, not the user's -- and nothing about whether the task
|
|
5
|
+
worked, what changed, or what it cost. The report answers four questions, and
|
|
6
|
+
derives every line from the same records the precise output uses, because a
|
|
7
|
+
friendly line computed some other way is how a tool starts saying confidently
|
|
8
|
+
wrong things (`docs/0039`):
|
|
9
|
+
|
|
10
|
+
outcome did it work? only what was CHECKED; never implied
|
|
11
|
+
changed what is different? from git, against the run's start
|
|
12
|
+
used what did it cost? requests, tokens, money -- labelled
|
|
13
|
+
needs you what is left? blocked effects, with the command
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import subprocess
|
|
18
|
+
import time
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
# ── the vocabulary (`docs/0043` §3, Phase 3) ──────────────────────────────
|
|
23
|
+
#: The gate's verdicts, as a user would say them.
|
|
24
|
+
VERDICT_WORDS = {
|
|
25
|
+
"EXECUTE": "ran",
|
|
26
|
+
"SUBSTITUTE": "already done; reused the recorded result",
|
|
27
|
+
"BLOCK": "paused; needs your decision",
|
|
28
|
+
"ESCALATE": "paused; needs your decision",
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
#: Effect classes, as kinds of action.
|
|
32
|
+
CLASS_WORDS = {
|
|
33
|
+
"PURE_READ": "read",
|
|
34
|
+
"IDEMPOTENT_WRITE": "file write",
|
|
35
|
+
"NON_IDEMPOTENT_WRITE": "command",
|
|
36
|
+
"EXTERNAL": "command",
|
|
37
|
+
"DESTRUCTIVE": "dangerous command",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def plural(n: int, word: str) -> str:
|
|
42
|
+
return f"{n} {word}{'' if n == 1 else 's'}"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# ── what changed ──────────────────────────────────────────────────────────
|
|
46
|
+
def _git(ws: Path, *args: str) -> str | None:
|
|
47
|
+
try:
|
|
48
|
+
r = subprocess.run(["git", *args], cwd=str(ws), capture_output=True,
|
|
49
|
+
text=True, encoding="utf-8", errors="replace",
|
|
50
|
+
timeout=30)
|
|
51
|
+
except Exception: # noqa: BLE001
|
|
52
|
+
return None
|
|
53
|
+
return r.stdout if r.returncode == 0 else None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _porcelain(ws: Path, prefix: str = "") -> set[str]:
|
|
57
|
+
"""Changed paths under `ws`, relative to it. `prefix` is `ws` relative to
|
|
58
|
+
the repository's top level: porcelain paths are always top-level ones."""
|
|
59
|
+
out = _git(ws, "status", "--porcelain=v1", "--untracked-files=all",
|
|
60
|
+
"--", ".") or ""
|
|
61
|
+
paths = set()
|
|
62
|
+
for line in out.splitlines():
|
|
63
|
+
path = line[3:].strip().strip('"')
|
|
64
|
+
if prefix and path.startswith(prefix):
|
|
65
|
+
path = path[len(prefix):]
|
|
66
|
+
if path and not path.startswith(".agentctl"):
|
|
67
|
+
paths.add(path)
|
|
68
|
+
return paths
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass(frozen=True)
|
|
72
|
+
class Start:
|
|
73
|
+
"""The workspace as the run found it."""
|
|
74
|
+
is_git: bool
|
|
75
|
+
head: str | None # None: a repo with no commits yet
|
|
76
|
+
dirty: frozenset[str] # paths already changed before the run
|
|
77
|
+
at: float
|
|
78
|
+
#: `ws` relative to the repository's top level, "" when it IS the top.
|
|
79
|
+
#: Non-empty means the workspace sits inside a bigger repository -- on
|
|
80
|
+
#: the machine this was built on, the home directory itself -- and every
|
|
81
|
+
#: git question must be scoped to the workspace, or the report describes
|
|
82
|
+
#: someone else's files (`docs/0048`).
|
|
83
|
+
prefix: str = ""
|
|
84
|
+
toplevel: str = ""
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def snapshot(ws: str | Path) -> Start:
|
|
88
|
+
ws = Path(ws)
|
|
89
|
+
if _git(ws, "rev-parse", "--is-inside-work-tree") is None:
|
|
90
|
+
return Start(False, None, frozenset(), time.time())
|
|
91
|
+
head = (_git(ws, "rev-parse", "HEAD") or "").strip() or None
|
|
92
|
+
prefix = (_git(ws, "rev-parse", "--show-prefix") or "").strip()
|
|
93
|
+
top = (_git(ws, "rev-parse", "--show-toplevel") or "").strip()
|
|
94
|
+
return Start(True, head, frozenset(_porcelain(ws, prefix)), time.time(),
|
|
95
|
+
prefix, top)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@dataclass
|
|
99
|
+
class Changes:
|
|
100
|
+
files: list[tuple[str, int | None, int | None]] = field(default_factory=list)
|
|
101
|
+
commits: int = 0
|
|
102
|
+
already_dirty: int = 0
|
|
103
|
+
note: str = ""
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def added(self) -> int:
|
|
107
|
+
return sum(a or 0 for _, a, _ in self.files)
|
|
108
|
+
|
|
109
|
+
@property
|
|
110
|
+
def removed(self) -> int:
|
|
111
|
+
return sum(d or 0 for _, _, d in self.files)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def changes(ws: str | Path, start: Start) -> Changes:
|
|
115
|
+
"""Files that differ from where the run started, commits included.
|
|
116
|
+
|
|
117
|
+
Measured against the starting commit, so work the agent committed counts
|
|
118
|
+
as well as work it left uncommitted. Files that were ALREADY modified
|
|
119
|
+
before the run are counted only if they differ now, and their number is
|
|
120
|
+
reported, because their diff mixes the user's edits with the agent's.
|
|
121
|
+
"""
|
|
122
|
+
ws = Path(ws)
|
|
123
|
+
if not start.is_git:
|
|
124
|
+
return Changes(note="not a git repository, so changes are not tracked")
|
|
125
|
+
c = Changes(already_dirty=len(start.dirty))
|
|
126
|
+
if start.prefix:
|
|
127
|
+
c.note = f"(inside the repository at {start.toplevel})"
|
|
128
|
+
seen: set[str] = set()
|
|
129
|
+
if start.head:
|
|
130
|
+
c.commits = int((_git(ws, "rev-list", "--count", f"{start.head}..HEAD",
|
|
131
|
+
"--", ".") or "0").strip() or 0)
|
|
132
|
+
# --relative: limited to the workspace, with paths relative to it.
|
|
133
|
+
for line in (_git(ws, "diff", "--numstat", "--relative", start.head)
|
|
134
|
+
or "").splitlines():
|
|
135
|
+
parts = line.split("\t")
|
|
136
|
+
if len(parts) != 3 or parts[2].startswith(".agentctl"):
|
|
137
|
+
continue
|
|
138
|
+
a, d, path = parts
|
|
139
|
+
c.files.append((path, None if a == "-" else int(a),
|
|
140
|
+
None if d == "-" else int(d)))
|
|
141
|
+
seen.add(path)
|
|
142
|
+
for path in sorted(_porcelain(ws, start.prefix) - start.dirty - seen):
|
|
143
|
+
p = ws / path
|
|
144
|
+
try:
|
|
145
|
+
n = sum(1 for _ in p.open(encoding="utf-8", errors="replace"))
|
|
146
|
+
except Exception: # noqa: BLE001
|
|
147
|
+
n = None
|
|
148
|
+
c.files.append((path, n, 0))
|
|
149
|
+
return c
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
# ── what it cost ──────────────────────────────────────────────────────────
|
|
153
|
+
@dataclass(frozen=True)
|
|
154
|
+
class Usage:
|
|
155
|
+
requests: int
|
|
156
|
+
tokens: int
|
|
157
|
+
cost: float
|
|
158
|
+
|
|
159
|
+
@staticmethod
|
|
160
|
+
def of(conv) -> "Usage":
|
|
161
|
+
try:
|
|
162
|
+
m = conv.state.stats.get_combined_metrics()
|
|
163
|
+
t = m.accumulated_token_usage
|
|
164
|
+
tokens = ((getattr(t, "prompt_tokens", 0) or 0)
|
|
165
|
+
+ (getattr(t, "completion_tokens", 0) or 0)) if t else 0
|
|
166
|
+
return Usage(len(m.response_latencies or []), tokens,
|
|
167
|
+
float(m.accumulated_cost or 0.0))
|
|
168
|
+
except Exception: # noqa: BLE001
|
|
169
|
+
return Usage(0, 0, 0.0)
|
|
170
|
+
|
|
171
|
+
def __sub__(self, o: "Usage") -> "Usage":
|
|
172
|
+
return Usage(self.requests - o.requests, self.tokens - o.tokens,
|
|
173
|
+
self.cost - o.cost)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def describe_cost(model: str, cost: float) -> str:
|
|
177
|
+
"""A money figure with what it can and cannot be trusted for.
|
|
178
|
+
|
|
179
|
+
`docs/0042` I-10: litellm puts LIST prices on free-tier calls and reports
|
|
180
|
+
nothing for OpenRouter's `:free` ids, so the bare number is wrong in both
|
|
181
|
+
directions. Three honest states instead of one dishonest one.
|
|
182
|
+
"""
|
|
183
|
+
from agentctl.control.providers import PROVIDERS
|
|
184
|
+
from agentctl.control.proxy import POOL
|
|
185
|
+
|
|
186
|
+
if model.endswith(":free") or model.split("/", 1)[-1].startswith(POOL):
|
|
187
|
+
return "$0.00 (free-tier model)"
|
|
188
|
+
provider = next((p for p in PROVIDERS if model.startswith(p.prefix)), None)
|
|
189
|
+
if provider and provider.free_tier:
|
|
190
|
+
return (f"${cost:.4f} at list price; a free-tier key is not billed "
|
|
191
|
+
f"(per the provider's plan, unverified here)"
|
|
192
|
+
if cost else "$0.00 (free tier, unpriced)")
|
|
193
|
+
if not cost:
|
|
194
|
+
return "unknown (the provider's price was not reported)"
|
|
195
|
+
return f"${cost:.4f} (as litellm priced it)"
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _k(n: int) -> str:
|
|
199
|
+
return f"{n / 1000:.1f}K" if n >= 1000 else str(n)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _duration(s: float) -> str:
|
|
203
|
+
s = int(round(s))
|
|
204
|
+
return f"{s // 60}m {s % 60:02d}s" if s >= 60 else f"{s}s"
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
# ── the report ────────────────────────────────────────────────────────────
|
|
208
|
+
@dataclass
|
|
209
|
+
class Report:
|
|
210
|
+
outcome: str # "PASS" | "FAIL" | "not checked"
|
|
211
|
+
accept: dict | None
|
|
212
|
+
changes: Changes
|
|
213
|
+
agent_said: str | None
|
|
214
|
+
usage: Usage
|
|
215
|
+
model: str
|
|
216
|
+
seconds: float
|
|
217
|
+
decisions: list[dict]
|
|
218
|
+
blocked: list[str]
|
|
219
|
+
ledger: str
|
|
220
|
+
conversation_id: str
|
|
221
|
+
workspace: str
|
|
222
|
+
#: (tool_call_id, what it would do) for actions queued for approval
|
|
223
|
+
#: rather than refused -- a different question from an ambiguous effect.
|
|
224
|
+
awaiting: list[tuple[str, str]] = field(default_factory=list)
|
|
225
|
+
#: Stopped by Ctrl-C after a step. Not done, and resumable.
|
|
226
|
+
paused: bool = False
|
|
227
|
+
|
|
228
|
+
@property
|
|
229
|
+
def ok(self) -> bool:
|
|
230
|
+
"""Done, as far as anything here can tell: nothing failed a check,
|
|
231
|
+
nothing waits on a human, and the user did not pause it. "not
|
|
232
|
+
checked" is not a failure -- and is never printed as a success."""
|
|
233
|
+
return self.outcome != "FAIL" and not self.blocked and not self.paused
|
|
234
|
+
|
|
235
|
+
def to_json(self) -> dict:
|
|
236
|
+
"""What a program reading the report needs (`--report-json`, the
|
|
237
|
+
GitHub Action of `docs/0053`). `text` is the report as printed, so a
|
|
238
|
+
reader never re-renders it and drifts from the terminal."""
|
|
239
|
+
return {
|
|
240
|
+
"outcome": self.outcome, "ok": self.ok, "paused": self.paused,
|
|
241
|
+
"accept": ({k: self.accept.get(k) for k in ("command", "exit", "passed")}
|
|
242
|
+
if self.accept else None),
|
|
243
|
+
"files": [p for p, _, _ in self.changes.files],
|
|
244
|
+
"commits": self.changes.commits,
|
|
245
|
+
"awaiting": [{"id": t, "what": w} for t, w in self.awaiting],
|
|
246
|
+
"blocked": len(self.blocked),
|
|
247
|
+
"model": self.model, "requests": self.usage.requests,
|
|
248
|
+
"tokens": self.usage.tokens, "seconds": round(self.seconds, 1),
|
|
249
|
+
"conversation_id": self.conversation_id,
|
|
250
|
+
"text": "\n".join(self.lines()).strip("\n"),
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
def lines(self) -> list[str]:
|
|
254
|
+
out = ["", " ── result " + "─" * 58]
|
|
255
|
+
if self.accept:
|
|
256
|
+
a = self.accept
|
|
257
|
+
out.append(f" outcome {self.outcome} `{a['command']}` exited "
|
|
258
|
+
f"{a['exit']} (run by agentctl after the agent finished)")
|
|
259
|
+
if self.outcome == "FAIL" and a.get("tail"):
|
|
260
|
+
out += [f" {l}" for l in a["tail"]]
|
|
261
|
+
else:
|
|
262
|
+
out.append(" outcome not checked (add --accept \"<test command>\" "
|
|
263
|
+
"to have agentctl check it)")
|
|
264
|
+
|
|
265
|
+
c = self.changes
|
|
266
|
+
if c.note and not c.note.startswith("(inside"):
|
|
267
|
+
out.append(f" changed {c.note}")
|
|
268
|
+
elif not c.files:
|
|
269
|
+
out.append(" changed nothing" + (f" ({plural(c.commits, 'commit')})"
|
|
270
|
+
if c.commits else ""))
|
|
271
|
+
else:
|
|
272
|
+
head = f" changed {plural(len(c.files), 'file')} +{c.added} -{c.removed}"
|
|
273
|
+
if c.commits:
|
|
274
|
+
head += f" ({plural(c.commits, 'commit')})"
|
|
275
|
+
out.append(head)
|
|
276
|
+
width = min(max(len(p) for p, _, _ in c.files), 40)
|
|
277
|
+
for path, a, d in c.files[:8]:
|
|
278
|
+
stat = "binary" if a is None else f"+{a} -{d}"
|
|
279
|
+
out.append(f" {path:<{width}} {stat}")
|
|
280
|
+
if len(c.files) > 8:
|
|
281
|
+
out.append(f" ... and {len(c.files) - 8} more")
|
|
282
|
+
if c.already_dirty:
|
|
283
|
+
out.append(f" ({plural(c.already_dirty, 'file')} "
|
|
284
|
+
f"already had changes before the run)")
|
|
285
|
+
if c.note.startswith("(inside"):
|
|
286
|
+
out.append(f" {c.note}")
|
|
287
|
+
|
|
288
|
+
if self.agent_said:
|
|
289
|
+
said = [l for l in self.agent_said.strip().splitlines() if l.strip()]
|
|
290
|
+
first = said[0][:96] + ("..." if len(said[0]) > 96 else "")
|
|
291
|
+
out.append(f" agent said \"{first}\"")
|
|
292
|
+
if len(said) > 1:
|
|
293
|
+
out.append(f" (+{len(said) - 1} more line(s) above)")
|
|
294
|
+
|
|
295
|
+
u = self.usage
|
|
296
|
+
out.append(f" used {plural(u.requests, 'request')} · "
|
|
297
|
+
f"{_k(u.tokens)} tokens · {describe_cost(self.model, u.cost)} · "
|
|
298
|
+
f"{_duration(self.seconds)}")
|
|
299
|
+
|
|
300
|
+
kinds: dict[str, int] = {}
|
|
301
|
+
verdicts: dict[str, int] = {}
|
|
302
|
+
for d in self.decisions:
|
|
303
|
+
w = CLASS_WORDS.get(d.get("class") or "", "action")
|
|
304
|
+
kinds[w] = kinds.get(w, 0) + 1
|
|
305
|
+
verdicts[d["verdict"]] = verdicts.get(d["verdict"], 0) + 1
|
|
306
|
+
if kinds:
|
|
307
|
+
parts = ", ".join(plural(n, w) for w, n in sorted(kinds.items(),
|
|
308
|
+
key=lambda kv: -kv[1]))
|
|
309
|
+
extra = []
|
|
310
|
+
if (n := verdicts.get("SUBSTITUTE")):
|
|
311
|
+
extra.append(f"{n} already done, reused")
|
|
312
|
+
if (n := verdicts.get("BLOCK", 0) + verdicts.get("ESCALATE", 0)):
|
|
313
|
+
extra.append(f"{n} paused")
|
|
314
|
+
out.append(f" actions {plural(len(self.decisions), 'action')}: "
|
|
315
|
+
f"{parts}" + (" · " + ", ".join(extra) if extra else ""))
|
|
316
|
+
|
|
317
|
+
waiting = {tid for tid, _ in self.awaiting}
|
|
318
|
+
other = [b for b in self.blocked if b not in waiting]
|
|
319
|
+
if not self.blocked:
|
|
320
|
+
out.append(" needs you nothing")
|
|
321
|
+
if self.awaiting:
|
|
322
|
+
out.append(f" needs you {plural(len(self.awaiting), 'action')} "
|
|
323
|
+
f"waiting for your approval:")
|
|
324
|
+
for tid, what in self.awaiting:
|
|
325
|
+
short = tid[:12]
|
|
326
|
+
out.append(f" {what[:70]}")
|
|
327
|
+
out.append(f" agentctl approve {short} | "
|
|
328
|
+
f"agentctl deny {short}")
|
|
329
|
+
if other:
|
|
330
|
+
label = " " if self.awaiting else " needs you "
|
|
331
|
+
out.append(f"{label} {plural(len(other), 'action')} whose outcome "
|
|
332
|
+
f"is unknown: agentctl blocked")
|
|
333
|
+
short_cid = self.conversation_id[:8]
|
|
334
|
+
if self.paused:
|
|
335
|
+
out.append(f" paused by you. Continue: agentctl resume {short_cid}")
|
|
336
|
+
elif self.awaiting:
|
|
337
|
+
out.append(f" then agentctl resume {short_cid} (the agent is "
|
|
338
|
+
f"told what you decided)")
|
|
339
|
+
else:
|
|
340
|
+
out.append(f" resume agentctl resume {short_cid}")
|
|
341
|
+
return out
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
# ── acceptance ────────────────────────────────────────────────────────────
|
|
345
|
+
def accept(command: str, ws: str | Path, timeout_s: float = 900.0) -> dict:
|
|
346
|
+
"""Run the user's check, OUTSIDE the agent and outside the ledger.
|
|
347
|
+
|
|
348
|
+
It is not an effect the agent chose, so it is not gated, and its result is
|
|
349
|
+
the harness's own observation rather than the agent's claim about itself.
|
|
350
|
+
"""
|
|
351
|
+
t0 = time.time()
|
|
352
|
+
try:
|
|
353
|
+
r = subprocess.run(command, shell=True, cwd=str(ws), capture_output=True,
|
|
354
|
+
text=True, encoding="utf-8", errors="replace",
|
|
355
|
+
timeout=timeout_s)
|
|
356
|
+
code, text = r.returncode, (r.stdout or "") + (r.stderr or "")
|
|
357
|
+
except subprocess.TimeoutExpired:
|
|
358
|
+
code, text = None, f"timed out after {timeout_s:.0f}s"
|
|
359
|
+
lines = [l for l in text.strip().splitlines() if l.strip()]
|
|
360
|
+
return {"command": command, "exit": code, "seconds": time.time() - t0,
|
|
361
|
+
"passed": code == 0, "tail": lines[-6:]}
|