handcode 0.3.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. agentctl/__init__.py +0 -0
  2. agentctl/adapters/__init__.py +0 -0
  3. agentctl/adapters/litellm/__init__.py +9 -0
  4. agentctl/adapters/litellm/hook.py +49 -0
  5. agentctl/adapters/litellm/recorder.py +187 -0
  6. agentctl/adapters/openhands/__init__.py +169 -0
  7. agentctl/adapters/openhands/handoff.py +155 -0
  8. agentctl/adapters/openhands/seam_b.py +259 -0
  9. agentctl/adapters/openhands/seam_c.py +209 -0
  10. agentctl/cli.py +1450 -0
  11. agentctl/control/__init__.py +0 -0
  12. agentctl/control/cost/__init__.py +4 -0
  13. agentctl/control/cost/ledger.py +210 -0
  14. agentctl/control/dash.py +697 -0
  15. agentctl/control/keys.py +440 -0
  16. agentctl/control/matrix/__init__.py +0 -0
  17. agentctl/control/matrix/data/tools.yaml +149 -0
  18. agentctl/control/policy/__init__.py +10 -0
  19. agentctl/control/policy/compile.py +258 -0
  20. agentctl/control/policy/data/policy.compiled.json +38 -0
  21. agentctl/control/policy/data/policy.yaml +46 -0
  22. agentctl/control/probe.py +399 -0
  23. agentctl/control/providers.py +293 -0
  24. agentctl/control/proxy.py +536 -0
  25. agentctl/control/proxyenv.py +309 -0
  26. agentctl/control/replay/__init__.py +14 -0
  27. agentctl/control/replay/cassette.py +281 -0
  28. agentctl/control/replay/server.py +109 -0
  29. agentctl/demo/__init__.py +214 -0
  30. agentctl/demo/child.py +84 -0
  31. agentctl/demo/mock.py +79 -0
  32. agentctl/demo/tool.py +62 -0
  33. agentctl/gha.py +488 -0
  34. agentctl/kernel/__init__.py +0 -0
  35. agentctl/kernel/classify.py +170 -0
  36. agentctl/kernel/gate.py +391 -0
  37. agentctl/kernel/hook.py +229 -0
  38. agentctl/kernel/ledger/__init__.py +0 -0
  39. agentctl/kernel/ledger/models.py +160 -0
  40. agentctl/kernel/ledger/schema.sql +62 -0
  41. agentctl/kernel/ledger/store.py +596 -0
  42. agentctl/kernel/paths.py +203 -0
  43. agentctl/kernel/policy.py +160 -0
  44. agentctl/kernel/reconcile/__init__.py +31 -0
  45. agentctl/kernel/reconcile/base.py +106 -0
  46. agentctl/kernel/reconcile/external.py +137 -0
  47. agentctl/kernel/reconcile/filesystem.py +162 -0
  48. agentctl/kernel/reconcile/git.py +162 -0
  49. agentctl/runtime/__init__.py +20 -0
  50. agentctl/runtime/citations.py +179 -0
  51. agentctl/runtime/config.py +97 -0
  52. agentctl/runtime/doctor.py +335 -0
  53. agentctl/runtime/init.py +148 -0
  54. agentctl/runtime/lease.py +143 -0
  55. agentctl/runtime/orchestrate.py +187 -0
  56. agentctl/runtime/plugins.py +130 -0
  57. agentctl/runtime/report.py +361 -0
  58. agentctl/runtime/runner.py +787 -0
  59. agentctl/runtime/runs.py +191 -0
  60. agentctl/runtime/subagent.py +274 -0
  61. agentctl/runtime/tools.py +350 -0
  62. handcode-0.3.0rc1.dist-info/METADATA +659 -0
  63. handcode-0.3.0rc1.dist-info/RECORD +67 -0
  64. handcode-0.3.0rc1.dist-info/WHEEL +5 -0
  65. handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
  66. handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
  67. handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,187 @@
1
+ r"""Fan a read-only recon job out across sources, by scarcity.
2
+
3
+ `docs/0040` §6.1: **allocate work inversely to quota scarcity, not evenly.**
4
+ That single scheduling choice is the difference between ×1.55 and ×4.33 more
5
+ recon tasks per day on the owner's pool, with no extra quota bought. An even
6
+ split puts the same load on a leg with one allowance as on a leg with six, and
7
+ the one-allowance leg then decides the throughput of the whole job.
8
+
9
+ ## Why this allocates by quota count and not by rate limit
10
+
11
+ The obvious allocator divides work by each source's requests-per-day. This one
12
+ cannot, and the reason is a rule rather than an oversight:
13
+ `control/providers.py` deliberately records no rate limits, because they go
14
+ stale within weeks and a confidently wrong number is worse than none.
15
+
16
+ So the allocator uses the one fact the registry *will* vouch for — how many
17
+ **independent quotas** a source has (`providers.quotas_for`) — and splits in
18
+ proportion to that. It is structural, it does not rot, and on the measured
19
+ pool it recovers most of the available gain:
20
+
21
+ Gemini 1 quota : Mistral 6 quotas -> 12 files split 2 / 10
22
+ ×3.61 of the ×4.33 ceiling
23
+
24
+ Reaching the last 20% needs each source's real RPD, which is exactly the
25
+ number this project refuses to hardcode. `ratio=` accepts one if the caller
26
+ has measured it today.
27
+
28
+ ## Sequential, deliberately
29
+
30
+ `docs/0039` §5 records that subagent concurrency is unproven: the two globals
31
+ that raced are fixed and three further candidates were refuted, but nothing
32
+ exercises the concurrent path. `research/phase-10-5`'s build order puts
33
+ sequential fan-out before concurrent for that reason. Concurrency is a change
34
+ to this file only, once something tests it.
35
+ """
36
+ from __future__ import annotations
37
+
38
+ from dataclasses import dataclass, field
39
+ from typing import Any, Callable
40
+
41
+
42
+ @dataclass
43
+ class Leg:
44
+ """One source's share of the job, and what came back."""
45
+ source: str
46
+ quotas: int
47
+ items: list[str] = field(default_factory=list)
48
+ report: str | None = None
49
+ error: str | None = None
50
+
51
+ @property
52
+ def requests(self) -> int:
53
+ """One request per item, plus the report it ends with."""
54
+ return len(self.items) + 1 if self.items else 0
55
+
56
+
57
+ def weights(sources: list[str], ratio: dict[str, int] | None = None,
58
+ env: dict | None = None) -> dict[str, int]:
59
+ """How much work each source should carry, relatively.
60
+
61
+ Defaults to its independent quota count. `ratio` overrides per source, for
62
+ a caller who has measured real limits and wants to use them — pass the
63
+ numbers, not a promise that they are current.
64
+ """
65
+ from agentctl.control.providers import BY_NAME, quotas_for
66
+
67
+ out: dict[str, int] = {}
68
+ for name in sources:
69
+ if ratio and name in ratio:
70
+ out[name] = max(int(ratio[name]), 0)
71
+ continue
72
+ p = BY_NAME.get(name)
73
+ out[name] = quotas_for(p, env) if p is not None else 1
74
+ return out
75
+
76
+
77
+ def allocate(items: list[str], sources: list[str],
78
+ ratio: dict[str, int] | None = None,
79
+ env: dict | None = None) -> list[Leg]:
80
+ """Split `items` across `sources` in proportion to their quotas.
81
+
82
+ Largest-remainder, so the split is exact rather than drifting by rounding,
83
+ and deterministic — the same inputs always give the same plan, which is
84
+ what makes a fan-out reproducible enough to compare two runs.
85
+
86
+ A source with no quota gets no work. A source that *has* quota gets at
87
+ least one item, because a leg with zero items still costs a report request
88
+ and returns nothing: paying one request to learn nothing is strictly worse
89
+ than not dispatching it, so it is dropped instead.
90
+ """
91
+ w = weights(sources, ratio, env)
92
+ live = [s for s in sources if w.get(s, 0) > 0]
93
+ if not items or not live:
94
+ return [Leg(s, w.get(s, 0)) for s in sources]
95
+
96
+ # More legs than items would leave some carrying nothing. Keep the
97
+ # highest-weighted, so a scarce leg is dropped before a plentiful one.
98
+ if len(live) > len(items):
99
+ live = sorted(live, key=lambda s: (-w[s], s))[:len(items)]
100
+
101
+ total = sum(w[s] for s in live)
102
+ exact = {s: len(items) * w[s] / total for s in live}
103
+ base = {s: max(int(exact[s]), 1) for s in live}
104
+
105
+ # Largest remainder, then trim from the most plentiful leg if the
106
+ # one-item floor overshot.
107
+ short = len(items) - sum(base.values())
108
+ order = sorted(live, key=lambda s: (-(exact[s] - int(exact[s])), s))
109
+ i = 0
110
+ while short > 0:
111
+ base[order[i % len(order)]] += 1
112
+ short -= 1
113
+ i += 1
114
+ while short < 0:
115
+ victim = max((s for s in live if base[s] > 1),
116
+ key=lambda s: (w[s], s), default=None)
117
+ if victim is None:
118
+ break
119
+ base[victim] -= 1
120
+ short += 1
121
+
122
+ legs, cut = [], 0
123
+ for s in sources:
124
+ n = base.get(s, 0)
125
+ legs.append(Leg(s, w.get(s, 0), items[cut:cut + n]))
126
+ cut += n
127
+ return legs
128
+
129
+
130
+ def describe(legs: list[Leg]) -> str:
131
+ """The plan, and why it is uneven. Printed before anything is spent."""
132
+ live = [lg for lg in legs if lg.items]
133
+ if not live:
134
+ return " nothing to dispatch"
135
+ out = []
136
+ for lg in legs:
137
+ if not lg.items:
138
+ why = "no quota" if not lg.quotas else "no share at this size"
139
+ out.append(f" -- {lg.source:<14} {why}")
140
+ continue
141
+ out.append(f" -> {lg.source:<14} {len(lg.items):>2} item(s), "
142
+ f"{lg.requests:>2} request(s) "
143
+ f"{lg.quotas} quota(s)")
144
+ scarce = min(live, key=lambda lg: lg.quotas)
145
+ rich = max(live, key=lambda lg: lg.quotas)
146
+ if scarce.quotas != rich.quotas:
147
+ # ASCII only in printed output. Non-ASCII in a rendered string has
148
+ # broken this project on a cp437 console five times; the dashboard
149
+ # carries a test for exactly this.
150
+ out.append(f" {scarce.source} carries less because it has "
151
+ f"{scarce.quotas} quota(s) to {rich.source}'s {rich.quotas} "
152
+ f"-- an even split would let it set the pace for all of "
153
+ f"them (docs/0040 sec 6.1).")
154
+ return "\n".join(out)
155
+
156
+
157
+ def fan_out(definition: Any, question: str, legs: list[Leg], *,
158
+ workspace: str = ".", base_url: str | None = None,
159
+ runner: Callable[..., str] | None = None,
160
+ on_leg: Callable[[Leg], None] | None = None) -> list[Leg]:
161
+ """Run each leg's scout in turn, filling in `report` or `error`.
162
+
163
+ Sequential. One leg failing does not stop the others — a recon job that
164
+ loses a scout should return what the rest found, clearly short, rather
165
+ than nothing at all. The caller sees which legs are missing because
166
+ `error` is set, not because the report is quietly thinner.
167
+ """
168
+ from agentctl.control.proxy import SOURCE_PREFIX
169
+
170
+ from .subagent import run as run_subagent
171
+
172
+ run = runner or run_subagent
173
+ for leg in legs:
174
+ if not leg.items:
175
+ continue
176
+ task = (f"{question}\n\nLook only at these, and report on them:\n"
177
+ + "\n".join(f" - {i}" for i in leg.items))
178
+ try:
179
+ leg.report = run(
180
+ definition, task, workspace=workspace,
181
+ model=f"openai/{SOURCE_PREFIX}{leg.source}" if base_url else None,
182
+ base_url=base_url)
183
+ except Exception as e: # noqa: BLE001
184
+ leg.error = f"{type(e).__name__}: {e}"
185
+ if on_leg is not None:
186
+ on_leg(leg)
187
+ return legs
@@ -0,0 +1,130 @@
1
+ r"""Claude Code plugins, admitted one capability at a time.
2
+
3
+ `docs/0010` §9.1 gave plugins the only **Partial BUILD** in an otherwise
4
+ SKIP-heavy table, and §9.2 said exactly which slice: packaging is
5
+ harness-specific and not our job, but *"policy over what may be installed and
6
+ reached is not."* This is that slice, at its smallest useful size.
7
+
8
+ A Claude Code plugin directory can contribute six things:
9
+
10
+ agents/ subagent definitions
11
+ hooks/ event handlers
12
+ commands/ slash commands
13
+ skills/ agent skills
14
+ .mcp.json MCP servers
15
+ plugin.json the manifest
16
+
17
+ **Five of those six are refused here, and the refusal is the feature.** A
18
+ plugin is third-party content: `docs/0010` §8.2 records that MCP amplifies
19
+ the effect problem, and hooks and commands are arbitrary behaviour attached
20
+ to an agent loop that this project spends its whole correctness budget
21
+ guarding. Admitting them because they happened to be in the folder would be
22
+ the opposite of a broker.
23
+
24
+ So the rule is **default-deny with an itemised receipt**. Only read-only agent
25
+ definitions are admitted, every one re-validated by
26
+ `subagent.validate` rather than trusted for having arrived in a manifest, and
27
+ everything declined is counted and named. A broker that silently dropped what
28
+ it would not run would leave you believing you had installed something you
29
+ had not.
30
+
31
+ ## What is still trusted, and should be said plainly
32
+
33
+ An admitted definition carries a third-party `system_prompt`, and that becomes
34
+ instructions to a model that can read your files. Read-only bounds the damage
35
+ to *reading* — it cannot write, shell out, or reach the network — and the
36
+ workspace is scoped per task, so it reads where you pointed it and nowhere
37
+ else. That is a real bound, not a complete one. Pin plugin versions and do not
38
+ install one you would not read.
39
+ """
40
+ from __future__ import annotations
41
+
42
+ from pathlib import Path
43
+ from typing import Any
44
+
45
+ #: The one capability a plugin may contribute.
46
+ ADMITTED = "agents"
47
+
48
+ #: Everything else, with why. Named rather than silently skipped — the point of
49
+ #: a broker is that you can audit what it refused.
50
+ REFUSED: dict[str, str] = {
51
+ "mcp_config": "an MCP server is an arbitrary tool surface reached over a "
52
+ "pipe; nothing here can inspect what it would do "
53
+ "(`docs/0010` §8.2)",
54
+ "hooks": "a hook is arbitrary behaviour attached to the agent loop, which "
55
+ "is the thing the gate exists to guard",
56
+ "commands": "slash commands belong to a harness UI, not to a control "
57
+ "plane (`docs/0010` §9.1)",
58
+ "skills": "a skill injects instructions into the context window; the "
59
+ "harness owns that and does it better (`0007`)",
60
+ }
61
+
62
+
63
+ def load(plugin_dir: str | Path) -> dict[str, Any]:
64
+ """Inspect a plugin. Reports what would be admitted and what is refused.
65
+
66
+ Loads nothing into a live agent — this is the audit step. `agentctl
67
+ plugins` prints it, and a caller that wants the definitions takes
68
+ `admitted`.
69
+ """
70
+ from openhands.sdk.plugin.format.claude_code import ClaudeCodePluginFormat
71
+
72
+ from .subagent import NotReadOnly, validate
73
+
74
+ d = Path(plugin_dir)
75
+ if not d.is_dir():
76
+ return {"path": str(d), "error": f"no such directory: {d}"}
77
+
78
+ fmt = ClaudeCodePluginFormat()
79
+ try:
80
+ manifest = fmt.load_manifest(d)
81
+ except Exception as e: # noqa: BLE001
82
+ return {"path": str(d), "error": f"unreadable manifest: {e}"}
83
+
84
+ admitted, rejected = [], []
85
+ try:
86
+ for defn in fmt.load_agents(d):
87
+ try:
88
+ validate(defn)
89
+ admitted.append(defn)
90
+ except NotReadOnly as e:
91
+ rejected.append((getattr(defn, "name", "<unnamed>"), str(e)))
92
+ except Exception as e: # noqa: BLE001
93
+ rejected.append(("<agents/>", f"could not be read: {e}"))
94
+
95
+ return {
96
+ "path": str(d),
97
+ "name": getattr(manifest, "name", d.name),
98
+ "version": getattr(manifest, "version", "?"),
99
+ "description": getattr(manifest, "description", ""),
100
+ "admitted": admitted,
101
+ "rejected": rejected,
102
+ "refused": _refused_counts(fmt, d),
103
+ }
104
+
105
+
106
+ def _refused_counts(fmt: Any, d: Path) -> list[tuple[str, int, str]]:
107
+ """What the plugin carries that this project will not run.
108
+
109
+ Counted, not just listed. "declines MCP servers" is a policy statement;
110
+ "declines 3 MCP servers" is a fact about the thing in front of you, and
111
+ only the second tells you whether refusing it matters.
112
+ """
113
+ out = []
114
+ for cap, why in REFUSED.items():
115
+ try:
116
+ loader = {
117
+ "mcp_config": fmt.load_mcp_config,
118
+ "hooks": fmt.load_hooks,
119
+ "commands": fmt.load_commands,
120
+ "skills": lambda p: [], # assembled by the base class' load()
121
+ }[cap]
122
+ got = loader(d)
123
+ except Exception: # noqa: BLE001
124
+ continue
125
+ if got is None:
126
+ continue
127
+ n = len(got) if hasattr(got, "__len__") else 1
128
+ if n:
129
+ out.append((cap, n, why))
130
+ return out
@@ -0,0 +1,361 @@
1
+ """What a run did, in words a user can act on. `docs/0043` Phase 3.
2
+
3
+ Phase 0 (`docs/0044` F6, F10) ended every run with `decisions {'EXECUTE': 9}`
4
+ -- the gate's vocabulary, not the user's -- and nothing about whether the task
5
+ worked, what changed, or what it cost. The report answers four questions, and
6
+ derives every line from the same records the precise output uses, because a
7
+ friendly line computed some other way is how a tool starts saying confidently
8
+ wrong things (`docs/0039`):
9
+
10
+ outcome did it work? only what was CHECKED; never implied
11
+ changed what is different? from git, against the run's start
12
+ used what did it cost? requests, tokens, money -- labelled
13
+ needs you what is left? blocked effects, with the command
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import subprocess
18
+ import time
19
+ from dataclasses import dataclass, field
20
+ from pathlib import Path
21
+
22
+ # ── the vocabulary (`docs/0043` §3, Phase 3) ──────────────────────────────
23
+ #: The gate's verdicts, as a user would say them.
24
+ VERDICT_WORDS = {
25
+ "EXECUTE": "ran",
26
+ "SUBSTITUTE": "already done; reused the recorded result",
27
+ "BLOCK": "paused; needs your decision",
28
+ "ESCALATE": "paused; needs your decision",
29
+ }
30
+
31
+ #: Effect classes, as kinds of action.
32
+ CLASS_WORDS = {
33
+ "PURE_READ": "read",
34
+ "IDEMPOTENT_WRITE": "file write",
35
+ "NON_IDEMPOTENT_WRITE": "command",
36
+ "EXTERNAL": "command",
37
+ "DESTRUCTIVE": "dangerous command",
38
+ }
39
+
40
+
41
+ def plural(n: int, word: str) -> str:
42
+ return f"{n} {word}{'' if n == 1 else 's'}"
43
+
44
+
45
+ # ── what changed ──────────────────────────────────────────────────────────
46
+ def _git(ws: Path, *args: str) -> str | None:
47
+ try:
48
+ r = subprocess.run(["git", *args], cwd=str(ws), capture_output=True,
49
+ text=True, encoding="utf-8", errors="replace",
50
+ timeout=30)
51
+ except Exception: # noqa: BLE001
52
+ return None
53
+ return r.stdout if r.returncode == 0 else None
54
+
55
+
56
+ def _porcelain(ws: Path, prefix: str = "") -> set[str]:
57
+ """Changed paths under `ws`, relative to it. `prefix` is `ws` relative to
58
+ the repository's top level: porcelain paths are always top-level ones."""
59
+ out = _git(ws, "status", "--porcelain=v1", "--untracked-files=all",
60
+ "--", ".") or ""
61
+ paths = set()
62
+ for line in out.splitlines():
63
+ path = line[3:].strip().strip('"')
64
+ if prefix and path.startswith(prefix):
65
+ path = path[len(prefix):]
66
+ if path and not path.startswith(".agentctl"):
67
+ paths.add(path)
68
+ return paths
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class Start:
73
+ """The workspace as the run found it."""
74
+ is_git: bool
75
+ head: str | None # None: a repo with no commits yet
76
+ dirty: frozenset[str] # paths already changed before the run
77
+ at: float
78
+ #: `ws` relative to the repository's top level, "" when it IS the top.
79
+ #: Non-empty means the workspace sits inside a bigger repository -- on
80
+ #: the machine this was built on, the home directory itself -- and every
81
+ #: git question must be scoped to the workspace, or the report describes
82
+ #: someone else's files (`docs/0048`).
83
+ prefix: str = ""
84
+ toplevel: str = ""
85
+
86
+
87
+ def snapshot(ws: str | Path) -> Start:
88
+ ws = Path(ws)
89
+ if _git(ws, "rev-parse", "--is-inside-work-tree") is None:
90
+ return Start(False, None, frozenset(), time.time())
91
+ head = (_git(ws, "rev-parse", "HEAD") or "").strip() or None
92
+ prefix = (_git(ws, "rev-parse", "--show-prefix") or "").strip()
93
+ top = (_git(ws, "rev-parse", "--show-toplevel") or "").strip()
94
+ return Start(True, head, frozenset(_porcelain(ws, prefix)), time.time(),
95
+ prefix, top)
96
+
97
+
98
+ @dataclass
99
+ class Changes:
100
+ files: list[tuple[str, int | None, int | None]] = field(default_factory=list)
101
+ commits: int = 0
102
+ already_dirty: int = 0
103
+ note: str = ""
104
+
105
+ @property
106
+ def added(self) -> int:
107
+ return sum(a or 0 for _, a, _ in self.files)
108
+
109
+ @property
110
+ def removed(self) -> int:
111
+ return sum(d or 0 for _, _, d in self.files)
112
+
113
+
114
+ def changes(ws: str | Path, start: Start) -> Changes:
115
+ """Files that differ from where the run started, commits included.
116
+
117
+ Measured against the starting commit, so work the agent committed counts
118
+ as well as work it left uncommitted. Files that were ALREADY modified
119
+ before the run are counted only if they differ now, and their number is
120
+ reported, because their diff mixes the user's edits with the agent's.
121
+ """
122
+ ws = Path(ws)
123
+ if not start.is_git:
124
+ return Changes(note="not a git repository, so changes are not tracked")
125
+ c = Changes(already_dirty=len(start.dirty))
126
+ if start.prefix:
127
+ c.note = f"(inside the repository at {start.toplevel})"
128
+ seen: set[str] = set()
129
+ if start.head:
130
+ c.commits = int((_git(ws, "rev-list", "--count", f"{start.head}..HEAD",
131
+ "--", ".") or "0").strip() or 0)
132
+ # --relative: limited to the workspace, with paths relative to it.
133
+ for line in (_git(ws, "diff", "--numstat", "--relative", start.head)
134
+ or "").splitlines():
135
+ parts = line.split("\t")
136
+ if len(parts) != 3 or parts[2].startswith(".agentctl"):
137
+ continue
138
+ a, d, path = parts
139
+ c.files.append((path, None if a == "-" else int(a),
140
+ None if d == "-" else int(d)))
141
+ seen.add(path)
142
+ for path in sorted(_porcelain(ws, start.prefix) - start.dirty - seen):
143
+ p = ws / path
144
+ try:
145
+ n = sum(1 for _ in p.open(encoding="utf-8", errors="replace"))
146
+ except Exception: # noqa: BLE001
147
+ n = None
148
+ c.files.append((path, n, 0))
149
+ return c
150
+
151
+
152
+ # ── what it cost ──────────────────────────────────────────────────────────
153
+ @dataclass(frozen=True)
154
+ class Usage:
155
+ requests: int
156
+ tokens: int
157
+ cost: float
158
+
159
+ @staticmethod
160
+ def of(conv) -> "Usage":
161
+ try:
162
+ m = conv.state.stats.get_combined_metrics()
163
+ t = m.accumulated_token_usage
164
+ tokens = ((getattr(t, "prompt_tokens", 0) or 0)
165
+ + (getattr(t, "completion_tokens", 0) or 0)) if t else 0
166
+ return Usage(len(m.response_latencies or []), tokens,
167
+ float(m.accumulated_cost or 0.0))
168
+ except Exception: # noqa: BLE001
169
+ return Usage(0, 0, 0.0)
170
+
171
+ def __sub__(self, o: "Usage") -> "Usage":
172
+ return Usage(self.requests - o.requests, self.tokens - o.tokens,
173
+ self.cost - o.cost)
174
+
175
+
176
+ def describe_cost(model: str, cost: float) -> str:
177
+ """A money figure with what it can and cannot be trusted for.
178
+
179
+ `docs/0042` I-10: litellm puts LIST prices on free-tier calls and reports
180
+ nothing for OpenRouter's `:free` ids, so the bare number is wrong in both
181
+ directions. Three honest states instead of one dishonest one.
182
+ """
183
+ from agentctl.control.providers import PROVIDERS
184
+ from agentctl.control.proxy import POOL
185
+
186
+ if model.endswith(":free") or model.split("/", 1)[-1].startswith(POOL):
187
+ return "$0.00 (free-tier model)"
188
+ provider = next((p for p in PROVIDERS if model.startswith(p.prefix)), None)
189
+ if provider and provider.free_tier:
190
+ return (f"${cost:.4f} at list price; a free-tier key is not billed "
191
+ f"(per the provider's plan, unverified here)"
192
+ if cost else "$0.00 (free tier, unpriced)")
193
+ if not cost:
194
+ return "unknown (the provider's price was not reported)"
195
+ return f"${cost:.4f} (as litellm priced it)"
196
+
197
+
198
+ def _k(n: int) -> str:
199
+ return f"{n / 1000:.1f}K" if n >= 1000 else str(n)
200
+
201
+
202
+ def _duration(s: float) -> str:
203
+ s = int(round(s))
204
+ return f"{s // 60}m {s % 60:02d}s" if s >= 60 else f"{s}s"
205
+
206
+
207
+ # ── the report ────────────────────────────────────────────────────────────
208
+ @dataclass
209
+ class Report:
210
+ outcome: str # "PASS" | "FAIL" | "not checked"
211
+ accept: dict | None
212
+ changes: Changes
213
+ agent_said: str | None
214
+ usage: Usage
215
+ model: str
216
+ seconds: float
217
+ decisions: list[dict]
218
+ blocked: list[str]
219
+ ledger: str
220
+ conversation_id: str
221
+ workspace: str
222
+ #: (tool_call_id, what it would do) for actions queued for approval
223
+ #: rather than refused -- a different question from an ambiguous effect.
224
+ awaiting: list[tuple[str, str]] = field(default_factory=list)
225
+ #: Stopped by Ctrl-C after a step. Not done, and resumable.
226
+ paused: bool = False
227
+
228
+ @property
229
+ def ok(self) -> bool:
230
+ """Done, as far as anything here can tell: nothing failed a check,
231
+ nothing waits on a human, and the user did not pause it. "not
232
+ checked" is not a failure -- and is never printed as a success."""
233
+ return self.outcome != "FAIL" and not self.blocked and not self.paused
234
+
235
+ def to_json(self) -> dict:
236
+ """What a program reading the report needs (`--report-json`, the
237
+ GitHub Action of `docs/0053`). `text` is the report as printed, so a
238
+ reader never re-renders it and drifts from the terminal."""
239
+ return {
240
+ "outcome": self.outcome, "ok": self.ok, "paused": self.paused,
241
+ "accept": ({k: self.accept.get(k) for k in ("command", "exit", "passed")}
242
+ if self.accept else None),
243
+ "files": [p for p, _, _ in self.changes.files],
244
+ "commits": self.changes.commits,
245
+ "awaiting": [{"id": t, "what": w} for t, w in self.awaiting],
246
+ "blocked": len(self.blocked),
247
+ "model": self.model, "requests": self.usage.requests,
248
+ "tokens": self.usage.tokens, "seconds": round(self.seconds, 1),
249
+ "conversation_id": self.conversation_id,
250
+ "text": "\n".join(self.lines()).strip("\n"),
251
+ }
252
+
253
+ def lines(self) -> list[str]:
254
+ out = ["", " ── result " + "─" * 58]
255
+ if self.accept:
256
+ a = self.accept
257
+ out.append(f" outcome {self.outcome} `{a['command']}` exited "
258
+ f"{a['exit']} (run by agentctl after the agent finished)")
259
+ if self.outcome == "FAIL" and a.get("tail"):
260
+ out += [f" {l}" for l in a["tail"]]
261
+ else:
262
+ out.append(" outcome not checked (add --accept \"<test command>\" "
263
+ "to have agentctl check it)")
264
+
265
+ c = self.changes
266
+ if c.note and not c.note.startswith("(inside"):
267
+ out.append(f" changed {c.note}")
268
+ elif not c.files:
269
+ out.append(" changed nothing" + (f" ({plural(c.commits, 'commit')})"
270
+ if c.commits else ""))
271
+ else:
272
+ head = f" changed {plural(len(c.files), 'file')} +{c.added} -{c.removed}"
273
+ if c.commits:
274
+ head += f" ({plural(c.commits, 'commit')})"
275
+ out.append(head)
276
+ width = min(max(len(p) for p, _, _ in c.files), 40)
277
+ for path, a, d in c.files[:8]:
278
+ stat = "binary" if a is None else f"+{a} -{d}"
279
+ out.append(f" {path:<{width}} {stat}")
280
+ if len(c.files) > 8:
281
+ out.append(f" ... and {len(c.files) - 8} more")
282
+ if c.already_dirty:
283
+ out.append(f" ({plural(c.already_dirty, 'file')} "
284
+ f"already had changes before the run)")
285
+ if c.note.startswith("(inside"):
286
+ out.append(f" {c.note}")
287
+
288
+ if self.agent_said:
289
+ said = [l for l in self.agent_said.strip().splitlines() if l.strip()]
290
+ first = said[0][:96] + ("..." if len(said[0]) > 96 else "")
291
+ out.append(f" agent said \"{first}\"")
292
+ if len(said) > 1:
293
+ out.append(f" (+{len(said) - 1} more line(s) above)")
294
+
295
+ u = self.usage
296
+ out.append(f" used {plural(u.requests, 'request')} · "
297
+ f"{_k(u.tokens)} tokens · {describe_cost(self.model, u.cost)} · "
298
+ f"{_duration(self.seconds)}")
299
+
300
+ kinds: dict[str, int] = {}
301
+ verdicts: dict[str, int] = {}
302
+ for d in self.decisions:
303
+ w = CLASS_WORDS.get(d.get("class") or "", "action")
304
+ kinds[w] = kinds.get(w, 0) + 1
305
+ verdicts[d["verdict"]] = verdicts.get(d["verdict"], 0) + 1
306
+ if kinds:
307
+ parts = ", ".join(plural(n, w) for w, n in sorted(kinds.items(),
308
+ key=lambda kv: -kv[1]))
309
+ extra = []
310
+ if (n := verdicts.get("SUBSTITUTE")):
311
+ extra.append(f"{n} already done, reused")
312
+ if (n := verdicts.get("BLOCK", 0) + verdicts.get("ESCALATE", 0)):
313
+ extra.append(f"{n} paused")
314
+ out.append(f" actions {plural(len(self.decisions), 'action')}: "
315
+ f"{parts}" + (" · " + ", ".join(extra) if extra else ""))
316
+
317
+ waiting = {tid for tid, _ in self.awaiting}
318
+ other = [b for b in self.blocked if b not in waiting]
319
+ if not self.blocked:
320
+ out.append(" needs you nothing")
321
+ if self.awaiting:
322
+ out.append(f" needs you {plural(len(self.awaiting), 'action')} "
323
+ f"waiting for your approval:")
324
+ for tid, what in self.awaiting:
325
+ short = tid[:12]
326
+ out.append(f" {what[:70]}")
327
+ out.append(f" agentctl approve {short} | "
328
+ f"agentctl deny {short}")
329
+ if other:
330
+ label = " " if self.awaiting else " needs you "
331
+ out.append(f"{label} {plural(len(other), 'action')} whose outcome "
332
+ f"is unknown: agentctl blocked")
333
+ short_cid = self.conversation_id[:8]
334
+ if self.paused:
335
+ out.append(f" paused by you. Continue: agentctl resume {short_cid}")
336
+ elif self.awaiting:
337
+ out.append(f" then agentctl resume {short_cid} (the agent is "
338
+ f"told what you decided)")
339
+ else:
340
+ out.append(f" resume agentctl resume {short_cid}")
341
+ return out
342
+
343
+
344
+ # ── acceptance ────────────────────────────────────────────────────────────
345
+ def accept(command: str, ws: str | Path, timeout_s: float = 900.0) -> dict:
346
+ """Run the user's check, OUTSIDE the agent and outside the ledger.
347
+
348
+ It is not an effect the agent chose, so it is not gated, and its result is
349
+ the harness's own observation rather than the agent's claim about itself.
350
+ """
351
+ t0 = time.time()
352
+ try:
353
+ r = subprocess.run(command, shell=True, cwd=str(ws), capture_output=True,
354
+ text=True, encoding="utf-8", errors="replace",
355
+ timeout=timeout_s)
356
+ code, text = r.returncode, (r.stdout or "") + (r.stderr or "")
357
+ except subprocess.TimeoutExpired:
358
+ code, text = None, f"timed out after {timeout_s:.0f}s"
359
+ lines = [l for l in text.strip().splitlines() if l.strip()]
360
+ return {"command": command, "exit": code, "seconds": time.time() - t0,
361
+ "passed": code == 0, "tail": lines[-6:]}