handcode 0.3.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentctl/__init__.py +0 -0
- agentctl/adapters/__init__.py +0 -0
- agentctl/adapters/litellm/__init__.py +9 -0
- agentctl/adapters/litellm/hook.py +49 -0
- agentctl/adapters/litellm/recorder.py +187 -0
- agentctl/adapters/openhands/__init__.py +169 -0
- agentctl/adapters/openhands/handoff.py +155 -0
- agentctl/adapters/openhands/seam_b.py +259 -0
- agentctl/adapters/openhands/seam_c.py +209 -0
- agentctl/cli.py +1450 -0
- agentctl/control/__init__.py +0 -0
- agentctl/control/cost/__init__.py +4 -0
- agentctl/control/cost/ledger.py +210 -0
- agentctl/control/dash.py +697 -0
- agentctl/control/keys.py +440 -0
- agentctl/control/matrix/__init__.py +0 -0
- agentctl/control/matrix/data/tools.yaml +149 -0
- agentctl/control/policy/__init__.py +10 -0
- agentctl/control/policy/compile.py +258 -0
- agentctl/control/policy/data/policy.compiled.json +38 -0
- agentctl/control/policy/data/policy.yaml +46 -0
- agentctl/control/probe.py +399 -0
- agentctl/control/providers.py +293 -0
- agentctl/control/proxy.py +536 -0
- agentctl/control/proxyenv.py +309 -0
- agentctl/control/replay/__init__.py +14 -0
- agentctl/control/replay/cassette.py +281 -0
- agentctl/control/replay/server.py +109 -0
- agentctl/demo/__init__.py +214 -0
- agentctl/demo/child.py +84 -0
- agentctl/demo/mock.py +79 -0
- agentctl/demo/tool.py +62 -0
- agentctl/gha.py +488 -0
- agentctl/kernel/__init__.py +0 -0
- agentctl/kernel/classify.py +170 -0
- agentctl/kernel/gate.py +391 -0
- agentctl/kernel/hook.py +229 -0
- agentctl/kernel/ledger/__init__.py +0 -0
- agentctl/kernel/ledger/models.py +160 -0
- agentctl/kernel/ledger/schema.sql +62 -0
- agentctl/kernel/ledger/store.py +596 -0
- agentctl/kernel/paths.py +203 -0
- agentctl/kernel/policy.py +160 -0
- agentctl/kernel/reconcile/__init__.py +31 -0
- agentctl/kernel/reconcile/base.py +106 -0
- agentctl/kernel/reconcile/external.py +137 -0
- agentctl/kernel/reconcile/filesystem.py +162 -0
- agentctl/kernel/reconcile/git.py +162 -0
- agentctl/runtime/__init__.py +20 -0
- agentctl/runtime/citations.py +179 -0
- agentctl/runtime/config.py +97 -0
- agentctl/runtime/doctor.py +335 -0
- agentctl/runtime/init.py +148 -0
- agentctl/runtime/lease.py +143 -0
- agentctl/runtime/orchestrate.py +187 -0
- agentctl/runtime/plugins.py +130 -0
- agentctl/runtime/report.py +361 -0
- agentctl/runtime/runner.py +787 -0
- agentctl/runtime/runs.py +191 -0
- agentctl/runtime/subagent.py +274 -0
- agentctl/runtime/tools.py +350 -0
- handcode-0.3.0rc1.dist-info/METADATA +659 -0
- handcode-0.3.0rc1.dist-info/RECORD +67 -0
- handcode-0.3.0rc1.dist-info/WHEEL +5 -0
- handcode-0.3.0rc1.dist-info/entry_points.txt +3 -0
- handcode-0.3.0rc1.dist-info/licenses/LICENSE +21 -0
- handcode-0.3.0rc1.dist-info/top_level.txt +1 -0
agentctl/runtime/runs.py
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""Every run, in one place, so no follow-up command needs a path. `docs/0042` I-17.
|
|
2
|
+
|
|
3
|
+
Phase 0 (`docs/0044` F4): runs wrote `<workspace>/.agentctl/ledger.db`, and
|
|
4
|
+
`status`, `blocked`, `resolve` and `dash` read `./ledger.db` -- "no ledger at
|
|
5
|
+
ledger.db" from inside the very workspace that had one. Resume meant copying a
|
|
6
|
+
UUID out of scrollback (F5).
|
|
7
|
+
|
|
8
|
+
`~/.agentctl/runs.db` has one row per run ATTEMPT (a resume is another row of
|
|
9
|
+
the same conversation):
|
|
10
|
+
|
|
11
|
+
started written before the agent's first step
|
|
12
|
+
ended written when the run returns, however it returns
|
|
13
|
+
|
|
14
|
+
A row still `running` whose process is gone is a run that DIED -- a crash, a
|
|
15
|
+
killed terminal, a closed laptop -- and is shown as such. That is the point of
|
|
16
|
+
writing the start first: absence of an end is evidence.
|
|
17
|
+
|
|
18
|
+
Local only. It holds task text and paths, so it lives under `~/.agentctl`,
|
|
19
|
+
outside every repository.
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import os
|
|
24
|
+
import sqlite3
|
|
25
|
+
import time
|
|
26
|
+
from contextlib import contextmanager
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
SCHEMA = """
|
|
31
|
+
PRAGMA journal_mode = WAL;
|
|
32
|
+
CREATE TABLE IF NOT EXISTS run (
|
|
33
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
34
|
+
conversation_id TEXT NOT NULL,
|
|
35
|
+
workspace TEXT NOT NULL,
|
|
36
|
+
ledger TEXT NOT NULL,
|
|
37
|
+
model TEXT,
|
|
38
|
+
base_url TEXT,
|
|
39
|
+
task TEXT,
|
|
40
|
+
pid INTEGER,
|
|
41
|
+
started REAL NOT NULL,
|
|
42
|
+
ended REAL,
|
|
43
|
+
-- running | done | needs_you | failed | interrupted | rate_limited | error
|
|
44
|
+
status TEXT NOT NULL,
|
|
45
|
+
outcome TEXT,
|
|
46
|
+
requests INTEGER,
|
|
47
|
+
detail TEXT
|
|
48
|
+
);
|
|
49
|
+
CREATE INDEX IF NOT EXISTS ix_run_ws ON run(workspace, started);
|
|
50
|
+
CREATE INDEX IF NOT EXISTS ix_run_conv ON run(conversation_id);
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def path() -> Path:
|
|
55
|
+
base = os.environ.get("AGENTCTL_HOME") or str(Path.home() / ".agentctl")
|
|
56
|
+
return Path(base) / "runs.db"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@contextmanager
|
|
60
|
+
def _db():
|
|
61
|
+
"""A connection that is CLOSED afterwards. `with sqlite3.connect()` only
|
|
62
|
+
commits, and on Windows an open handle keeps the file locked."""
|
|
63
|
+
p = path()
|
|
64
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
65
|
+
db = sqlite3.connect(str(p), isolation_level=None, timeout=10)
|
|
66
|
+
try:
|
|
67
|
+
db.row_factory = sqlite3.Row
|
|
68
|
+
db.executescript(SCHEMA)
|
|
69
|
+
yield db
|
|
70
|
+
finally:
|
|
71
|
+
db.close()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass(frozen=True)
|
|
75
|
+
class Run:
|
|
76
|
+
id: int
|
|
77
|
+
conversation_id: str
|
|
78
|
+
workspace: str
|
|
79
|
+
ledger: str
|
|
80
|
+
model: str | None
|
|
81
|
+
base_url: str | None
|
|
82
|
+
task: str | None
|
|
83
|
+
pid: int | None
|
|
84
|
+
started: float
|
|
85
|
+
ended: float | None
|
|
86
|
+
status: str
|
|
87
|
+
outcome: str | None
|
|
88
|
+
requests: int | None
|
|
89
|
+
detail: str | None
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def state(self) -> str:
|
|
93
|
+
"""`status`, except that a `running` row whose process is gone DIED."""
|
|
94
|
+
if self.status == "running" and self.pid:
|
|
95
|
+
from agentctl.runtime.lease import pid_alive
|
|
96
|
+
if not pid_alive(self.pid):
|
|
97
|
+
return "died"
|
|
98
|
+
return self.status
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _row(r: sqlite3.Row) -> Run:
|
|
102
|
+
return Run(**{k: r[k] for k in r.keys()})
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def start(conversation_id: str, workspace: str | Path, ledger: str | Path,
|
|
106
|
+
model: str | None, base_url: str | None, task: str | None) -> int | None:
|
|
107
|
+
"""Record a run beginning. Never raises: the index is a convenience, and a
|
|
108
|
+
run must not fail because the convenience could not be written."""
|
|
109
|
+
try:
|
|
110
|
+
with _db() as db:
|
|
111
|
+
cur = db.execute(
|
|
112
|
+
"INSERT INTO run(conversation_id, workspace, ledger, model, "
|
|
113
|
+
"base_url, task, pid, started, status) VALUES(?,?,?,?,?,?,?,?,?)",
|
|
114
|
+
(conversation_id, str(Path(workspace).resolve()), str(ledger),
|
|
115
|
+
model, base_url, (task or "")[:300] or None, os.getpid(),
|
|
116
|
+
time.time(), "running"))
|
|
117
|
+
return cur.lastrowid
|
|
118
|
+
except Exception: # noqa: BLE001
|
|
119
|
+
return None
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def end(run_id: int | None, status: str, outcome: str | None = None,
|
|
123
|
+
requests: int | None = None, detail: str | None = None) -> None:
|
|
124
|
+
if run_id is None:
|
|
125
|
+
return
|
|
126
|
+
try:
|
|
127
|
+
with _db() as db:
|
|
128
|
+
db.execute("UPDATE run SET ended=?, status=?, outcome=?, requests=?, "
|
|
129
|
+
"detail=? WHERE id=?",
|
|
130
|
+
(time.time(), status, outcome, requests,
|
|
131
|
+
(detail or "")[:500] or None, run_id))
|
|
132
|
+
except Exception: # noqa: BLE001
|
|
133
|
+
pass
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def recent(limit: int = 20, workspace: str | Path | None = None) -> list[Run]:
|
|
137
|
+
if not path().exists():
|
|
138
|
+
return []
|
|
139
|
+
with _db() as db:
|
|
140
|
+
if workspace is not None:
|
|
141
|
+
rows = db.execute("SELECT * FROM run WHERE workspace=? ORDER BY "
|
|
142
|
+
"started DESC LIMIT ?",
|
|
143
|
+
(str(Path(workspace).resolve()), limit)).fetchall()
|
|
144
|
+
else:
|
|
145
|
+
rows = db.execute("SELECT * FROM run ORDER BY started DESC LIMIT ?",
|
|
146
|
+
(limit,)).fetchall()
|
|
147
|
+
return [_row(r) for r in rows]
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def find(given: str | None = None, workspace: str | Path | None = None) -> Run:
|
|
151
|
+
"""The run `agentctl resume` means.
|
|
152
|
+
|
|
153
|
+
No argument: the latest run in this workspace, or else the latest
|
|
154
|
+
anywhere. Otherwise a conversation id or an unambiguous prefix of one.
|
|
155
|
+
Raises SystemExit with the reason, never guesses.
|
|
156
|
+
"""
|
|
157
|
+
if not path().exists():
|
|
158
|
+
raise SystemExit("no runs recorded yet. Start one: agentctl run \"<task>\"")
|
|
159
|
+
with _db() as db:
|
|
160
|
+
if given:
|
|
161
|
+
rows = db.execute("SELECT * FROM run WHERE replace(conversation_id,'-','') "
|
|
162
|
+
"LIKE ? ORDER BY started DESC",
|
|
163
|
+
(given.replace("-", "") + "%",)).fetchall()
|
|
164
|
+
convs = {r["conversation_id"] for r in rows}
|
|
165
|
+
if not rows:
|
|
166
|
+
raise SystemExit(f"no run with an id starting {given!r}. "
|
|
167
|
+
f"`agentctl status` lists them.")
|
|
168
|
+
if len(convs) > 1:
|
|
169
|
+
raise SystemExit(f"{given!r} matches {len(convs)} conversations: "
|
|
170
|
+
+ ", ".join(sorted(c[:13] for c in convs))
|
|
171
|
+
+ ". Give more characters.")
|
|
172
|
+
return _row(rows[0])
|
|
173
|
+
if workspace is not None:
|
|
174
|
+
r = db.execute("SELECT * FROM run WHERE workspace=? ORDER BY started "
|
|
175
|
+
"DESC LIMIT 1", (str(Path(workspace).resolve()),)).fetchone()
|
|
176
|
+
if r is not None:
|
|
177
|
+
return _row(r)
|
|
178
|
+
r = db.execute("SELECT * FROM run ORDER BY started DESC LIMIT 1").fetchone()
|
|
179
|
+
if r is None:
|
|
180
|
+
raise SystemExit("no runs recorded yet.")
|
|
181
|
+
return _row(r)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def ledgers() -> list[Path]:
|
|
185
|
+
"""Every ledger a recorded run wrote, newest first, that still exists."""
|
|
186
|
+
if not path().exists():
|
|
187
|
+
return []
|
|
188
|
+
with _db() as db:
|
|
189
|
+
rows = db.execute("SELECT ledger, MAX(started) s FROM run GROUP BY ledger "
|
|
190
|
+
"ORDER BY s DESC").fetchall()
|
|
191
|
+
return [Path(r["ledger"]) for r in rows if Path(r["ledger"]).exists()]
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
r"""Read-only subagents: the affordable slice of multi-agent.
|
|
2
|
+
|
|
3
|
+
`docs/0038` §4 skips multi-agent, and the reason is not cost — it is that four
|
|
4
|
+
correctness mechanisms assume a single writer. The lease keys on
|
|
5
|
+
`conversation_id`, `find_by_intent` is conversation-scoped, the git probe
|
|
6
|
+
infers "HEAD moved, so my commit landed", and `SubstitutionHandoff`
|
|
7
|
+
fingerprints carry no agent. Two workers editing one workspace break all four,
|
|
8
|
+
and fixing that was priced at 24 engineering days.
|
|
9
|
+
|
|
10
|
+
**A subagent that cannot produce an effect needs none of them.** No effect
|
|
11
|
+
means nothing to duplicate, nothing to reconcile, nothing to lease, and
|
|
12
|
+
nothing for a probe to be wrong about. That is the whole argument for this
|
|
13
|
+
module, and it holds only for exactly as long as "cannot produce an effect"
|
|
14
|
+
stays true — so that property is enforced twice, below.
|
|
15
|
+
|
|
16
|
+
## What this uses, and what it had to build
|
|
17
|
+
|
|
18
|
+
The SDK ships the *format*: `AgentDefinition` reads Markdown frontmatter with
|
|
19
|
+
`model`, `tools`, `max_iteration_per_run`, `max_budget_per_run`, hooks and a
|
|
20
|
+
system prompt, and `openhands.sdk.subagent.load` discovers them from a
|
|
21
|
+
directory. That is the Claude Code subagent format, and this module is the
|
|
22
|
+
first thing in `agentctl/` that calls it.
|
|
23
|
+
|
|
24
|
+
It does not ship the *runtime*. `TaskManager` appears in two docstrings and no
|
|
25
|
+
module; there is no `task` tool. So dispatch is ours.
|
|
26
|
+
|
|
27
|
+
## Two enforcements, not one
|
|
28
|
+
|
|
29
|
+
1. **`validate()` refuses** a definition asking for anything outside
|
|
30
|
+
`READ_ONLY_TOOLS`, and refuses one carrying `mcp_config` at all — an MCP
|
|
31
|
+
server is a tool with effects behind it that no allowlist here can see.
|
|
32
|
+
2. **The tool list handed to the agent is built from the constant**, never
|
|
33
|
+
from `definition.tools`. If validation were ever bypassed or wrongly
|
|
34
|
+
relaxed, the subagent still receives only read tools.
|
|
35
|
+
|
|
36
|
+
The second is what makes the first a check rather than the mechanism. A
|
|
37
|
+
safety property that depends on one function returning correctly is one
|
|
38
|
+
refactor away from not being a safety property.
|
|
39
|
+
|
|
40
|
+
## `max_budget_per_run` is inert here, deliberately not relied on
|
|
41
|
+
|
|
42
|
+
It compares `accumulated_cost`, which LiteLLM reports as `0.0` on every
|
|
43
|
+
unpriced free endpoint (`docs/0038` §2). A budget cap reading zero is not a
|
|
44
|
+
cap, so iteration count is the bound that actually binds.
|
|
45
|
+
"""
|
|
46
|
+
from __future__ import annotations
|
|
47
|
+
|
|
48
|
+
from pathlib import Path
|
|
49
|
+
from typing import Any
|
|
50
|
+
|
|
51
|
+
#: Tools a subagent may hold. `read_file` only.
|
|
52
|
+
#:
|
|
53
|
+
#: `execute_bash` is deliberately absent and is not a close call: a shell is
|
|
54
|
+
#: not a read tool, it is every tool. The capability matrix classifies
|
|
55
|
+
#: individual commands, but it is regex over a command string and an
|
|
56
|
+
#: interpreter (`python -c "..."`) is opaque to it by construction (README,
|
|
57
|
+
#: "What it does not do yet"). Handing a read-only agent a shell would make
|
|
58
|
+
#: this module's entire safety argument depend on that classifier being
|
|
59
|
+
#: complete, which it is not claimed to be.
|
|
60
|
+
READ_ONLY_TOOLS = frozenset({"read_file"})
|
|
61
|
+
|
|
62
|
+
#: Where definitions live, relative to the workspace.
|
|
63
|
+
AGENTS_DIR = Path(".agentctl") / "agents"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class NotReadOnly(ValueError):
|
|
67
|
+
"""A definition asked for something that could change the world."""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def validate(definition: Any) -> None:
|
|
71
|
+
"""Refuse anything that could produce an effect. Raises `NotReadOnly`.
|
|
72
|
+
|
|
73
|
+
Deliberately strict about `mcp_config`: an MCP server is an arbitrary
|
|
74
|
+
tool surface reached over a pipe, and nothing here can inspect what it
|
|
75
|
+
would do. `docs/0010` §8.2 already records that MCP amplifies the effect
|
|
76
|
+
problem; a read-only agent is not the place to find out.
|
|
77
|
+
"""
|
|
78
|
+
name = getattr(definition, "name", "<unnamed>")
|
|
79
|
+
|
|
80
|
+
tools = set(getattr(definition, "tools", None) or ())
|
|
81
|
+
forbidden = tools - READ_ONLY_TOOLS
|
|
82
|
+
if forbidden:
|
|
83
|
+
raise NotReadOnly(
|
|
84
|
+
f"subagent {name!r} asks for {sorted(forbidden)}, which can "
|
|
85
|
+
f"change the world. A read-only subagent may hold only "
|
|
86
|
+
f"{sorted(READ_ONLY_TOOLS)}. It runs outside the effect ledger "
|
|
87
|
+
f"precisely because it cannot produce an effect, so this is not "
|
|
88
|
+
f"a restriction that can be relaxed without building the "
|
|
89
|
+
f"24 days of prerequisites in docs/0038 §4.2 first."
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
if getattr(definition, "mcp_config", None):
|
|
93
|
+
raise NotReadOnly(
|
|
94
|
+
f"subagent {name!r} declares MCP servers. An MCP server is a "
|
|
95
|
+
f"tool surface this module cannot inspect, so it cannot be "
|
|
96
|
+
f"admitted to a read-only agent (`docs/0010` §8.2)."
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def discover(workspace: str | Path = ".") -> list[Any]:
|
|
101
|
+
"""Definitions under `<workspace>/.agentctl/agents/`, valid ones only.
|
|
102
|
+
|
|
103
|
+
An invalid definition is skipped rather than raising, so one bad file
|
|
104
|
+
does not hide every good one — but it is never silently skipped: the
|
|
105
|
+
reason is attached for the caller to print.
|
|
106
|
+
"""
|
|
107
|
+
from openhands.sdk.subagent.load import load_agents_from_dir
|
|
108
|
+
|
|
109
|
+
d = Path(workspace) / AGENTS_DIR
|
|
110
|
+
if not d.is_dir():
|
|
111
|
+
return []
|
|
112
|
+
out = []
|
|
113
|
+
for defn in load_agents_from_dir(d):
|
|
114
|
+
try:
|
|
115
|
+
validate(defn)
|
|
116
|
+
except NotReadOnly as e:
|
|
117
|
+
defn.__dict__["_agentctl_rejected"] = str(e)
|
|
118
|
+
out.append(defn)
|
|
119
|
+
return out
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def rejection(definition: Any) -> str | None:
|
|
123
|
+
"""Why `discover` would not run this one, or None."""
|
|
124
|
+
return definition.__dict__.get("_agentctl_rejected")
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def run(definition: Any, task: str, *, workspace: str | Path = ".",
|
|
128
|
+
model: str | None = None, api_key: str | None = None,
|
|
129
|
+
base_url: str | None = None, max_iterations: int = 15) -> str:
|
|
130
|
+
"""Run a read-only subagent on `task`. Returns what it reported.
|
|
131
|
+
|
|
132
|
+
No ledger, no gate, no lease — see the module docstring. `validate` runs
|
|
133
|
+
again here rather than trusting `discover`, because this is a public
|
|
134
|
+
entry point and the caller may have built the definition itself.
|
|
135
|
+
"""
|
|
136
|
+
from openhands.sdk import LLM, Conversation
|
|
137
|
+
|
|
138
|
+
from . import tools as rt
|
|
139
|
+
from .runner import _build_agent
|
|
140
|
+
|
|
141
|
+
validate(definition)
|
|
142
|
+
# The SDK resolves tools by name from a process-global registry, so they
|
|
143
|
+
# must be registered before an Agent naming them is constructed.
|
|
144
|
+
# `agentctl run` does this on its own path; a subagent started straight
|
|
145
|
+
# from the CLI has no parent run to have done it.
|
|
146
|
+
#
|
|
147
|
+
# `overwrite=False` is load-bearing. Seam C registers GATED tools under
|
|
148
|
+
# these same names, and a plain re-registration silently replaces them --
|
|
149
|
+
# un-gating every effect the parent's guard was wrapping, with nothing
|
|
150
|
+
# failing to say so. Never clobber a registration that already exists.
|
|
151
|
+
rt.register_all(overwrite=False)
|
|
152
|
+
|
|
153
|
+
ws = Path(workspace).resolve()
|
|
154
|
+
# `read_file` resolves against this, not against the Conversation's
|
|
155
|
+
# workspace argument -- deliberately, since it is configuration and never
|
|
156
|
+
# model input (`docs/0023` §3). Setting only the latter left the subagent
|
|
157
|
+
# hunting for `/gate.py` and, to its credit, refusing to guess.
|
|
158
|
+
#
|
|
159
|
+
# Scoped to this task, not the process: the env var version leaked a file
|
|
160
|
+
# from one workspace into a subagent scoped to another, and leaked the
|
|
161
|
+
# subagent's workspace back to the parent after it returned.
|
|
162
|
+
token = rt._scoped_workspace.set(str(ws))
|
|
163
|
+
try:
|
|
164
|
+
return _run_scoped(definition, task, ws, model, api_key, base_url,
|
|
165
|
+
max_iterations)
|
|
166
|
+
finally:
|
|
167
|
+
# Even when construction fails. A subagent that raised before it ever
|
|
168
|
+
# reached the model must not leave the parent's next `read_file`
|
|
169
|
+
# resolving against the subagent's workspace.
|
|
170
|
+
rt._scoped_workspace.reset(token)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _run_scoped(definition, task, ws, model, api_key, base_url,
|
|
174
|
+
max_iterations) -> str:
|
|
175
|
+
"""The body of `run`, with the workspace already scoped.
|
|
176
|
+
|
|
177
|
+
Split out so the scope is released by one `finally` covering every failure
|
|
178
|
+
path -- including `_build_agent` and `LLM(...)` construction, which is
|
|
179
|
+
where a broken model id or a missing key actually raises.
|
|
180
|
+
"""
|
|
181
|
+
from openhands.sdk import LLM, Conversation
|
|
182
|
+
|
|
183
|
+
from .runner import _build_agent
|
|
184
|
+
|
|
185
|
+
# NOT `definition.tools`. See the module docstring: the allowlist is the
|
|
186
|
+
# mechanism, the validation is the check.
|
|
187
|
+
tools = sorted(READ_ONLY_TOOLS)
|
|
188
|
+
|
|
189
|
+
declared = getattr(definition, "model", "inherit")
|
|
190
|
+
# `inherit` means "whatever the parent is using". Run straight from the
|
|
191
|
+
# CLI there is no parent, and handing `None` to a validated `LLM` raised a
|
|
192
|
+
# pydantic traceback from the very command `--init` tells you to type.
|
|
193
|
+
# Fall back to the same default `agentctl run` uses.
|
|
194
|
+
from .runner import DEFAULT_MODEL
|
|
195
|
+
chosen = (model or DEFAULT_MODEL) if declared in ("inherit", "", None) else declared
|
|
196
|
+
|
|
197
|
+
if api_key is None and not base_url:
|
|
198
|
+
# Direct calls need the key for THIS model's provider. Without this
|
|
199
|
+
# the subagent inherited whatever litellm happened to find, which is
|
|
200
|
+
# the wrong account as often as not. A proxy run needs no key at all:
|
|
201
|
+
# the proxy holds them (`docs/0038` §5.2).
|
|
202
|
+
from .runner import _key_for
|
|
203
|
+
api_key, _ = _key_for(chosen)
|
|
204
|
+
if base_url and api_key is None:
|
|
205
|
+
# litellm still wants something; the proxy ignores it.
|
|
206
|
+
api_key = "proxy-holds-the-credentials"
|
|
207
|
+
|
|
208
|
+
llm = LLM(model=chosen, api_key=api_key, base_url=base_url,
|
|
209
|
+
service_id=f"agentctl-subagent-{definition.name}",
|
|
210
|
+
temperature=0.0, num_retries=2, max_output_tokens=4096)
|
|
211
|
+
|
|
212
|
+
agent = _build_agent(llm, tools)
|
|
213
|
+
if (prompt := getattr(definition, "system_prompt", "")):
|
|
214
|
+
task = f"{prompt}\n\n---\n\n{task}"
|
|
215
|
+
|
|
216
|
+
from .runner import state_dir
|
|
217
|
+
conv = Conversation(
|
|
218
|
+
agent=agent, workspace=str(ws),
|
|
219
|
+
persistence_dir=str(state_dir(ws) / "subagents" / definition.name),
|
|
220
|
+
delete_on_close=False,
|
|
221
|
+
# The bound that actually binds; max_budget_per_run does not.
|
|
222
|
+
max_iteration_per_run=(
|
|
223
|
+
getattr(definition, "max_iteration_per_run", None) or max_iterations),
|
|
224
|
+
)
|
|
225
|
+
conv.send_message(task)
|
|
226
|
+
conv.run()
|
|
227
|
+
return _final_text(conv)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _content_text(item: Any) -> str:
|
|
231
|
+
"""Text out of one content part.
|
|
232
|
+
|
|
233
|
+
In memory a part is a `TextContent` with `.text`; once persisted and
|
|
234
|
+
reloaded it is a plain dict. Both shapes reach this function depending on
|
|
235
|
+
whether the caller is holding a live conversation or a resumed one, so it
|
|
236
|
+
handles both rather than assuming the happy one.
|
|
237
|
+
"""
|
|
238
|
+
if isinstance(item, dict):
|
|
239
|
+
return item.get("text") or ""
|
|
240
|
+
return getattr(item, "text", "") or ""
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _final_text(conv: Any) -> str:
|
|
244
|
+
"""The last thing the subagent *said*, as text.
|
|
245
|
+
|
|
246
|
+
Only agent messages count. An earlier version took the newest event
|
|
247
|
+
carrying any content at all and so returned the user's own prompt back --
|
|
248
|
+
which reads like an answer, is not one, and would have been spliced into a
|
|
249
|
+
parent agent's context as though the subagent had reported something.
|
|
250
|
+
|
|
251
|
+
Returns a plain marker rather than an empty string when there is nothing,
|
|
252
|
+
because a caller must be able to tell "it found nothing" from "it never
|
|
253
|
+
ran".
|
|
254
|
+
"""
|
|
255
|
+
try:
|
|
256
|
+
events = list(getattr(conv.state, "events", []) or [])
|
|
257
|
+
except Exception: # noqa: BLE001
|
|
258
|
+
return "[subagent produced no readable transcript]"
|
|
259
|
+
|
|
260
|
+
for ev in reversed(events):
|
|
261
|
+
msg = getattr(ev, "llm_message", None)
|
|
262
|
+
if msg is None:
|
|
263
|
+
continue
|
|
264
|
+
source = getattr(ev, "source", None)
|
|
265
|
+
role = getattr(msg, "role", None) or (
|
|
266
|
+
msg.get("role") if isinstance(msg, dict) else None)
|
|
267
|
+
if source != "agent" and role != "assistant":
|
|
268
|
+
continue
|
|
269
|
+
content = (getattr(msg, "content", None)
|
|
270
|
+
or (msg.get("content") if isinstance(msg, dict) else None))
|
|
271
|
+
parts = [t for t in (_content_text(c) for c in (content or [])) if t]
|
|
272
|
+
if parts:
|
|
273
|
+
return "\n".join(parts).strip()
|
|
274
|
+
return "[subagent produced no readable transcript]"
|