rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,481 @@
|
|
|
1
|
+
"""Headless exec mode: rocky as a subagent for OTHER coding agents.
|
|
2
|
+
|
|
3
|
+
`rockycode exec "task"` runs ONE task to completion with no human present,
|
|
4
|
+
under a strict machine contract (design: docs pending, see the exec command
|
|
5
|
+
help). It is the fifth consumer of Engine.run_turn(), after the TUI, bench,
|
|
6
|
+
serve, and goal mode.
|
|
7
|
+
|
|
8
|
+
stdout — JSONL only, one event per line. First line = `meta` (schema,
|
|
9
|
+
session rk_ id, effective profile), last line = the `result`
|
|
10
|
+
envelope. Anything human-shaped goes to stderr.
|
|
11
|
+
approver — the caller can't answer prompts, so goal mode's command
|
|
12
|
+
classifier decides: safe/moderate runs; an ask-tier action stops
|
|
13
|
+
the run with a `blocked_on` grant token the caller can re-invoke
|
|
14
|
+
with (exit 2 — the resume-with-grant loop); block-tier is refused
|
|
15
|
+
UNCONDITIONALLY, no flag disables it, and the model must find a
|
|
16
|
+
reversible path.
|
|
17
|
+
exit — 0 done | 1 error | 2 blocked on a grant | 3 step budget spent.
|
|
18
|
+
|
|
19
|
+
The calling agent is the supervisor: the envelope carries EVIDENCE (files
|
|
20
|
+
changed, commands run, refusals) — never verdicts — so the caller does its
|
|
21
|
+
own verification. Rocky never blocks waiting and never silently self-limits.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import json
|
|
26
|
+
import os
|
|
27
|
+
import re
|
|
28
|
+
import subprocess
|
|
29
|
+
import sys
|
|
30
|
+
import traceback
|
|
31
|
+
from pathlib import Path
|
|
32
|
+
from typing import Callable, Optional
|
|
33
|
+
|
|
34
|
+
from rockycode.engine.events import (
|
|
35
|
+
Compacted,
|
|
36
|
+
EngineError,
|
|
37
|
+
TextDelta,
|
|
38
|
+
ThinkingDelta,
|
|
39
|
+
ToolFinished,
|
|
40
|
+
ToolStarted,
|
|
41
|
+
TurnFinished,
|
|
42
|
+
TurnStarted,
|
|
43
|
+
)
|
|
44
|
+
from rockycode.banner import fail, info
|
|
45
|
+
from rockycode.engine.loop import Engine
|
|
46
|
+
from rockycode.engine.safety import classify_command
|
|
47
|
+
|
|
48
|
+
SCHEMA = "rockyexec/1"
|
|
49
|
+
|
|
50
|
+
EXIT_DONE = 0
|
|
51
|
+
EXIT_ERROR = 1
|
|
52
|
+
EXIT_BLOCKED = 2
|
|
53
|
+
EXIT_BUDGET = 3
|
|
54
|
+
|
|
55
|
+
# Headless-only ask tier: deletion. classify_command deliberately allows
|
|
56
|
+
# `rm -rf build/` because goal mode runs on an isolated copy — exec runs on
|
|
57
|
+
# the REAL working tree, so every delete needs an explicit `delete` grant.
|
|
58
|
+
_DELETE = re.compile(
|
|
59
|
+
r"(?:^|[\s;&|(])(?:rm|rmdir|unlink)\b"
|
|
60
|
+
r"|\bgit\s+clean\b"
|
|
61
|
+
r"|\bfind\b[^\n|;]*\s-delete\b"
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
# Cap for string values inside emitted tool args (a write_file `content` can
|
|
65
|
+
# be an entire file; the caller only needs enough to see what happened).
|
|
66
|
+
_ARG_CLIP = 2000
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class HeadlessApprover:
|
|
70
|
+
"""The approver for runs where the CALLER, not a human, supervises.
|
|
71
|
+
|
|
72
|
+
Never waits: an ask-tier action records `blocked` (the exec driver stops
|
|
73
|
+
the run and exits 2 so the caller can re-invoke with the grant) and a
|
|
74
|
+
block-tier action records `refused` (the model gets the denial and keeps
|
|
75
|
+
going). The grant token IS the safety pattern name, so `blocked_on.grant`
|
|
76
|
+
tells the caller exactly what to pass to --allow.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
def __init__(self, registry: dict, grants: frozenset[str] = frozenset()):
|
|
80
|
+
self.registry = registry
|
|
81
|
+
self.grants = grants
|
|
82
|
+
self.blocked: Optional[dict] = None # first ask-tier hit ends the run
|
|
83
|
+
self.refused: list[dict] = [] # block-tier denials (evidence)
|
|
84
|
+
|
|
85
|
+
async def __call__(self, name: str, args: dict) -> bool:
|
|
86
|
+
if name == "bash":
|
|
87
|
+
cmd = str(args.get("command", ""))
|
|
88
|
+
v = classify_command(cmd)
|
|
89
|
+
if v.action == "block":
|
|
90
|
+
self.refused.append({"command": cmd, "reason": v.reason})
|
|
91
|
+
return False
|
|
92
|
+
if v.action == "ask" and v.pattern not in self.grants:
|
|
93
|
+
self.blocked = {"grant": v.pattern, "reason": v.reason, "command": cmd}
|
|
94
|
+
return False
|
|
95
|
+
if _DELETE.search(cmd) and "delete" not in self.grants:
|
|
96
|
+
self.blocked = {
|
|
97
|
+
"grant": "delete",
|
|
98
|
+
"reason": "deletes files on the real working tree",
|
|
99
|
+
"command": cmd,
|
|
100
|
+
}
|
|
101
|
+
return False
|
|
102
|
+
return True
|
|
103
|
+
# File tools are jailed to workdir + --allow-dir by the registry, so
|
|
104
|
+
# safe/moderate tiers are exactly workspace-write. Anything unclassified
|
|
105
|
+
# (dynamic MCP tools etc.) is gated like an ask.
|
|
106
|
+
risk = getattr(self.registry.get(name), "risk", "risky")
|
|
107
|
+
if risk in ("safe", "moderate"):
|
|
108
|
+
return True
|
|
109
|
+
# Honor the grant the docstring promises: "tool:<name>" tokens work
|
|
110
|
+
# exactly like bash pattern grants (the routine envelope relies on it —
|
|
111
|
+
# this path was unexercised until routines re-invoked with grants).
|
|
112
|
+
if f"tool:{name}" in self.grants:
|
|
113
|
+
return True
|
|
114
|
+
self.blocked = {"grant": f"tool:{name}", "reason": f"unclassified risky tool '{name}'"}
|
|
115
|
+
return False
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _scrub(text: str) -> str:
|
|
119
|
+
"""Redact secrets + the home path from anything crossing stdout. Unlike
|
|
120
|
+
serve there is no debug bypass — the consumer is always another program."""
|
|
121
|
+
try:
|
|
122
|
+
from rockycode.engine.redact import redact
|
|
123
|
+
text = redact(text)
|
|
124
|
+
except Exception: # noqa: BLE001 — best-effort
|
|
125
|
+
pass
|
|
126
|
+
try:
|
|
127
|
+
home = str(Path.home())
|
|
128
|
+
if home and home != "/":
|
|
129
|
+
text = text.replace(home, "~")
|
|
130
|
+
except Exception: # noqa: BLE001
|
|
131
|
+
pass
|
|
132
|
+
return text
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _parse_args(ev_args: dict) -> dict:
|
|
136
|
+
"""ToolStarted carries the model's arguments UNPARSED as {'raw': json-str}
|
|
137
|
+
(both loop.py branches yield before decoding). Decode for the caller;
|
|
138
|
+
anything malformed passes through as-is."""
|
|
139
|
+
raw = ev_args.get("raw")
|
|
140
|
+
if isinstance(raw, str) and len(ev_args) == 1:
|
|
141
|
+
try:
|
|
142
|
+
parsed = json.loads(raw) if raw.strip() else {}
|
|
143
|
+
if isinstance(parsed, dict):
|
|
144
|
+
return parsed
|
|
145
|
+
except json.JSONDecodeError:
|
|
146
|
+
pass
|
|
147
|
+
return ev_args
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _clip_args(args: dict) -> dict:
|
|
151
|
+
out = {}
|
|
152
|
+
for k, v in args.items():
|
|
153
|
+
if isinstance(v, str) and len(v) > _ARG_CLIP:
|
|
154
|
+
v = v[:_ARG_CLIP] + f"… [{len(v) - _ARG_CLIP} chars clipped]"
|
|
155
|
+
out[k] = _scrub(v) if isinstance(v, str) else v
|
|
156
|
+
return out
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def event_to_line(ev, *, include_thinking: bool = False) -> Optional[dict]:
|
|
160
|
+
"""One engine Event → one JSONL dict, or None to skip. Thinking deltas are
|
|
161
|
+
presentation-only and omitted by default — the engine's request history is
|
|
162
|
+
a separate layer and never contains reasoning_content either way."""
|
|
163
|
+
if isinstance(ev, TurnStarted):
|
|
164
|
+
return {"type": "turn.started"}
|
|
165
|
+
if isinstance(ev, (ThinkingDelta, TextDelta)):
|
|
166
|
+
return None # deltas are buffered by drive() — one event per block,
|
|
167
|
+
# not a 2023-style line per token
|
|
168
|
+
if isinstance(ev, ToolStarted):
|
|
169
|
+
return {"type": "tool.started", "call_id": ev.call_id, "tool": ev.tool,
|
|
170
|
+
"args": _clip_args(_parse_args(ev.args))}
|
|
171
|
+
if isinstance(ev, ToolFinished):
|
|
172
|
+
return {"type": "tool.finished", "call_id": ev.call_id, "tool": ev.tool,
|
|
173
|
+
"ok": ev.ok, "duration_s": round(ev.duration_s, 3),
|
|
174
|
+
**_clip_output(_scrub(ev.output))}
|
|
175
|
+
if isinstance(ev, Compacted):
|
|
176
|
+
return {"type": "compacted", "strategy": ev.strategy}
|
|
177
|
+
if isinstance(ev, TurnFinished):
|
|
178
|
+
return {"type": "turn.finished", "steps": ev.steps, "usage": ev.usage}
|
|
179
|
+
if isinstance(ev, EngineError):
|
|
180
|
+
return {"type": "error", "message": _scrub(ev.message)}
|
|
181
|
+
return None # StateChanged / ContextReminder: pet-and-TUI noise
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
_OUTPUT_CAP = 2_000 # chars — the stream is a receipt; full output lives in the trajectory
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _clip_output(text: str) -> dict:
|
|
188
|
+
if len(text) <= _OUTPUT_CAP:
|
|
189
|
+
return {"output": text}
|
|
190
|
+
return {"output": text[:_OUTPUT_CAP], "output_truncated": True,
|
|
191
|
+
"output_chars": len(text)}
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _scrub_obj(obj):
|
|
195
|
+
"""Recursively _scrub every string in a dict/list (the result envelope's
|
|
196
|
+
evidence fields — commands, refused, blocked_on — which the caller parses)."""
|
|
197
|
+
if isinstance(obj, str):
|
|
198
|
+
return _scrub(obj)
|
|
199
|
+
if isinstance(obj, dict):
|
|
200
|
+
return {k: _scrub_obj(v) for k, v in obj.items()}
|
|
201
|
+
if isinstance(obj, list):
|
|
202
|
+
return [_scrub_obj(v) for v in obj]
|
|
203
|
+
return obj
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _stdout_line(obj: dict) -> None:
|
|
207
|
+
"""stdout carries ONLY event JSON — the purity contract callers parse by."""
|
|
208
|
+
sys.stdout.write(json.dumps(obj, ensure_ascii=False, default=str) + "\n")
|
|
209
|
+
sys.stdout.flush()
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _caller_attribution(originator: str) -> dict:
|
|
213
|
+
"""Best-effort record of WHO invoked this run, for the trajectory's audit
|
|
214
|
+
trail: a self-declared originator plus the parent process. Post-hoc
|
|
215
|
+
attribution is the realistic defense for unattended abuse — see the
|
|
216
|
+
exec design memory."""
|
|
217
|
+
info: dict = {"originator": originator or os.getenv("ROCKYCODE_ORIGINATOR", "")}
|
|
218
|
+
try:
|
|
219
|
+
ppid = os.getppid()
|
|
220
|
+
info["caller_pid"] = ppid
|
|
221
|
+
name = subprocess.run(
|
|
222
|
+
["ps", "-p", str(ppid), "-o", "comm="],
|
|
223
|
+
capture_output=True, text=True, timeout=2,
|
|
224
|
+
).stdout.strip()
|
|
225
|
+
if name:
|
|
226
|
+
info["caller_process"] = name
|
|
227
|
+
except Exception: # noqa: BLE001 — attribution must never break the run
|
|
228
|
+
pass
|
|
229
|
+
return info
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _final_text(history: list[dict]) -> str:
|
|
233
|
+
"""The last plain assistant message — the model's answer to the caller."""
|
|
234
|
+
for m in reversed(history):
|
|
235
|
+
if m.get("role") == "assistant" and m.get("content") and not m.get("tool_calls"):
|
|
236
|
+
return str(m["content"])
|
|
237
|
+
return ""
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def build_exec_engine(
|
|
241
|
+
*,
|
|
242
|
+
model: str,
|
|
243
|
+
workdir: Path,
|
|
244
|
+
allowed_roots: tuple[Path, ...] = (),
|
|
245
|
+
grants: frozenset[str] = frozenset(),
|
|
246
|
+
max_steps: int = 30,
|
|
247
|
+
originator: str = "",
|
|
248
|
+
client=None,
|
|
249
|
+
registry=None,
|
|
250
|
+
sandbox_meta: Optional[dict] = None,
|
|
251
|
+
extra_meta: Optional[dict] = None,
|
|
252
|
+
) -> tuple[Engine, HeadlessApprover]:
|
|
253
|
+
"""An Engine wired for headless exec: capped steps, headless approver,
|
|
254
|
+
attribution in the trajectory meta. client/registry injection is for tests
|
|
255
|
+
(fake DeepSeek stream, recording tools) — same seams bench uses.
|
|
256
|
+
extra_meta lets a caller stamp identity onto the trajectory — the routine
|
|
257
|
+
runner adds project_id + runner="routine" so the dream can grade runs."""
|
|
258
|
+
engine = Engine(
|
|
259
|
+
model=model,
|
|
260
|
+
workdir=workdir,
|
|
261
|
+
allowed_roots=allowed_roots,
|
|
262
|
+
max_steps=max_steps,
|
|
263
|
+
client=client,
|
|
264
|
+
registry=registry,
|
|
265
|
+
trajectory_meta={"headless_exec": True, **(sandbox_meta or {}),
|
|
266
|
+
**_caller_attribution(originator), **(extra_meta or {})},
|
|
267
|
+
)
|
|
268
|
+
engine._exec_sandbox_meta = sandbox_meta or {"sandbox": False, "network": False}
|
|
269
|
+
approver = HeadlessApprover(engine.registry, grants)
|
|
270
|
+
engine.approver = approver
|
|
271
|
+
return engine, approver
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
async def drive(
|
|
275
|
+
engine: Engine,
|
|
276
|
+
approver: HeadlessApprover,
|
|
277
|
+
prompt: str,
|
|
278
|
+
*,
|
|
279
|
+
write: Optional[Callable[[dict], None]] = None,
|
|
280
|
+
include_thinking: bool = False,
|
|
281
|
+
output_last_message: Optional[Path] = None,
|
|
282
|
+
) -> int:
|
|
283
|
+
"""Run one exec turn: meta line, event stream, result envelope, exit code."""
|
|
284
|
+
from rockycode.session import public_id
|
|
285
|
+
write = write or _stdout_line
|
|
286
|
+
rk = public_id(engine.trajectory.session_id)
|
|
287
|
+
|
|
288
|
+
write({
|
|
289
|
+
"type": "meta",
|
|
290
|
+
"schema": SCHEMA,
|
|
291
|
+
"session": rk,
|
|
292
|
+
"model": engine.model,
|
|
293
|
+
"workdir": _scrub(str(engine.workdir)),
|
|
294
|
+
# False warns the caller: overwrites here have no git safety net.
|
|
295
|
+
"git": (engine.workdir / ".git").exists(),
|
|
296
|
+
"profile": {
|
|
297
|
+
"mode": "workspace-write",
|
|
298
|
+
"grants": sorted(approver.grants),
|
|
299
|
+
"max_steps": engine.max_steps,
|
|
300
|
+
**getattr(engine, "_exec_sandbox_meta", {"sandbox": False, "network": False}),
|
|
301
|
+
},
|
|
302
|
+
})
|
|
303
|
+
|
|
304
|
+
files_changed: list[str] = []
|
|
305
|
+
commands: list[dict] = []
|
|
306
|
+
started_cmds: dict[str, str] = {} # call_id → bash command
|
|
307
|
+
steps = 0
|
|
308
|
+
usage: dict = {}
|
|
309
|
+
budget_hit = False
|
|
310
|
+
error_msg = ""
|
|
311
|
+
|
|
312
|
+
text_buf: list[str] = []
|
|
313
|
+
think_buf: list[str] = []
|
|
314
|
+
|
|
315
|
+
def _flush_deltas() -> None:
|
|
316
|
+
# one coherent event per block — flushed when something else happens
|
|
317
|
+
if think_buf:
|
|
318
|
+
write({"type": "thinking", "text": "".join(think_buf)})
|
|
319
|
+
think_buf.clear()
|
|
320
|
+
if text_buf:
|
|
321
|
+
write({"type": "text", "text": "".join(text_buf)})
|
|
322
|
+
text_buf.clear()
|
|
323
|
+
|
|
324
|
+
gen = engine.run_turn(prompt)
|
|
325
|
+
try:
|
|
326
|
+
async for ev in gen:
|
|
327
|
+
if isinstance(ev, TextDelta):
|
|
328
|
+
text_buf.append(ev.text)
|
|
329
|
+
continue
|
|
330
|
+
if isinstance(ev, ThinkingDelta):
|
|
331
|
+
if include_thinking:
|
|
332
|
+
think_buf.append(ev.text)
|
|
333
|
+
continue
|
|
334
|
+
if isinstance(ev, (ToolStarted, TurnFinished, EngineError)):
|
|
335
|
+
_flush_deltas()
|
|
336
|
+
if isinstance(ev, ToolStarted):
|
|
337
|
+
args = _parse_args(ev.args)
|
|
338
|
+
if ev.tool == "bash":
|
|
339
|
+
started_cmds[ev.call_id] = str(args.get("command", ""))
|
|
340
|
+
elif ev.tool in ("write_file", "edit_file") and args.get("path"):
|
|
341
|
+
p = str(args["path"])
|
|
342
|
+
if p not in files_changed:
|
|
343
|
+
files_changed.append(p)
|
|
344
|
+
elif isinstance(ev, ToolFinished) and ev.call_id in started_cmds:
|
|
345
|
+
commands.append({"command": started_cmds.pop(ev.call_id), "ok": ev.ok})
|
|
346
|
+
elif isinstance(ev, TurnFinished):
|
|
347
|
+
steps, usage = ev.steps, ev.usage
|
|
348
|
+
elif isinstance(ev, EngineError):
|
|
349
|
+
# Coupling: loop.py words its step-cap error "step limit …" —
|
|
350
|
+
# that's a budget stop (resumable checkpoint), not a failure.
|
|
351
|
+
if ev.message.startswith("step limit"):
|
|
352
|
+
budget_hit = True
|
|
353
|
+
else:
|
|
354
|
+
error_msg = ev.message
|
|
355
|
+
line = event_to_line(ev, include_thinking=include_thinking)
|
|
356
|
+
if line:
|
|
357
|
+
write(line)
|
|
358
|
+
if approver.blocked:
|
|
359
|
+
break # fail fast: the caller decides, then resumes with a grant
|
|
360
|
+
except Exception as e: # noqa: BLE001 — envelope + exit 1, never a stdout traceback
|
|
361
|
+
error_msg = f"{type(e).__name__}: {e}"
|
|
362
|
+
print(traceback.format_exc(), file=sys.stderr, flush=True)
|
|
363
|
+
finally:
|
|
364
|
+
# Run the engine's cleanup (history repair + trajectory stubs) NOW —
|
|
365
|
+
# a broken-out async-for otherwise defers it to GC.
|
|
366
|
+
await gen.aclose()
|
|
367
|
+
_flush_deltas() # anything still buffered (blocked break / early end)
|
|
368
|
+
|
|
369
|
+
if approver.blocked:
|
|
370
|
+
status, code = "blocked", EXIT_BLOCKED
|
|
371
|
+
elif error_msg:
|
|
372
|
+
status, code = "error", EXIT_ERROR
|
|
373
|
+
elif budget_hit:
|
|
374
|
+
status, code = "budget", EXIT_BUDGET
|
|
375
|
+
else:
|
|
376
|
+
status, code = "done", EXIT_DONE
|
|
377
|
+
|
|
378
|
+
summary = _scrub(_final_text(engine.history))
|
|
379
|
+
# F5: the caller PARSES these fields, so they get the same scrub as the
|
|
380
|
+
# event stream — a token embedded in a command URL must not leak here.
|
|
381
|
+
result = {
|
|
382
|
+
"type": "result",
|
|
383
|
+
"status": status,
|
|
384
|
+
"session": rk,
|
|
385
|
+
"summary": summary,
|
|
386
|
+
"blocked_on": _scrub_obj(approver.blocked),
|
|
387
|
+
"evidence": {
|
|
388
|
+
"files_changed": [_scrub(p) for p in files_changed],
|
|
389
|
+
"commands": _scrub_obj(commands),
|
|
390
|
+
"refused": _scrub_obj(approver.refused),
|
|
391
|
+
},
|
|
392
|
+
"steps": steps,
|
|
393
|
+
"usage": usage,
|
|
394
|
+
}
|
|
395
|
+
if error_msg:
|
|
396
|
+
result["error"] = _scrub(error_msg)
|
|
397
|
+
write(result)
|
|
398
|
+
engine.trajectory.outcome({k: v for k, v in result.items() if k != "type"})
|
|
399
|
+
# Heuristic reward line LAST (self-evolve): readers take the last outcome
|
|
400
|
+
# record and the dream's judge gate keys on source="heuristic" — this is
|
|
401
|
+
# what makes exec/routine runs gradeable. The envelope above stays in the
|
|
402
|
+
# file as evidence.
|
|
403
|
+
engine.finalize_outcome()
|
|
404
|
+
|
|
405
|
+
if output_last_message is not None:
|
|
406
|
+
try:
|
|
407
|
+
output_last_message.write_text(summary, encoding="utf-8")
|
|
408
|
+
except OSError as e:
|
|
409
|
+
print(f"could not write --output-last-message: {e}", file=sys.stderr)
|
|
410
|
+
|
|
411
|
+
return code
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
async def run_exec(
|
|
415
|
+
*,
|
|
416
|
+
prompt: str,
|
|
417
|
+
model: str,
|
|
418
|
+
workdir: Path,
|
|
419
|
+
allowed_roots: tuple[Path, ...] = (),
|
|
420
|
+
grants: frozenset[str] = frozenset(),
|
|
421
|
+
max_steps: int = 30,
|
|
422
|
+
originator: str = "",
|
|
423
|
+
include_thinking: bool = False,
|
|
424
|
+
output_last_message: Optional[Path] = None,
|
|
425
|
+
write: Optional[Callable[[dict], None]] = None,
|
|
426
|
+
client=None,
|
|
427
|
+
registry=None,
|
|
428
|
+
sandbox: bool = True,
|
|
429
|
+
network: bool = False,
|
|
430
|
+
err=None,
|
|
431
|
+
extra_meta: Optional[dict] = None,
|
|
432
|
+
) -> int:
|
|
433
|
+
"""Provision the sandbox (default), build the engine, drive one task.
|
|
434
|
+
|
|
435
|
+
The task can originate from an untrusted source (an issue, a page a
|
|
436
|
+
delegating agent read), so by default every tool runs inside a Docker
|
|
437
|
+
container with NO network: a `rm -rf /` hits the container's root, home
|
|
438
|
+
secrets aren't mounted, and there's no egress to exfiltrate over. The
|
|
439
|
+
command classifier stays on top as defense-in-depth, its real designed
|
|
440
|
+
role. --no-sandbox is the explicit, loud host escape hatch.
|
|
441
|
+
"""
|
|
442
|
+
sb = None
|
|
443
|
+
sandbox_meta = {"sandbox": False, "network": False}
|
|
444
|
+
if sandbox and registry is None: # registry injected → tests, already wired
|
|
445
|
+
try:
|
|
446
|
+
from rockycode.engine.sandbox import ChatSandbox, build_sandbox_registry
|
|
447
|
+
sb = await ChatSandbox.start(workdir, network=network)
|
|
448
|
+
registry = build_sandbox_registry(sb)
|
|
449
|
+
sandbox_meta = {"sandbox": True, "network": bool(network),
|
|
450
|
+
"container": sb.container_id[:12]}
|
|
451
|
+
if err is not None:
|
|
452
|
+
err.print(f"[dim]· sandbox: on (container {sb.container_id[:12]}…) · "
|
|
453
|
+
f"network {'on' if network else 'off'}[/]")
|
|
454
|
+
except Exception as e: # noqa: BLE001 — Docker missing/broken: fail clearly
|
|
455
|
+
if err is not None:
|
|
456
|
+
fail(err, f"exec needs Docker for the sandbox — {type(e).__name__}: {e}")
|
|
457
|
+
info(err, "start Docker Desktop, or pass --no-sandbox to run on the "
|
|
458
|
+
"host (UNSAFE for untrusted input).")
|
|
459
|
+
return EXIT_ERROR
|
|
460
|
+
elif not sandbox and registry is None and err is not None:
|
|
461
|
+
err.print("[bold yellow]⚠ --no-sandbox: tools run on the HOST with no "
|
|
462
|
+
"isolation. The command classifier is a denylist, not a security "
|
|
463
|
+
"boundary — only use with input you fully trust.[/]")
|
|
464
|
+
|
|
465
|
+
try:
|
|
466
|
+
engine, approver = build_exec_engine(
|
|
467
|
+
model=model, workdir=workdir, allowed_roots=allowed_roots, grants=grants,
|
|
468
|
+
max_steps=max_steps, originator=originator, client=client, registry=registry,
|
|
469
|
+
sandbox_meta=sandbox_meta, extra_meta=extra_meta,
|
|
470
|
+
)
|
|
471
|
+
return await drive(
|
|
472
|
+
engine, approver, prompt,
|
|
473
|
+
write=write, include_thinking=include_thinking,
|
|
474
|
+
output_last_message=output_last_message,
|
|
475
|
+
)
|
|
476
|
+
finally:
|
|
477
|
+
if sb is not None:
|
|
478
|
+
try:
|
|
479
|
+
await sb.stop()
|
|
480
|
+
except Exception: # noqa: BLE001 — best-effort teardown
|
|
481
|
+
pass
|