rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/routines.py
ADDED
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
"""Routines — recurring, pre-approved autonomous work (self-evolve phase 2).
|
|
2
|
+
|
|
3
|
+
A routine is a directory under the global $ROCKYCODE_HOME/routines/<name>/:
|
|
4
|
+
|
|
5
|
+
routine.toml — the DECLARATION: what it does, when it's due, and the
|
|
6
|
+
grant envelope the user approved once on the enable card
|
|
7
|
+
(network, tools, isolation, budgets). Declarative and
|
|
8
|
+
hand-editable; never mutated by runs.
|
|
9
|
+
SKILL.md — the HOW: the playbook the runner follows.
|
|
10
|
+
state.json — the RUNTIME state: last run, lease spend, run history.
|
|
11
|
+
Machine-owned, separate on purpose so the declaration
|
|
12
|
+
stays reviewable.
|
|
13
|
+
|
|
14
|
+
Scheduling is catch-up-on-launch (no daemon): at launch, due routines show a
|
|
15
|
+
card — click to run. `auto = true` is a LEASE, not a switch (locked design,
|
|
16
|
+
2026-07-17): it expires after at most MAX_LEASE_DAYS or when the lease
|
|
17
|
+
budget is spent, whichever first, then the routine falls back to
|
|
18
|
+
click-to-run until the lease is renewed (one click, spend shown). Trust
|
|
19
|
+
decays; it must be re-earned.
|
|
20
|
+
|
|
21
|
+
Execution (slice 3) is exec-shaped: headless engine run where the approver
|
|
22
|
+
IS the grant envelope — out-of-grant means a clean stop and a card at next
|
|
23
|
+
launch, never a mid-run hang. Every run writes a trajectory with
|
|
24
|
+
project_id + runner="routine", so the dream grades routines like chats.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import json
|
|
29
|
+
import os
|
|
30
|
+
import time
|
|
31
|
+
import tomllib
|
|
32
|
+
from dataclasses import dataclass, field
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import Optional
|
|
35
|
+
|
|
36
|
+
MAX_LEASE_DAYS = 7 # hard ceiling — a lease is never longer than this
|
|
37
|
+
CADENCES = {"daily": 86_400.0, "weekly": 7 * 86_400.0}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def routines_dir() -> Path:
|
|
41
|
+
base = os.environ.get("ROCKYCODE_HOME")
|
|
42
|
+
root = Path(base).expanduser() if base else Path.home() / ".rockycode"
|
|
43
|
+
return root / "routines"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class Routine:
|
|
48
|
+
name: str
|
|
49
|
+
description: str = ""
|
|
50
|
+
cadence: str = "daily" # daily | weekly
|
|
51
|
+
prompt: str = "" # the task line handed to the runner
|
|
52
|
+
workdir: str = "" # where it runs
|
|
53
|
+
project_id: str = "" # groups its trajectories with a project
|
|
54
|
+
output_dir: str = "" # where results land (relative to workdir)
|
|
55
|
+
# -- the grant envelope (approved once, on the enable card) --------------
|
|
56
|
+
network: bool = False
|
|
57
|
+
tools: list[str] = field(default_factory=list)
|
|
58
|
+
isolation: bool = False # worktree + branch delivery (mutating routines)
|
|
59
|
+
budget_run: float = 0.10 # per-run spend cap (session currency)
|
|
60
|
+
max_steps: int = 30 # per-run step cap (exec: never unbounded)
|
|
61
|
+
# -- the auto lease -------------------------------------------------------
|
|
62
|
+
auto: bool = False
|
|
63
|
+
lease_deadline: float = 0.0 # unix; 0 = no lease ever granted
|
|
64
|
+
budget_lease: float = 1.00 # cross-run cap while the lease is active
|
|
65
|
+
enabled: bool = True
|
|
66
|
+
path: Optional[Path] = None # the routine's directory
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _emit_toml(r: Routine) -> str:
|
|
70
|
+
def s(v: str) -> str:
|
|
71
|
+
return '"' + v.replace("\\", "\\\\").replace('"', '\\"') + '"'
|
|
72
|
+
|
|
73
|
+
lines = [
|
|
74
|
+
f"name = {s(r.name)}",
|
|
75
|
+
f"description = {s(r.description)}",
|
|
76
|
+
f"cadence = {s(r.cadence)}",
|
|
77
|
+
f"prompt = {s(r.prompt)}",
|
|
78
|
+
f"workdir = {s(r.workdir)}",
|
|
79
|
+
f"project_id = {s(r.project_id)}",
|
|
80
|
+
f"output_dir = {s(r.output_dir)}",
|
|
81
|
+
f"network = {'true' if r.network else 'false'}",
|
|
82
|
+
"tools = [" + ", ".join(s(t) for t in r.tools) + "]",
|
|
83
|
+
f"isolation = {'true' if r.isolation else 'false'}",
|
|
84
|
+
f"budget_run = {r.budget_run}",
|
|
85
|
+
f"max_steps = {r.max_steps}",
|
|
86
|
+
f"auto = {'true' if r.auto else 'false'}",
|
|
87
|
+
f"lease_deadline = {r.lease_deadline}",
|
|
88
|
+
f"budget_lease = {r.budget_lease}",
|
|
89
|
+
f"enabled = {'true' if r.enabled else 'false'}",
|
|
90
|
+
]
|
|
91
|
+
return "\n".join(lines) + "\n"
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
@dataclass
|
|
95
|
+
class RoutineState:
|
|
96
|
+
last_run: float = 0.0 # unix; 0 = never ran
|
|
97
|
+
lease_spent: float = 0.0 # spend since the CURRENT lease started
|
|
98
|
+
runs: list[dict] = field(default_factory=list) # {sid, t, cost, status}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class RoutineStore:
|
|
102
|
+
"""Directory-per-routine under one global root. The toml is the contract,
|
|
103
|
+
state.json the odometer — same files-are-the-truth rule as everything."""
|
|
104
|
+
|
|
105
|
+
def __init__(self, root: Optional[Path] = None) -> None:
|
|
106
|
+
self.root = root or routines_dir()
|
|
107
|
+
|
|
108
|
+
# -- declarations ---------------------------------------------------------
|
|
109
|
+
|
|
110
|
+
def list(self, project_id: Optional[str] = None) -> list[Routine]:
|
|
111
|
+
if not self.root.is_dir():
|
|
112
|
+
return []
|
|
113
|
+
out = []
|
|
114
|
+
for d in sorted(self.root.iterdir()):
|
|
115
|
+
r = self.load(d.name)
|
|
116
|
+
if r is None or not r.enabled:
|
|
117
|
+
continue
|
|
118
|
+
if project_id is None or r.project_id in ("", project_id):
|
|
119
|
+
out.append(r)
|
|
120
|
+
return out
|
|
121
|
+
|
|
122
|
+
def load(self, name: str) -> Optional[Routine]:
|
|
123
|
+
path = self.root / name / "routine.toml"
|
|
124
|
+
try:
|
|
125
|
+
data = tomllib.loads(path.read_text(encoding="utf-8"))
|
|
126
|
+
except (OSError, tomllib.TOMLDecodeError):
|
|
127
|
+
return None
|
|
128
|
+
known = {f for f in Routine.__dataclass_fields__ if f != "path"}
|
|
129
|
+
clean = {k: v for k, v in data.items() if k in known}
|
|
130
|
+
r = Routine(**{"name": name, **clean})
|
|
131
|
+
if r.cadence not in CADENCES:
|
|
132
|
+
r.cadence = "daily"
|
|
133
|
+
r.path = self.root / name
|
|
134
|
+
return r
|
|
135
|
+
|
|
136
|
+
def save(self, r: Routine, skill_md: str = "") -> Path:
|
|
137
|
+
d = self.root / r.name
|
|
138
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
139
|
+
(d / "routine.toml").write_text(_emit_toml(r), encoding="utf-8")
|
|
140
|
+
if skill_md:
|
|
141
|
+
(d / "SKILL.md").write_text(skill_md, encoding="utf-8")
|
|
142
|
+
r.path = d
|
|
143
|
+
return d
|
|
144
|
+
|
|
145
|
+
# -- runtime state --------------------------------------------------------
|
|
146
|
+
|
|
147
|
+
def state(self, r: Routine) -> RoutineState:
|
|
148
|
+
try:
|
|
149
|
+
raw = json.loads((self.root / r.name / "state.json").read_text(encoding="utf-8"))
|
|
150
|
+
return RoutineState(
|
|
151
|
+
last_run=float(raw.get("last_run", 0.0)),
|
|
152
|
+
lease_spent=float(raw.get("lease_spent", 0.0)),
|
|
153
|
+
runs=list(raw.get("runs", [])),
|
|
154
|
+
)
|
|
155
|
+
except (OSError, ValueError, json.JSONDecodeError):
|
|
156
|
+
return RoutineState()
|
|
157
|
+
|
|
158
|
+
def _write_state(self, r: Routine, st: RoutineState) -> None:
|
|
159
|
+
(self.root / r.name).mkdir(parents=True, exist_ok=True)
|
|
160
|
+
(self.root / r.name / "state.json").write_text(json.dumps({
|
|
161
|
+
"last_run": st.last_run, "lease_spent": st.lease_spent,
|
|
162
|
+
"runs": st.runs[-50:], # a bounded odometer, not a log store
|
|
163
|
+
}, indent=1), encoding="utf-8")
|
|
164
|
+
|
|
165
|
+
# -- due + lease ----------------------------------------------------------
|
|
166
|
+
|
|
167
|
+
def due(self, project_id: Optional[str] = None, now: Optional[float] = None) -> list[Routine]:
|
|
168
|
+
"""Routines whose cadence has elapsed. Missed runs never stack — a
|
|
169
|
+
routine is due once, no matter how long the machine slept."""
|
|
170
|
+
now = time.time() if now is None else now
|
|
171
|
+
out = []
|
|
172
|
+
for r in self.list(project_id):
|
|
173
|
+
st = self.state(r)
|
|
174
|
+
if now - st.last_run >= CADENCES[r.cadence]:
|
|
175
|
+
out.append(r)
|
|
176
|
+
return out
|
|
177
|
+
|
|
178
|
+
def lease_active(self, r: Routine, now: Optional[float] = None) -> bool:
|
|
179
|
+
"""The auto lease holds only while BOTH the deadline and the lease
|
|
180
|
+
budget hold — expiry of either falls back to click-to-run."""
|
|
181
|
+
now = time.time() if now is None else now
|
|
182
|
+
if not (r.auto and r.enabled):
|
|
183
|
+
return False
|
|
184
|
+
if now >= r.lease_deadline:
|
|
185
|
+
return False
|
|
186
|
+
return self.state(r).lease_spent < r.budget_lease
|
|
187
|
+
|
|
188
|
+
def grant_lease(self, r: Routine, days: float, budget: float) -> Routine:
|
|
189
|
+
"""Start (or renew) the auto lease: at most MAX_LEASE_DAYS, always
|
|
190
|
+
with a budget. Renewal resets the lease odometer."""
|
|
191
|
+
days = max(0.0, min(float(days), MAX_LEASE_DAYS))
|
|
192
|
+
r.auto = True
|
|
193
|
+
r.lease_deadline = time.time() + days * 86_400.0
|
|
194
|
+
r.budget_lease = float(budget)
|
|
195
|
+
st = self.state(r)
|
|
196
|
+
st.lease_spent = 0.0
|
|
197
|
+
self.save(r)
|
|
198
|
+
self._write_state(r, st)
|
|
199
|
+
return r
|
|
200
|
+
|
|
201
|
+
def revoke_lease(self, r: Routine) -> Routine:
|
|
202
|
+
r.auto = False
|
|
203
|
+
r.lease_deadline = 0.0
|
|
204
|
+
self.save(r)
|
|
205
|
+
return r
|
|
206
|
+
|
|
207
|
+
def record_run(self, r: Routine, *, session_id: str, cost: float, status: str,
|
|
208
|
+
now: Optional[float] = None) -> RoutineState:
|
|
209
|
+
"""Odometer tick after a run: last_run moves, lease spend accumulates,
|
|
210
|
+
the run lands in the (bounded) history with its trajectory id — the
|
|
211
|
+
dream finds the full story there."""
|
|
212
|
+
now = time.time() if now is None else now
|
|
213
|
+
st = self.state(r)
|
|
214
|
+
st.last_run = now
|
|
215
|
+
st.lease_spent += max(0.0, float(cost))
|
|
216
|
+
st.runs.append({"sid": session_id, "t": now, "cost": cost, "status": status})
|
|
217
|
+
self._write_state(r, st)
|
|
218
|
+
return st
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
222
|
+
# the runner — exec's headless machinery, driven by a routine's contract
|
|
223
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _grant_tokens(tools: list[str]) -> frozenset[str]:
|
|
227
|
+
"""routine.toml lists what the user granted; exec's HeadlessApprover
|
|
228
|
+
speaks grant tokens — bare tool names become "tool:<name>", anything
|
|
229
|
+
already token-shaped (a bash safety-pattern name, "tool:x") passes as-is."""
|
|
230
|
+
return frozenset(t if ":" in t or "-" in t else f"tool:{t}" for t in tools)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
async def run_routine(store: RoutineStore, r: Routine, *, model: str,
|
|
234
|
+
client=None, registry=None, err=None) -> dict:
|
|
235
|
+
"""Run one routine through exec (sandboxed ALWAYS — no host fallback for
|
|
236
|
+
unattended work; Docker down = the run fails loudly, the card says so).
|
|
237
|
+
Settles the odometer from the result envelope and writes last-run.md
|
|
238
|
+
beside the routine for the "ready" line. Returns a summary dict."""
|
|
239
|
+
from rockycode.engine.headless import run_exec
|
|
240
|
+
from rockycode.pricing import UsageLedger
|
|
241
|
+
from rockycode.session import get_project
|
|
242
|
+
|
|
243
|
+
workdir = Path(r.workdir).expanduser() if r.workdir else Path.cwd()
|
|
244
|
+
skill = ""
|
|
245
|
+
if r.path is not None:
|
|
246
|
+
try:
|
|
247
|
+
skill = (r.path / "SKILL.md").read_text(encoding="utf-8", errors="replace")
|
|
248
|
+
except OSError:
|
|
249
|
+
skill = ""
|
|
250
|
+
out_note = (f"\nWrite your results into the directory '{r.output_dir}' "
|
|
251
|
+
f"(relative to the working directory)." if r.output_dir else "")
|
|
252
|
+
prompt = f"{r.prompt}{out_note}"
|
|
253
|
+
if skill.strip():
|
|
254
|
+
prompt += f"\n\nFollow this playbook:\n\n{skill.strip()}"
|
|
255
|
+
|
|
256
|
+
project = get_project(workdir)
|
|
257
|
+
lines: list[dict] = []
|
|
258
|
+
code = await run_exec(
|
|
259
|
+
prompt=prompt, model=model, workdir=workdir,
|
|
260
|
+
grants=_grant_tokens(r.tools), max_steps=r.max_steps,
|
|
261
|
+
originator=f"routine:{r.name}",
|
|
262
|
+
sandbox=True, network=r.network,
|
|
263
|
+
write=lines.append, client=client, registry=registry, err=err,
|
|
264
|
+
extra_meta={"runner": "routine", "routine": r.name,
|
|
265
|
+
"project_id": r.project_id or project.id,
|
|
266
|
+
"project_name": project.name},
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
meta = next((l for l in lines if l.get("type") == "meta"), {})
|
|
270
|
+
result = next((l for l in lines if l.get("type") == "result"), {})
|
|
271
|
+
status = {0: "done", 2: "blocked", 3: "budget"}.get(code, "error")
|
|
272
|
+
ledger = UsageLedger()
|
|
273
|
+
if result.get("usage"):
|
|
274
|
+
ledger.add(model, result["usage"])
|
|
275
|
+
cost = ledger.cost("usd")
|
|
276
|
+
store.record_run(r, session_id=meta.get("session", ""), cost=cost, status=status)
|
|
277
|
+
|
|
278
|
+
summary = str(result.get("summary", "") or result.get("error", ""))
|
|
279
|
+
if r.path is not None:
|
|
280
|
+
try:
|
|
281
|
+
(r.path / "last-run.md").write_text(
|
|
282
|
+
f"# {r.name} — last run\n\nstatus: {status} · cost: ${cost:.4f} · "
|
|
283
|
+
f"session: {meta.get('session', '?')}\n\n{summary}\n", encoding="utf-8")
|
|
284
|
+
except OSError:
|
|
285
|
+
pass
|
|
286
|
+
return {"status": status, "cost": cost, "summary": summary,
|
|
287
|
+
"session": meta.get("session", ""), "blocked_on": result.get("blocked_on")}
|
|
File without changes
|
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
"""The rockycode harness runner: the agent loop on SWE-bench tasks.
|
|
2
|
+
|
|
3
|
+
Per task: pull the official SWE-bench image → start a container → Rocky
|
|
4
|
+
works inside it (bash/read/write/edit via docker exec) → `git diff` is the
|
|
5
|
+
prediction. Same images the scorer uses (namespace "swebench" on Docker
|
|
6
|
+
Hub), so nothing is built locally and the cache is shared.
|
|
7
|
+
|
|
8
|
+
Tasks run sequentially on purpose: containers run under emulation on
|
|
9
|
+
Apple Silicon, and one task at a time keeps API spend observable.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import asyncio
|
|
14
|
+
import json
|
|
15
|
+
from datetime import datetime
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Optional
|
|
18
|
+
|
|
19
|
+
from rich.console import Console
|
|
20
|
+
|
|
21
|
+
from rockycode.banner import amaze, confused, fail, info
|
|
22
|
+
from rockycode.engine.container import DockerSession, build_session_registry, extract_patch
|
|
23
|
+
from rockycode.palette import RED
|
|
24
|
+
from rockycode.engine.events import (
|
|
25
|
+
Compacted,
|
|
26
|
+
EngineError,
|
|
27
|
+
TextDelta,
|
|
28
|
+
ToolFinished,
|
|
29
|
+
ToolStarted,
|
|
30
|
+
TurnFinished,
|
|
31
|
+
)
|
|
32
|
+
from rockycode.engine.loop import Engine
|
|
33
|
+
from rockycode.prompts.rocky import BENCH_TASK, ROCKY_SYSTEM
|
|
34
|
+
from rockycode.runners.data import load_verified
|
|
35
|
+
|
|
36
|
+
PREDICTIONS_DIR = Path("results") / "predictions"
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _image_for(instance: dict) -> str:
|
|
40
|
+
from swebench.harness.test_spec.test_spec import make_test_spec
|
|
41
|
+
|
|
42
|
+
return make_test_spec(instance, namespace="swebench").instance_image_key
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
async def _ensure_image(image: str, console: Console) -> bool:
|
|
46
|
+
proc = await asyncio.create_subprocess_exec(
|
|
47
|
+
"docker", "image", "inspect", image,
|
|
48
|
+
stdout=asyncio.subprocess.DEVNULL, stderr=asyncio.subprocess.DEVNULL,
|
|
49
|
+
)
|
|
50
|
+
if await proc.wait() == 0:
|
|
51
|
+
return True
|
|
52
|
+
info(console, f"pulling {image} [dim](first time per task, can be ~1GB)[/dim]")
|
|
53
|
+
proc = await asyncio.create_subprocess_exec(
|
|
54
|
+
"docker", "pull", "--platform", "linux/amd64", image,
|
|
55
|
+
stdout=asyncio.subprocess.DEVNULL, stderr=asyncio.subprocess.PIPE,
|
|
56
|
+
)
|
|
57
|
+
_, err = await proc.communicate()
|
|
58
|
+
if proc.returncode != 0:
|
|
59
|
+
fail(console, f"pull failed: {err.decode(errors='replace').strip().splitlines()[-1]}")
|
|
60
|
+
return False
|
|
61
|
+
return True
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _task_prompt(instance: dict) -> str:
|
|
65
|
+
return BENCH_TASK.format(
|
|
66
|
+
repo=instance.get("repo", "unknown"),
|
|
67
|
+
problem_statement=instance.get("problem_statement", ""),
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
async def _run_instance(
|
|
72
|
+
instance: dict,
|
|
73
|
+
*,
|
|
74
|
+
model: str,
|
|
75
|
+
thinking: bool,
|
|
76
|
+
reasoning_effort: str,
|
|
77
|
+
max_tokens: int,
|
|
78
|
+
context_window: int,
|
|
79
|
+
max_steps: int,
|
|
80
|
+
system_prompt: str,
|
|
81
|
+
prompt_name: str,
|
|
82
|
+
prompt_sha: str,
|
|
83
|
+
console: Console,
|
|
84
|
+
) -> dict:
|
|
85
|
+
iid = instance["instance_id"]
|
|
86
|
+
image = _image_for(instance)
|
|
87
|
+
|
|
88
|
+
if not await _ensure_image(image, console):
|
|
89
|
+
return {"instance_id": iid, "patch": "", "steps": 0, "usage": {}, "error": "image pull failed"}
|
|
90
|
+
|
|
91
|
+
session = await DockerSession.start(image)
|
|
92
|
+
try:
|
|
93
|
+
registry = build_session_registry(session)
|
|
94
|
+
# Same generated "# Tools this session" section as chat, built from
|
|
95
|
+
# THIS registry — so bench prompts list exactly the container tools
|
|
96
|
+
# (no phantom web/artifact advertisements burning budget steps). No
|
|
97
|
+
# language/env/date appends: bench stays English + byte-reproducible.
|
|
98
|
+
from rockycode.prompts.rocky import tools_section
|
|
99
|
+
engine = Engine(
|
|
100
|
+
model=model,
|
|
101
|
+
thinking=thinking,
|
|
102
|
+
reasoning_effort=reasoning_effort,
|
|
103
|
+
max_tokens=max_tokens,
|
|
104
|
+
context_window=context_window,
|
|
105
|
+
max_steps=max_steps,
|
|
106
|
+
system_prompt=system_prompt + tools_section(registry),
|
|
107
|
+
registry=registry,
|
|
108
|
+
trajectory_meta={
|
|
109
|
+
"runner": "rockycode",
|
|
110
|
+
"instance_id": iid,
|
|
111
|
+
"image": image,
|
|
112
|
+
"prompt_name": prompt_name,
|
|
113
|
+
"prompt_sha": prompt_sha,
|
|
114
|
+
},
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
steps, usage, err = 0, {}, None
|
|
118
|
+
async for ev in engine.run_turn(_task_prompt(instance)):
|
|
119
|
+
if isinstance(ev, ToolStarted):
|
|
120
|
+
arg = (ev.args.get("raw") or "")[:70].replace("\n", " ")
|
|
121
|
+
console.print(f" [dim]⚒ {ev.tool} {arg}[/dim]")
|
|
122
|
+
elif isinstance(ev, ToolFinished) and not ev.ok:
|
|
123
|
+
first = ev.output.strip().splitlines()[0][:90] if ev.output.strip() else ""
|
|
124
|
+
console.print(f" [dim]✗ {ev.tool}: {first}[/dim]")
|
|
125
|
+
elif isinstance(ev, Compacted):
|
|
126
|
+
console.print(
|
|
127
|
+
f" [dim]♻ compacted ({ev.strategy}): "
|
|
128
|
+
f"~{ev.tokens_before:,} → ~{ev.tokens_after:,} tokens[/dim]"
|
|
129
|
+
)
|
|
130
|
+
elif isinstance(ev, TextDelta):
|
|
131
|
+
pass # final summary text; trajectory has it
|
|
132
|
+
elif isinstance(ev, EngineError):
|
|
133
|
+
err = ev.message
|
|
134
|
+
console.print(f" [{RED}]✗ {ev.message}[/]")
|
|
135
|
+
elif isinstance(ev, TurnFinished):
|
|
136
|
+
steps, usage = ev.steps, ev.usage
|
|
137
|
+
|
|
138
|
+
patch = await extract_patch(session)
|
|
139
|
+
engine.trajectory.outcome(
|
|
140
|
+
{
|
|
141
|
+
"instance_id": iid,
|
|
142
|
+
"steps": steps,
|
|
143
|
+
"patch_chars": len(patch),
|
|
144
|
+
"engine_error": err,
|
|
145
|
+
"usage": usage,
|
|
146
|
+
}
|
|
147
|
+
)
|
|
148
|
+
return {"instance_id": iid, "patch": patch, "steps": steps, "usage": usage, "error": err}
|
|
149
|
+
finally:
|
|
150
|
+
await session.stop()
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
async def _run_all(
|
|
154
|
+
instances: list[dict],
|
|
155
|
+
*,
|
|
156
|
+
model: str,
|
|
157
|
+
thinking: bool,
|
|
158
|
+
reasoning_effort: str,
|
|
159
|
+
max_tokens: int,
|
|
160
|
+
context_window: int,
|
|
161
|
+
max_steps: int,
|
|
162
|
+
token_budget: int,
|
|
163
|
+
system_prompt: str,
|
|
164
|
+
prompt_name: str,
|
|
165
|
+
prompt_sha: str,
|
|
166
|
+
out_path: Path,
|
|
167
|
+
safe_model: str,
|
|
168
|
+
console: Console,
|
|
169
|
+
) -> None:
|
|
170
|
+
totals: dict[str, int] = {"prompt_tokens": 0, "completion_tokens": 0, "prompt_cache_hit_tokens": 0}
|
|
171
|
+
n_patch = 0
|
|
172
|
+
|
|
173
|
+
with out_path.open("w") as f:
|
|
174
|
+
for i, instance in enumerate(instances, 1):
|
|
175
|
+
spent = totals["prompt_tokens"] + totals["completion_tokens"]
|
|
176
|
+
if token_budget and spent >= token_budget:
|
|
177
|
+
info(console, f"token budget reached ({spent:,} >= {token_budget:,}) — stopping.")
|
|
178
|
+
break
|
|
179
|
+
|
|
180
|
+
iid = instance["instance_id"]
|
|
181
|
+
console.print(f"\n[bold]task {i}/{len(instances)}[/bold] · {iid}")
|
|
182
|
+
try:
|
|
183
|
+
result = await _run_instance(
|
|
184
|
+
instance,
|
|
185
|
+
model=model,
|
|
186
|
+
thinking=thinking,
|
|
187
|
+
reasoning_effort=reasoning_effort,
|
|
188
|
+
max_tokens=max_tokens,
|
|
189
|
+
context_window=context_window,
|
|
190
|
+
max_steps=max_steps,
|
|
191
|
+
system_prompt=system_prompt,
|
|
192
|
+
prompt_name=prompt_name,
|
|
193
|
+
prompt_sha=prompt_sha,
|
|
194
|
+
console=console,
|
|
195
|
+
)
|
|
196
|
+
except Exception as e: # noqa: BLE001 — one task must not kill the run
|
|
197
|
+
fail(console, f"{iid}: {type(e).__name__}: {e}")
|
|
198
|
+
result = {"instance_id": iid, "patch": "", "steps": 0, "usage": {}, "error": str(e)}
|
|
199
|
+
|
|
200
|
+
patch = result["patch"]
|
|
201
|
+
if patch.strip():
|
|
202
|
+
n_patch += 1
|
|
203
|
+
amaze(console, f"patch ready · {result['steps']} steps · {len(patch):,} chars")
|
|
204
|
+
else:
|
|
205
|
+
confused(console, f"no patch produced ({result.get('error') or 'agent stopped without changes'})")
|
|
206
|
+
|
|
207
|
+
for k in totals:
|
|
208
|
+
totals[k] += result["usage"].get(k, 0) or 0
|
|
209
|
+
|
|
210
|
+
f.write(json.dumps({
|
|
211
|
+
"instance_id": iid,
|
|
212
|
+
"model_name_or_path": f"rockycode-{safe_model}",
|
|
213
|
+
"model_patch": patch,
|
|
214
|
+
}) + "\n")
|
|
215
|
+
f.flush()
|
|
216
|
+
|
|
217
|
+
console.print(
|
|
218
|
+
f"\n[dim]· {n_patch}/{len(instances)} tasks produced a patch · "
|
|
219
|
+
f"{totals['prompt_tokens']:,} in / {totals['completion_tokens']:,} out · "
|
|
220
|
+
f"cache hit {totals['prompt_cache_hit_tokens']:,}[/dim]"
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def run(
|
|
225
|
+
model: str,
|
|
226
|
+
instance_ids: Optional[list[str]],
|
|
227
|
+
console: Console,
|
|
228
|
+
*,
|
|
229
|
+
thinking: bool = True,
|
|
230
|
+
reasoning_effort: str = "max",
|
|
231
|
+
max_tokens: int = 16384,
|
|
232
|
+
context_window: int = 131_072,
|
|
233
|
+
max_steps: int = 50,
|
|
234
|
+
token_budget: int = 0,
|
|
235
|
+
system_prompt: Optional[str] = None,
|
|
236
|
+
prompt_name: str = "rocky-builtin",
|
|
237
|
+
prompt_sha: str = "",
|
|
238
|
+
task_label: str = "tasks",
|
|
239
|
+
output_dir: Path = PREDICTIONS_DIR,
|
|
240
|
+
) -> Path:
|
|
241
|
+
"""Run the harness on the requested instances. Returns predictions path."""
|
|
242
|
+
if system_prompt is None:
|
|
243
|
+
system_prompt = ROCKY_SYSTEM
|
|
244
|
+
ds = load_verified(console, instance_ids)
|
|
245
|
+
instances = list(ds)
|
|
246
|
+
console.print(f" → {len(instances)} tasks for the rockycode harness\n")
|
|
247
|
+
|
|
248
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
249
|
+
safe_model = model.replace("/", "-")
|
|
250
|
+
# prompt + task-set + timestamp: no run ever overwrites another
|
|
251
|
+
stamp = datetime.now().strftime("%Y%m%d-%H%M")
|
|
252
|
+
out_path = output_dir / f"rockycode-{safe_model}-{prompt_name}-{task_label}-{stamp}.jsonl"
|
|
253
|
+
|
|
254
|
+
asyncio.run(
|
|
255
|
+
_run_all(
|
|
256
|
+
instances,
|
|
257
|
+
model=model,
|
|
258
|
+
thinking=thinking,
|
|
259
|
+
reasoning_effort=reasoning_effort,
|
|
260
|
+
max_tokens=max_tokens,
|
|
261
|
+
context_window=context_window,
|
|
262
|
+
max_steps=max_steps,
|
|
263
|
+
token_budget=token_budget,
|
|
264
|
+
system_prompt=system_prompt,
|
|
265
|
+
prompt_name=prompt_name,
|
|
266
|
+
prompt_sha=prompt_sha,
|
|
267
|
+
out_path=out_path,
|
|
268
|
+
safe_model=safe_model,
|
|
269
|
+
console=console,
|
|
270
|
+
)
|
|
271
|
+
)
|
|
272
|
+
info(console, f"predictions at {out_path}; trajectories at .rockycode/trajectories/")
|
|
273
|
+
return out_path
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Shared SWE-bench Verified dataset loading for all runners."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
from rich.console import Console
|
|
7
|
+
|
|
8
|
+
from rockycode.banner import confused, fail, info
|
|
9
|
+
from rockycode.palette import BLUE
|
|
10
|
+
|
|
11
|
+
_NETWORK_HINTS = (
|
|
12
|
+
"connection", "timeout", "name resolution", "network", "unreachable",
|
|
13
|
+
"dns", "ssl", "max retries", "getaddrinfo", "proxy", "refused",
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
DATASET = "princeton-nlp/SWE-bench_Verified"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def load_verified(console: Console, instance_ids: Optional[list[str]]):
|
|
20
|
+
"""Load Verified (filtered to instance_ids if given). Exits with a
|
|
21
|
+
friendly message on network failure."""
|
|
22
|
+
import datasets
|
|
23
|
+
|
|
24
|
+
# HF's tqdm bars (download + filter) collide with our Rich status line —
|
|
25
|
+
# one renderer must own the terminal.
|
|
26
|
+
datasets.disable_progress_bars()
|
|
27
|
+
|
|
28
|
+
try:
|
|
29
|
+
with console.status(
|
|
30
|
+
f"loading [{BLUE}]{DATASET}[/] "
|
|
31
|
+
"[dim](first run pulls ~hundreds of MB from huggingface)[/dim]",
|
|
32
|
+
spinner="dots",
|
|
33
|
+
):
|
|
34
|
+
ds = datasets.load_dataset(DATASET, split="test")
|
|
35
|
+
except Exception as e:
|
|
36
|
+
fail(console, "could not load SWE-bench Verified from huggingface.")
|
|
37
|
+
msg = str(e).lower()
|
|
38
|
+
if any(k in msg for k in _NETWORK_HINTS):
|
|
39
|
+
confused(
|
|
40
|
+
console,
|
|
41
|
+
"looks like a network issue. check internet / vpn / proxy / "
|
|
42
|
+
"is huggingface.co reachable?",
|
|
43
|
+
)
|
|
44
|
+
else:
|
|
45
|
+
confused(console, f"underlying error: {type(e).__name__}: {e}")
|
|
46
|
+
info(console, "isolate the load step to debug:")
|
|
47
|
+
info(
|
|
48
|
+
console,
|
|
49
|
+
" uv run python -c \"from datasets import load_dataset; "
|
|
50
|
+
"load_dataset('princeton-nlp/SWE-bench_Verified', split='test')\"",
|
|
51
|
+
)
|
|
52
|
+
raise SystemExit(1)
|
|
53
|
+
|
|
54
|
+
if instance_ids:
|
|
55
|
+
wanted = set(instance_ids)
|
|
56
|
+
ds = ds.filter(lambda r: r["instance_id"] in wanted)
|
|
57
|
+
missing = wanted - set(ds["instance_id"])
|
|
58
|
+
if missing:
|
|
59
|
+
confused(console, f"{len(missing)} requested IDs not in Verified: {sorted(missing)[:3]}…")
|
|
60
|
+
|
|
61
|
+
return ds
|