rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/routines.py ADDED
@@ -0,0 +1,287 @@
1
+ """Routines — recurring, pre-approved autonomous work (self-evolve phase 2).
2
+
3
+ A routine is a directory under the global $ROCKYCODE_HOME/routines/<name>/:
4
+
5
+ routine.toml — the DECLARATION: what it does, when it's due, and the
6
+ grant envelope the user approved once on the enable card
7
+ (network, tools, isolation, budgets). Declarative and
8
+ hand-editable; never mutated by runs.
9
+ SKILL.md — the HOW: the playbook the runner follows.
10
+ state.json — the RUNTIME state: last run, lease spend, run history.
11
+ Machine-owned, separate on purpose so the declaration
12
+ stays reviewable.
13
+
14
+ Scheduling is catch-up-on-launch (no daemon): at launch, due routines show a
15
+ card — click to run. `auto = true` is a LEASE, not a switch (locked design,
16
+ 2026-07-17): it expires after at most MAX_LEASE_DAYS or when the lease
17
+ budget is spent, whichever first, then the routine falls back to
18
+ click-to-run until the lease is renewed (one click, spend shown). Trust
19
+ decays; it must be re-earned.
20
+
21
+ Execution (slice 3) is exec-shaped: headless engine run where the approver
22
+ IS the grant envelope — out-of-grant means a clean stop and a card at next
23
+ launch, never a mid-run hang. Every run writes a trajectory with
24
+ project_id + runner="routine", so the dream grades routines like chats.
25
+ """
26
+ from __future__ import annotations
27
+
28
+ import json
29
+ import os
30
+ import time
31
+ import tomllib
32
+ from dataclasses import dataclass, field
33
+ from pathlib import Path
34
+ from typing import Optional
35
+
36
+ MAX_LEASE_DAYS = 7 # hard ceiling — a lease is never longer than this
37
+ CADENCES = {"daily": 86_400.0, "weekly": 7 * 86_400.0}
38
+
39
+
40
+ def routines_dir() -> Path:
41
+ base = os.environ.get("ROCKYCODE_HOME")
42
+ root = Path(base).expanduser() if base else Path.home() / ".rockycode"
43
+ return root / "routines"
44
+
45
+
46
+ @dataclass
47
+ class Routine:
48
+ name: str
49
+ description: str = ""
50
+ cadence: str = "daily" # daily | weekly
51
+ prompt: str = "" # the task line handed to the runner
52
+ workdir: str = "" # where it runs
53
+ project_id: str = "" # groups its trajectories with a project
54
+ output_dir: str = "" # where results land (relative to workdir)
55
+ # -- the grant envelope (approved once, on the enable card) --------------
56
+ network: bool = False
57
+ tools: list[str] = field(default_factory=list)
58
+ isolation: bool = False # worktree + branch delivery (mutating routines)
59
+ budget_run: float = 0.10 # per-run spend cap (session currency)
60
+ max_steps: int = 30 # per-run step cap (exec: never unbounded)
61
+ # -- the auto lease -------------------------------------------------------
62
+ auto: bool = False
63
+ lease_deadline: float = 0.0 # unix; 0 = no lease ever granted
64
+ budget_lease: float = 1.00 # cross-run cap while the lease is active
65
+ enabled: bool = True
66
+ path: Optional[Path] = None # the routine's directory
67
+
68
+
69
+ def _emit_toml(r: Routine) -> str:
70
+ def s(v: str) -> str:
71
+ return '"' + v.replace("\\", "\\\\").replace('"', '\\"') + '"'
72
+
73
+ lines = [
74
+ f"name = {s(r.name)}",
75
+ f"description = {s(r.description)}",
76
+ f"cadence = {s(r.cadence)}",
77
+ f"prompt = {s(r.prompt)}",
78
+ f"workdir = {s(r.workdir)}",
79
+ f"project_id = {s(r.project_id)}",
80
+ f"output_dir = {s(r.output_dir)}",
81
+ f"network = {'true' if r.network else 'false'}",
82
+ "tools = [" + ", ".join(s(t) for t in r.tools) + "]",
83
+ f"isolation = {'true' if r.isolation else 'false'}",
84
+ f"budget_run = {r.budget_run}",
85
+ f"max_steps = {r.max_steps}",
86
+ f"auto = {'true' if r.auto else 'false'}",
87
+ f"lease_deadline = {r.lease_deadline}",
88
+ f"budget_lease = {r.budget_lease}",
89
+ f"enabled = {'true' if r.enabled else 'false'}",
90
+ ]
91
+ return "\n".join(lines) + "\n"
92
+
93
+
94
+ @dataclass
95
+ class RoutineState:
96
+ last_run: float = 0.0 # unix; 0 = never ran
97
+ lease_spent: float = 0.0 # spend since the CURRENT lease started
98
+ runs: list[dict] = field(default_factory=list) # {sid, t, cost, status}
99
+
100
+
101
+ class RoutineStore:
102
+ """Directory-per-routine under one global root. The toml is the contract,
103
+ state.json the odometer — same files-are-the-truth rule as everything."""
104
+
105
+ def __init__(self, root: Optional[Path] = None) -> None:
106
+ self.root = root or routines_dir()
107
+
108
+ # -- declarations ---------------------------------------------------------
109
+
110
+ def list(self, project_id: Optional[str] = None) -> list[Routine]:
111
+ if not self.root.is_dir():
112
+ return []
113
+ out = []
114
+ for d in sorted(self.root.iterdir()):
115
+ r = self.load(d.name)
116
+ if r is None or not r.enabled:
117
+ continue
118
+ if project_id is None or r.project_id in ("", project_id):
119
+ out.append(r)
120
+ return out
121
+
122
+ def load(self, name: str) -> Optional[Routine]:
123
+ path = self.root / name / "routine.toml"
124
+ try:
125
+ data = tomllib.loads(path.read_text(encoding="utf-8"))
126
+ except (OSError, tomllib.TOMLDecodeError):
127
+ return None
128
+ known = {f for f in Routine.__dataclass_fields__ if f != "path"}
129
+ clean = {k: v for k, v in data.items() if k in known}
130
+ r = Routine(**{"name": name, **clean})
131
+ if r.cadence not in CADENCES:
132
+ r.cadence = "daily"
133
+ r.path = self.root / name
134
+ return r
135
+
136
+ def save(self, r: Routine, skill_md: str = "") -> Path:
137
+ d = self.root / r.name
138
+ d.mkdir(parents=True, exist_ok=True)
139
+ (d / "routine.toml").write_text(_emit_toml(r), encoding="utf-8")
140
+ if skill_md:
141
+ (d / "SKILL.md").write_text(skill_md, encoding="utf-8")
142
+ r.path = d
143
+ return d
144
+
145
+ # -- runtime state --------------------------------------------------------
146
+
147
+ def state(self, r: Routine) -> RoutineState:
148
+ try:
149
+ raw = json.loads((self.root / r.name / "state.json").read_text(encoding="utf-8"))
150
+ return RoutineState(
151
+ last_run=float(raw.get("last_run", 0.0)),
152
+ lease_spent=float(raw.get("lease_spent", 0.0)),
153
+ runs=list(raw.get("runs", [])),
154
+ )
155
+ except (OSError, ValueError, json.JSONDecodeError):
156
+ return RoutineState()
157
+
158
+ def _write_state(self, r: Routine, st: RoutineState) -> None:
159
+ (self.root / r.name).mkdir(parents=True, exist_ok=True)
160
+ (self.root / r.name / "state.json").write_text(json.dumps({
161
+ "last_run": st.last_run, "lease_spent": st.lease_spent,
162
+ "runs": st.runs[-50:], # a bounded odometer, not a log store
163
+ }, indent=1), encoding="utf-8")
164
+
165
+ # -- due + lease ----------------------------------------------------------
166
+
167
+ def due(self, project_id: Optional[str] = None, now: Optional[float] = None) -> list[Routine]:
168
+ """Routines whose cadence has elapsed. Missed runs never stack — a
169
+ routine is due once, no matter how long the machine slept."""
170
+ now = time.time() if now is None else now
171
+ out = []
172
+ for r in self.list(project_id):
173
+ st = self.state(r)
174
+ if now - st.last_run >= CADENCES[r.cadence]:
175
+ out.append(r)
176
+ return out
177
+
178
+ def lease_active(self, r: Routine, now: Optional[float] = None) -> bool:
179
+ """The auto lease holds only while BOTH the deadline and the lease
180
+ budget hold — expiry of either falls back to click-to-run."""
181
+ now = time.time() if now is None else now
182
+ if not (r.auto and r.enabled):
183
+ return False
184
+ if now >= r.lease_deadline:
185
+ return False
186
+ return self.state(r).lease_spent < r.budget_lease
187
+
188
+ def grant_lease(self, r: Routine, days: float, budget: float) -> Routine:
189
+ """Start (or renew) the auto lease: at most MAX_LEASE_DAYS, always
190
+ with a budget. Renewal resets the lease odometer."""
191
+ days = max(0.0, min(float(days), MAX_LEASE_DAYS))
192
+ r.auto = True
193
+ r.lease_deadline = time.time() + days * 86_400.0
194
+ r.budget_lease = float(budget)
195
+ st = self.state(r)
196
+ st.lease_spent = 0.0
197
+ self.save(r)
198
+ self._write_state(r, st)
199
+ return r
200
+
201
+ def revoke_lease(self, r: Routine) -> Routine:
202
+ r.auto = False
203
+ r.lease_deadline = 0.0
204
+ self.save(r)
205
+ return r
206
+
207
+ def record_run(self, r: Routine, *, session_id: str, cost: float, status: str,
208
+ now: Optional[float] = None) -> RoutineState:
209
+ """Odometer tick after a run: last_run moves, lease spend accumulates,
210
+ the run lands in the (bounded) history with its trajectory id — the
211
+ dream finds the full story there."""
212
+ now = time.time() if now is None else now
213
+ st = self.state(r)
214
+ st.last_run = now
215
+ st.lease_spent += max(0.0, float(cost))
216
+ st.runs.append({"sid": session_id, "t": now, "cost": cost, "status": status})
217
+ self._write_state(r, st)
218
+ return st
219
+
220
+
221
+ # ─────────────────────────────────────────────────────────────────────────────
222
+ # the runner — exec's headless machinery, driven by a routine's contract
223
+ # ─────────────────────────────────────────────────────────────────────────────
224
+
225
+
226
+ def _grant_tokens(tools: list[str]) -> frozenset[str]:
227
+ """routine.toml lists what the user granted; exec's HeadlessApprover
228
+ speaks grant tokens — bare tool names become "tool:<name>", anything
229
+ already token-shaped (a bash safety-pattern name, "tool:x") passes as-is."""
230
+ return frozenset(t if ":" in t or "-" in t else f"tool:{t}" for t in tools)
231
+
232
+
233
+ async def run_routine(store: RoutineStore, r: Routine, *, model: str,
234
+ client=None, registry=None, err=None) -> dict:
235
+ """Run one routine through exec (sandboxed ALWAYS — no host fallback for
236
+ unattended work; Docker down = the run fails loudly, the card says so).
237
+ Settles the odometer from the result envelope and writes last-run.md
238
+ beside the routine for the "ready" line. Returns a summary dict."""
239
+ from rockycode.engine.headless import run_exec
240
+ from rockycode.pricing import UsageLedger
241
+ from rockycode.session import get_project
242
+
243
+ workdir = Path(r.workdir).expanduser() if r.workdir else Path.cwd()
244
+ skill = ""
245
+ if r.path is not None:
246
+ try:
247
+ skill = (r.path / "SKILL.md").read_text(encoding="utf-8", errors="replace")
248
+ except OSError:
249
+ skill = ""
250
+ out_note = (f"\nWrite your results into the directory '{r.output_dir}' "
251
+ f"(relative to the working directory)." if r.output_dir else "")
252
+ prompt = f"{r.prompt}{out_note}"
253
+ if skill.strip():
254
+ prompt += f"\n\nFollow this playbook:\n\n{skill.strip()}"
255
+
256
+ project = get_project(workdir)
257
+ lines: list[dict] = []
258
+ code = await run_exec(
259
+ prompt=prompt, model=model, workdir=workdir,
260
+ grants=_grant_tokens(r.tools), max_steps=r.max_steps,
261
+ originator=f"routine:{r.name}",
262
+ sandbox=True, network=r.network,
263
+ write=lines.append, client=client, registry=registry, err=err,
264
+ extra_meta={"runner": "routine", "routine": r.name,
265
+ "project_id": r.project_id or project.id,
266
+ "project_name": project.name},
267
+ )
268
+
269
+ meta = next((l for l in lines if l.get("type") == "meta"), {})
270
+ result = next((l for l in lines if l.get("type") == "result"), {})
271
+ status = {0: "done", 2: "blocked", 3: "budget"}.get(code, "error")
272
+ ledger = UsageLedger()
273
+ if result.get("usage"):
274
+ ledger.add(model, result["usage"])
275
+ cost = ledger.cost("usd")
276
+ store.record_run(r, session_id=meta.get("session", ""), cost=cost, status=status)
277
+
278
+ summary = str(result.get("summary", "") or result.get("error", ""))
279
+ if r.path is not None:
280
+ try:
281
+ (r.path / "last-run.md").write_text(
282
+ f"# {r.name} — last run\n\nstatus: {status} · cost: ${cost:.4f} · "
283
+ f"session: {meta.get('session', '?')}\n\n{summary}\n", encoding="utf-8")
284
+ except OSError:
285
+ pass
286
+ return {"status": status, "cost": cost, "summary": summary,
287
+ "session": meta.get("session", ""), "blocked_on": result.get("blocked_on")}
File without changes
@@ -0,0 +1,273 @@
1
+ """The rockycode harness runner: the agent loop on SWE-bench tasks.
2
+
3
+ Per task: pull the official SWE-bench image → start a container → Rocky
4
+ works inside it (bash/read/write/edit via docker exec) → `git diff` is the
5
+ prediction. Same images the scorer uses (namespace "swebench" on Docker
6
+ Hub), so nothing is built locally and the cache is shared.
7
+
8
+ Tasks run sequentially on purpose: containers run under emulation on
9
+ Apple Silicon, and one task at a time keeps API spend observable.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import asyncio
14
+ import json
15
+ from datetime import datetime
16
+ from pathlib import Path
17
+ from typing import Optional
18
+
19
+ from rich.console import Console
20
+
21
+ from rockycode.banner import amaze, confused, fail, info
22
+ from rockycode.engine.container import DockerSession, build_session_registry, extract_patch
23
+ from rockycode.palette import RED
24
+ from rockycode.engine.events import (
25
+ Compacted,
26
+ EngineError,
27
+ TextDelta,
28
+ ToolFinished,
29
+ ToolStarted,
30
+ TurnFinished,
31
+ )
32
+ from rockycode.engine.loop import Engine
33
+ from rockycode.prompts.rocky import BENCH_TASK, ROCKY_SYSTEM
34
+ from rockycode.runners.data import load_verified
35
+
36
+ PREDICTIONS_DIR = Path("results") / "predictions"
37
+
38
+
39
+ def _image_for(instance: dict) -> str:
40
+ from swebench.harness.test_spec.test_spec import make_test_spec
41
+
42
+ return make_test_spec(instance, namespace="swebench").instance_image_key
43
+
44
+
45
+ async def _ensure_image(image: str, console: Console) -> bool:
46
+ proc = await asyncio.create_subprocess_exec(
47
+ "docker", "image", "inspect", image,
48
+ stdout=asyncio.subprocess.DEVNULL, stderr=asyncio.subprocess.DEVNULL,
49
+ )
50
+ if await proc.wait() == 0:
51
+ return True
52
+ info(console, f"pulling {image} [dim](first time per task, can be ~1GB)[/dim]")
53
+ proc = await asyncio.create_subprocess_exec(
54
+ "docker", "pull", "--platform", "linux/amd64", image,
55
+ stdout=asyncio.subprocess.DEVNULL, stderr=asyncio.subprocess.PIPE,
56
+ )
57
+ _, err = await proc.communicate()
58
+ if proc.returncode != 0:
59
+ fail(console, f"pull failed: {err.decode(errors='replace').strip().splitlines()[-1]}")
60
+ return False
61
+ return True
62
+
63
+
64
+ def _task_prompt(instance: dict) -> str:
65
+ return BENCH_TASK.format(
66
+ repo=instance.get("repo", "unknown"),
67
+ problem_statement=instance.get("problem_statement", ""),
68
+ )
69
+
70
+
71
+ async def _run_instance(
72
+ instance: dict,
73
+ *,
74
+ model: str,
75
+ thinking: bool,
76
+ reasoning_effort: str,
77
+ max_tokens: int,
78
+ context_window: int,
79
+ max_steps: int,
80
+ system_prompt: str,
81
+ prompt_name: str,
82
+ prompt_sha: str,
83
+ console: Console,
84
+ ) -> dict:
85
+ iid = instance["instance_id"]
86
+ image = _image_for(instance)
87
+
88
+ if not await _ensure_image(image, console):
89
+ return {"instance_id": iid, "patch": "", "steps": 0, "usage": {}, "error": "image pull failed"}
90
+
91
+ session = await DockerSession.start(image)
92
+ try:
93
+ registry = build_session_registry(session)
94
+ # Same generated "# Tools this session" section as chat, built from
95
+ # THIS registry — so bench prompts list exactly the container tools
96
+ # (no phantom web/artifact advertisements burning budget steps). No
97
+ # language/env/date appends: bench stays English + byte-reproducible.
98
+ from rockycode.prompts.rocky import tools_section
99
+ engine = Engine(
100
+ model=model,
101
+ thinking=thinking,
102
+ reasoning_effort=reasoning_effort,
103
+ max_tokens=max_tokens,
104
+ context_window=context_window,
105
+ max_steps=max_steps,
106
+ system_prompt=system_prompt + tools_section(registry),
107
+ registry=registry,
108
+ trajectory_meta={
109
+ "runner": "rockycode",
110
+ "instance_id": iid,
111
+ "image": image,
112
+ "prompt_name": prompt_name,
113
+ "prompt_sha": prompt_sha,
114
+ },
115
+ )
116
+
117
+ steps, usage, err = 0, {}, None
118
+ async for ev in engine.run_turn(_task_prompt(instance)):
119
+ if isinstance(ev, ToolStarted):
120
+ arg = (ev.args.get("raw") or "")[:70].replace("\n", " ")
121
+ console.print(f" [dim]⚒ {ev.tool} {arg}[/dim]")
122
+ elif isinstance(ev, ToolFinished) and not ev.ok:
123
+ first = ev.output.strip().splitlines()[0][:90] if ev.output.strip() else ""
124
+ console.print(f" [dim]✗ {ev.tool}: {first}[/dim]")
125
+ elif isinstance(ev, Compacted):
126
+ console.print(
127
+ f" [dim]♻ compacted ({ev.strategy}): "
128
+ f"~{ev.tokens_before:,} → ~{ev.tokens_after:,} tokens[/dim]"
129
+ )
130
+ elif isinstance(ev, TextDelta):
131
+ pass # final summary text; trajectory has it
132
+ elif isinstance(ev, EngineError):
133
+ err = ev.message
134
+ console.print(f" [{RED}]✗ {ev.message}[/]")
135
+ elif isinstance(ev, TurnFinished):
136
+ steps, usage = ev.steps, ev.usage
137
+
138
+ patch = await extract_patch(session)
139
+ engine.trajectory.outcome(
140
+ {
141
+ "instance_id": iid,
142
+ "steps": steps,
143
+ "patch_chars": len(patch),
144
+ "engine_error": err,
145
+ "usage": usage,
146
+ }
147
+ )
148
+ return {"instance_id": iid, "patch": patch, "steps": steps, "usage": usage, "error": err}
149
+ finally:
150
+ await session.stop()
151
+
152
+
153
+ async def _run_all(
154
+ instances: list[dict],
155
+ *,
156
+ model: str,
157
+ thinking: bool,
158
+ reasoning_effort: str,
159
+ max_tokens: int,
160
+ context_window: int,
161
+ max_steps: int,
162
+ token_budget: int,
163
+ system_prompt: str,
164
+ prompt_name: str,
165
+ prompt_sha: str,
166
+ out_path: Path,
167
+ safe_model: str,
168
+ console: Console,
169
+ ) -> None:
170
+ totals: dict[str, int] = {"prompt_tokens": 0, "completion_tokens": 0, "prompt_cache_hit_tokens": 0}
171
+ n_patch = 0
172
+
173
+ with out_path.open("w") as f:
174
+ for i, instance in enumerate(instances, 1):
175
+ spent = totals["prompt_tokens"] + totals["completion_tokens"]
176
+ if token_budget and spent >= token_budget:
177
+ info(console, f"token budget reached ({spent:,} >= {token_budget:,}) — stopping.")
178
+ break
179
+
180
+ iid = instance["instance_id"]
181
+ console.print(f"\n[bold]task {i}/{len(instances)}[/bold] · {iid}")
182
+ try:
183
+ result = await _run_instance(
184
+ instance,
185
+ model=model,
186
+ thinking=thinking,
187
+ reasoning_effort=reasoning_effort,
188
+ max_tokens=max_tokens,
189
+ context_window=context_window,
190
+ max_steps=max_steps,
191
+ system_prompt=system_prompt,
192
+ prompt_name=prompt_name,
193
+ prompt_sha=prompt_sha,
194
+ console=console,
195
+ )
196
+ except Exception as e: # noqa: BLE001 — one task must not kill the run
197
+ fail(console, f"{iid}: {type(e).__name__}: {e}")
198
+ result = {"instance_id": iid, "patch": "", "steps": 0, "usage": {}, "error": str(e)}
199
+
200
+ patch = result["patch"]
201
+ if patch.strip():
202
+ n_patch += 1
203
+ amaze(console, f"patch ready · {result['steps']} steps · {len(patch):,} chars")
204
+ else:
205
+ confused(console, f"no patch produced ({result.get('error') or 'agent stopped without changes'})")
206
+
207
+ for k in totals:
208
+ totals[k] += result["usage"].get(k, 0) or 0
209
+
210
+ f.write(json.dumps({
211
+ "instance_id": iid,
212
+ "model_name_or_path": f"rockycode-{safe_model}",
213
+ "model_patch": patch,
214
+ }) + "\n")
215
+ f.flush()
216
+
217
+ console.print(
218
+ f"\n[dim]· {n_patch}/{len(instances)} tasks produced a patch · "
219
+ f"{totals['prompt_tokens']:,} in / {totals['completion_tokens']:,} out · "
220
+ f"cache hit {totals['prompt_cache_hit_tokens']:,}[/dim]"
221
+ )
222
+
223
+
224
+ def run(
225
+ model: str,
226
+ instance_ids: Optional[list[str]],
227
+ console: Console,
228
+ *,
229
+ thinking: bool = True,
230
+ reasoning_effort: str = "max",
231
+ max_tokens: int = 16384,
232
+ context_window: int = 131_072,
233
+ max_steps: int = 50,
234
+ token_budget: int = 0,
235
+ system_prompt: Optional[str] = None,
236
+ prompt_name: str = "rocky-builtin",
237
+ prompt_sha: str = "",
238
+ task_label: str = "tasks",
239
+ output_dir: Path = PREDICTIONS_DIR,
240
+ ) -> Path:
241
+ """Run the harness on the requested instances. Returns predictions path."""
242
+ if system_prompt is None:
243
+ system_prompt = ROCKY_SYSTEM
244
+ ds = load_verified(console, instance_ids)
245
+ instances = list(ds)
246
+ console.print(f" → {len(instances)} tasks for the rockycode harness\n")
247
+
248
+ output_dir.mkdir(parents=True, exist_ok=True)
249
+ safe_model = model.replace("/", "-")
250
+ # prompt + task-set + timestamp: no run ever overwrites another
251
+ stamp = datetime.now().strftime("%Y%m%d-%H%M")
252
+ out_path = output_dir / f"rockycode-{safe_model}-{prompt_name}-{task_label}-{stamp}.jsonl"
253
+
254
+ asyncio.run(
255
+ _run_all(
256
+ instances,
257
+ model=model,
258
+ thinking=thinking,
259
+ reasoning_effort=reasoning_effort,
260
+ max_tokens=max_tokens,
261
+ context_window=context_window,
262
+ max_steps=max_steps,
263
+ token_budget=token_budget,
264
+ system_prompt=system_prompt,
265
+ prompt_name=prompt_name,
266
+ prompt_sha=prompt_sha,
267
+ out_path=out_path,
268
+ safe_model=safe_model,
269
+ console=console,
270
+ )
271
+ )
272
+ info(console, f"predictions at {out_path}; trajectories at .rockycode/trajectories/")
273
+ return out_path
@@ -0,0 +1,61 @@
1
+ """Shared SWE-bench Verified dataset loading for all runners."""
2
+ from __future__ import annotations
3
+
4
+ from typing import Optional
5
+
6
+ from rich.console import Console
7
+
8
+ from rockycode.banner import confused, fail, info
9
+ from rockycode.palette import BLUE
10
+
11
+ _NETWORK_HINTS = (
12
+ "connection", "timeout", "name resolution", "network", "unreachable",
13
+ "dns", "ssl", "max retries", "getaddrinfo", "proxy", "refused",
14
+ )
15
+
16
+ DATASET = "princeton-nlp/SWE-bench_Verified"
17
+
18
+
19
+ def load_verified(console: Console, instance_ids: Optional[list[str]]):
20
+ """Load Verified (filtered to instance_ids if given). Exits with a
21
+ friendly message on network failure."""
22
+ import datasets
23
+
24
+ # HF's tqdm bars (download + filter) collide with our Rich status line —
25
+ # one renderer must own the terminal.
26
+ datasets.disable_progress_bars()
27
+
28
+ try:
29
+ with console.status(
30
+ f"loading [{BLUE}]{DATASET}[/] "
31
+ "[dim](first run pulls ~hundreds of MB from huggingface)[/dim]",
32
+ spinner="dots",
33
+ ):
34
+ ds = datasets.load_dataset(DATASET, split="test")
35
+ except Exception as e:
36
+ fail(console, "could not load SWE-bench Verified from huggingface.")
37
+ msg = str(e).lower()
38
+ if any(k in msg for k in _NETWORK_HINTS):
39
+ confused(
40
+ console,
41
+ "looks like a network issue. check internet / vpn / proxy / "
42
+ "is huggingface.co reachable?",
43
+ )
44
+ else:
45
+ confused(console, f"underlying error: {type(e).__name__}: {e}")
46
+ info(console, "isolate the load step to debug:")
47
+ info(
48
+ console,
49
+ " uv run python -c \"from datasets import load_dataset; "
50
+ "load_dataset('princeton-nlp/SWE-bench_Verified', split='test')\"",
51
+ )
52
+ raise SystemExit(1)
53
+
54
+ if instance_ids:
55
+ wanted = set(instance_ids)
56
+ ds = ds.filter(lambda r: r["instance_id"] in wanted)
57
+ missing = wanted - set(ds["instance_id"])
58
+ if missing:
59
+ confused(console, f"{len(missing)} requested IDs not in Verified: {sorted(missing)[:3]}…")
60
+
61
+ return ds