rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,541 @@
1
+ """Goal mode orchestrator: autonomous, budget-capped, sandbox-isolated runs.
2
+
3
+ Ties the phase-1/2 pieces together: safety (classify each bash command), budget
4
+ (stop on any cap), worktree (run on an isolated COPY of the repo). The loop:
5
+
6
+ plan the objective into milestones
7
+ → for each: work a turn, then VERIFY (explicit pass/fail via check_code)
8
+ → every REVIEW_EVERY turns (or after repeated stalls) a milestone REVIEW —
9
+ a (optionally stronger) model judges progress and can rewrite the remaining
10
+ plan to keep the goal moving
11
+ → finalize gracefully on: plan complete, any budget cap, or a hard stall.
12
+
13
+ The LLM-dependent steps (plan / work / verify / review) live behind the `Driver`
14
+ protocol, so this orchestration is fully testable with a fake driver; the real
15
+ one (EngineDriver) wires them to the agent Engine + models.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import re
20
+ from dataclasses import dataclass
21
+ from typing import Awaitable, Callable, Optional, Protocol
22
+
23
+ from rockycode.engine.budget import GoalBudget
24
+ from rockycode.engine.safety import Verdict, classify_command, pre_scan
25
+ from rockycode.engine.worktree import GoalWorkspace
26
+ from rockycode.pricing import UsageLedger
27
+
28
+
29
+ @dataclass
30
+ class GoalResult:
31
+ status: str # "done" | "budget" | "stalled" | "aborted"
32
+ reason: str
33
+ milestones_done: int
34
+ milestones_total: int
35
+ diff: str
36
+
37
+
38
+ class Driver(Protocol):
39
+ # Returns (milestones, requires): the plan, plus the planner's optional
40
+ # 'REQUIRES:' declaration text (network/push/sudo/install) for the pre-flight.
41
+ async def plan(self, objective: str) -> tuple[list[str], str]: ...
42
+ async def work(self, milestone: str, context: str) -> str: ...
43
+ # Snapshot the project's check state BEFORE any work — so verify can tell a
44
+ # pre-existing problem (not this milestone's fault) from a real regression.
45
+ async def capture_baseline(self) -> None: ...
46
+ async def verify(self, milestone: str, result: str) -> tuple[bool, str]: ...
47
+ async def review(
48
+ self, objective: str, remaining: list[str], done: list[str], note: str
49
+ ) -> Optional[list[str]]: ... # None = plan unchanged; a list = new remaining plan
50
+
51
+
52
+ @dataclass
53
+ class GoalRunner:
54
+ objective: str
55
+ driver: Driver
56
+ budget: GoalBudget
57
+ workspace: GoalWorkspace
58
+ ledger: UsageLedger
59
+ review_every: int = 3
60
+ max_stalls: int = 2
61
+ on_event: Optional[Callable[[str], None]] = None
62
+ # Called with the ask-tier verdicts found at plan time; return True to allow
63
+ # them for the run. Default (None) → deny (fail-safe): a headless run won't
64
+ # silently escalate. Used only on the legacy (non-preplanned) path.
65
+ on_approve: Optional[Callable[[list[Verdict]], Awaitable[bool]]] = None
66
+ # When set, planning + pre-flight already happened upstream (the CLI plans
67
+ # before the sandbox so network/permits are decided from the real plan, then
68
+ # provisions and hands us the plan). We skip straight to execution.
69
+ preplanned: Optional[list[str]] = None
70
+
71
+ async def run(self) -> GoalResult:
72
+ if self.preplanned is not None:
73
+ plan = list(self.preplanned)
74
+ total = len(plan)
75
+ else:
76
+ # Legacy path (tests / headless): plan + pre-flight here. The CLI uses
77
+ # the preplanned path so it can decide network/permits from the plan
78
+ # BEFORE the sandbox exists.
79
+ self._emit(f"planning: {self.objective}")
80
+ plan, _requires = await self.driver.plan(self.objective)
81
+ total = len(plan)
82
+ flags = pre_scan("\n".join(plan))
83
+ blocked = [v for v in flags if v.action == "block"]
84
+ if blocked:
85
+ return GoalResult("aborted", f"plan names a blocked action: {blocked[0].reason}",
86
+ 0, total, "")
87
+ asks = [v for v in flags if v.action == "ask"]
88
+ if asks:
89
+ approved = await self.on_approve(asks) if self.on_approve else False
90
+ if not approved:
91
+ names = ", ".join(v.reason for v in asks)
92
+ return GoalResult("aborted", f"needs up-front approval for: {names}", 0, total, "")
93
+
94
+ self.budget.start()
95
+ self._emit(f"budget: {self.budget.preflight_note()}")
96
+ # Snapshot the starting check state — planning doesn't touch files, so the
97
+ # workspace is still pristine here. verify() judges each milestone against
98
+ # this, so a pre-existing lint error a LATER milestone will fix can't fail
99
+ # an earlier one.
100
+ await self.driver.capture_baseline()
101
+ done: list[str] = []
102
+ stalls = 0
103
+ turn = 0
104
+
105
+ while plan:
106
+ over = self.budget.exceeded(self.ledger)
107
+ if over:
108
+ return self._finalize("budget", over, done, plan)
109
+
110
+ milestone = plan[0]
111
+ turn += 1
112
+ self._emit(f"[{turn}] working: {milestone}")
113
+ result = await self.driver.work(milestone, self._context(done))
114
+ ok, why = await self.driver.verify(milestone, result)
115
+
116
+ if ok:
117
+ self._emit(f"[{turn}] verified: {milestone}")
118
+ # Checkpoint the verified work onto the goal branch. Durable +
119
+ # crash-recoverable: a kill mid-run keeps every passed milestone.
120
+ if self.workspace.commit(f"goal: {milestone}"):
121
+ self._emit(f"[{turn}] committed")
122
+ done.append(plan.pop(0))
123
+ stalls = 0
124
+ else:
125
+ stalls += 1
126
+ self._emit(f"[{turn}] verify failed ({stalls}/{self.max_stalls}): {why}")
127
+ if stalls >= self.max_stalls:
128
+ new_plan = await self.driver.review(self.objective, plan, done, why)
129
+ if new_plan is not None:
130
+ plan, total = new_plan, len(done) + len(new_plan)
131
+ stalls = 0
132
+ self._emit("reviewer re-planned after stall")
133
+ else:
134
+ return self._finalize("stalled", f"stuck on '{milestone}': {why}", done, plan)
135
+
136
+ if turn % self.review_every == 0 and plan:
137
+ new_plan = await self.driver.review(self.objective, plan, done, "periodic checkpoint")
138
+ if new_plan is not None:
139
+ plan, total = new_plan, len(done) + len(new_plan)
140
+ self._emit("reviewer adjusted the plan")
141
+
142
+ return self._finalize("done", "all milestones complete", done, plan)
143
+
144
+ def _finalize(self, status: str, reason: str, done: list[str], remaining: list[str]) -> GoalResult:
145
+ self._emit(f"finalizing: {status} — {reason}")
146
+ return GoalResult(status, reason, len(done), len(done) + len(remaining), self.workspace.diff())
147
+
148
+ def _context(self, done: list[str]) -> str:
149
+ return "completed so far: " + "; ".join(done) if done else "(nothing done yet)"
150
+
151
+ def _emit(self, msg: str) -> None:
152
+ if self.on_event:
153
+ self.on_event(msg)
154
+
155
+
156
+ # ─────────────────────────────────────────────────────────────────────────────
157
+ # EngineDriver — the real Driver: model calls + the agent Engine, safety-gated.
158
+ # The parse_* helpers and safe_bash_tool are pure and unit-tested; the model /
159
+ # Engine wiring needs a live run to validate end-to-end.
160
+ # ─────────────────────────────────────────────────────────────────────────────
161
+
162
+ _PLAN_SYS = (
163
+ "You plan autonomous coding runs. You are given the objective and a snapshot "
164
+ "of the actual project files — plan against what's REALLY there (real file and "
165
+ "symbol names), never invent names. Break the objective into CONCRETE, "
166
+ "individually VERIFIABLE milestones — as FEW as the task genuinely needs "
167
+ "(a one-line change may be a single milestone; use more only for real scope, "
168
+ "up to 8). Don't pad. Do NOT add a separate 'run the linter', 'make it pass', "
169
+ "or 'verify it works/runs' milestone — passing the checks is an acceptance "
170
+ "criterion checked automatically after EVERY milestone, not a step of its own. "
171
+ "Each milestone is a PROSE description of WHAT to accomplish (e.g. 'Create "
172
+ "hello_gui.py with a Tkinter window that shows a Hello World label') — NOT "
173
+ "code. Never output source code, shell commands, here-docs, or file contents "
174
+ "as milestones; the agent writes the code itself. One line = one milestone, so "
175
+ "a single file is ONE milestone, not one per line. "
176
+ "Output the milestones one per line, imperative, no numbering, no preamble. "
177
+ "THEN, only if the plan needs elevated access the sandbox lacks by default, "
178
+ "add ONE final line starting 'REQUIRES:' listing needs and why — from "
179
+ "{network, git push, sudo, package install}. The sandbox is OFFLINE by "
180
+ "default, so anything that installs packages or hits the internet REQUIRES "
181
+ "network. If it needs none, omit the line."
182
+ )
183
+ _VERIFY_SYS = (
184
+ "You verify whether a coding milestone's OBJECTIVE has been achieved. You get "
185
+ "the milestone, the agent's summary, the checks NOW, and the BASELINE checks "
186
+ "(captured before the run). Decide FIRST: the FIRST line must be exactly PASS "
187
+ "or FAIL — no reasoning or hedging before it — then a one-line reason. Judge "
188
+ "the END STATE, not what the agent did:\n"
189
+ "• If the checks NOW are clean (no issues) and the objective is met, PASS. An "
190
+ "error that was in the BASELINE but is GONE now was FIXED — that is SUCCESS, "
191
+ "never an inconsistency or a discrepancy.\n"
192
+ "• 'No issues found' / 'nothing to do' is CORRECT when the state is already "
193
+ "clean (an earlier milestone may have handled it) — never fail an honest "
194
+ "'nothing to do'.\n"
195
+ "• An error present in BOTH baseline and now is pre-existing — it fails THIS "
196
+ "milestone only if fixing it was this milestone's stated objective.\n"
197
+ "• FAIL only if the objective is clearly unmet, or the checks NOW show a NEW "
198
+ "error that is absent from the baseline (a regression this work introduced)."
199
+ )
200
+ _REVIEW_SYS = (
201
+ "You review progress on an autonomous coding goal and keep it on track. Given "
202
+ "the objective, what's done, what remains, and a note, decide: if the plan is "
203
+ "still good, reply with the single word KEEP and nothing else. Otherwise reply "
204
+ "with ONLY the revised remaining milestones, one per line — no heading, no "
205
+ "numbering, no preamble. This replaces the remaining plan."
206
+ )
207
+ _DISCUSS_SYS = (
208
+ "You're refining an autonomous coding plan WITH the user before it runs — talk "
209
+ "to them like a colleague. You get the objective, the current milestone plan, "
210
+ "and the user's message (a QUESTION or a change request). Reply in two parts:\n"
211
+ "1) A SHORT, direct answer to the user (1–3 sentences): answer their question "
212
+ "from the plan/objective, or acknowledge their change.\n"
213
+ "2) Then a line containing exactly '---PLAN---', then the milestone list (one "
214
+ "per line, imperative, no numbering) — REVISED if they asked for a change, "
215
+ "otherwise the SAME plan unchanged. If it needs elevated access "
216
+ "(network / package install / git push / sudo) add a final 'REQUIRES:' line."
217
+ )
218
+
219
+ # A heading / preamble line, not a milestone — e.g. 'REVISED remaining-milestone
220
+ # list', 'Here is the plan', 'Milestones', 'the new plan'. Milestones are
221
+ # imperative ('Add…', 'Run…') and never match this.
222
+ _HEADER_RX = re.compile(
223
+ r"(?i)^\s*("
224
+ r"here\b.*"
225
+ r"|(the|a|an)?\s*(revised|updated|new)?\s*(remaining[-\s]?)?"
226
+ r"(milestone|plan|step)s?[-\s]?(list)?\s*"
227
+ r")$"
228
+ )
229
+
230
+
231
+ _FENCE_RX = re.compile(r"^\s*```")
232
+ # A heredoc opener: `<<EOF`, `<<-EOF`, `<< 'EOF'`, `<<"EOF"`. The body up to the
233
+ # delimiter is file CONTENT, not milestones — a planner that dumps
234
+ # `cat <<'EOF' > app.py … EOF` means ONE milestone (write the file), not one per
235
+ # line of the script. (Real bug: a tkinter heredoc became 8 per-line milestones.)
236
+ _HEREDOC_RX = re.compile(r"<<-?\s*(['\"]?)([A-Za-z_]\w*)\1")
237
+
238
+
239
+ def parse_plan(text: str) -> list[str]:
240
+ """Pull a milestone list out of a model reply (strips bullets/numbering, skips
241
+ blank lines, trailing-colon headers, and preamble like 'Here is the plan').
242
+
243
+ Code the planner shouldn't have emitted is collapsed, not exploded: a fenced
244
+ ```block``` and a here-doc body are kept WITH their owning line as a single
245
+ milestone instead of one milestone per line of code."""
246
+ out: list[str] = []
247
+ lines = text.splitlines()
248
+ i, n = 0, len(lines)
249
+ while i < n and len(out) < 8:
250
+ line = lines[i]
251
+ i += 1
252
+ # A bare code fence: swallow the whole fenced block (it's code, not steps).
253
+ # Attach it to the previous milestone if any; otherwise drop it.
254
+ if _FENCE_RX.match(line):
255
+ block = []
256
+ while i < n and not _FENCE_RX.match(lines[i]):
257
+ block.append(lines[i]); i += 1
258
+ i += 1 # closing fence
259
+ if out and block:
260
+ out[-1] = (out[-1] + "\n" + "\n".join(block)).strip()
261
+ continue
262
+ s = re.sub(r"^\s*(?:[-*•]|\d+[.)])\s*", "", line).strip()
263
+ if not s or s.endswith((":", ":")) or _HEADER_RX.match(s):
264
+ continue
265
+ hd = _HEREDOC_RX.search(s)
266
+ if hd:
267
+ # One milestone spans the whole here-doc, delimiter included.
268
+ delim, body = hd.group(2), [line]
269
+ while i < n and lines[i].strip() != delim:
270
+ body.append(lines[i]); i += 1
271
+ if i < n:
272
+ body.append(lines[i]); i += 1
273
+ out.append("\n".join(body).strip())
274
+ else:
275
+ out.append(s)
276
+ return out[:8]
277
+
278
+
279
+ _REQUIRES_RX = re.compile(r"(?i)^\s*(?:[-*•]\s*)?requires\s*[::]\s*(.*)$")
280
+
281
+
282
+ def split_plan(reply: str) -> tuple[list[str], str]:
283
+ """Split a plan reply into (milestones, requires-declaration). The planner may
284
+ append one 'REQUIRES: network (why); ...' line — pulled out so it isn't treated
285
+ as a milestone, and returned for the pre-flight approval scan."""
286
+ requires = ""
287
+ kept: list[str] = []
288
+ for line in reply.splitlines():
289
+ m = _REQUIRES_RX.match(line)
290
+ if m:
291
+ requires = m.group(1).strip()
292
+ continue
293
+ kept.append(line)
294
+ return parse_plan("\n".join(kept)), requires
295
+
296
+
297
+ _VERDICT_MARK = re.compile(
298
+ r"(?i)\b(?:revised\s+)?(?:judg?ment|verdict|conclusion|decision|final answer)"
299
+ r"\s*[:\-]?\s*\**\s*(pass|fail)\b"
300
+ )
301
+
302
+
303
+ _PLAN_STOP = {
304
+ "the", "a", "an", "and", "or", "to", "in", "of", "for", "with", "on", "at",
305
+ "by", "add", "remove", "fix", "update", "make", "create", "run", "then", "it",
306
+ "is", "that", "this", "use", "using", "into", "from", "so", "new", "all",
307
+ "any", "not", "should", "need", "please", "file", "code", "function", "class",
308
+ "test", "tests", "docstring",
309
+ }
310
+
311
+
312
+ def _objective_keywords(text: str) -> list[str]:
313
+ """Salient words from the objective, for ranking which files to show the
314
+ planner (drops stopwords + short tokens; keeps identifiers like snake_case)."""
315
+ seen: set[str] = set()
316
+ out: list[str] = []
317
+ for w in re.findall(r"[A-Za-z_][A-Za-z0-9_]{2,}", text or ""):
318
+ lw = w.lower()
319
+ if lw in _PLAN_STOP or lw in seen:
320
+ continue
321
+ seen.add(lw)
322
+ out.append(lw)
323
+ return out[:12]
324
+
325
+
326
+ def parse_verdict(text: str) -> tuple[bool, str]:
327
+ """Read a verify reply. A clean PASS/FAIL on the first line wins. If the model
328
+ reasoned first and stated its call at the end ('Revised judgment: PASS'), honor
329
+ that explicit marker. Otherwise ambiguity → FAIL (never pass on a guess)."""
330
+ body = text.strip()
331
+ first = (body.splitlines() or [""])[0].strip().upper()
332
+ if first.startswith("PASS") or first == "OK":
333
+ return True, body
334
+ if first.startswith("FAIL"):
335
+ return False, body
336
+ marks = _VERDICT_MARK.findall(body) # 'judgment: PASS' when it reasoned first
337
+ if marks:
338
+ return marks[-1].lower() == "pass", body
339
+ return False, body # conservative: no clean verdict → FAIL
340
+
341
+
342
+ _KEEP_RX = re.compile(
343
+ r"(?i)\b(keep|still (good|fine|on ?track|solid)|on track|no changes?|"
344
+ r"unchanged|looks good|plan is (still )?(good|fine|ok|solid|sound))\b"
345
+ )
346
+
347
+
348
+ def parse_review(text: str) -> Optional[list[str]]:
349
+ """Read a review reply: KEEP — or a paraphrase like 'the plan is still good'
350
+ — → None (plan unchanged); a real milestone list → the new plan."""
351
+ body = text.strip()
352
+ if not body or re.match(r"(?i)keep\b", body):
353
+ return None
354
+ plan = parse_plan(body)
355
+ # A reply that collapses to ≤1 line AND reads like an affirmation is a KEEP
356
+ # paraphrase, not a one-item revised plan (the live-run '[4] The plan is
357
+ # still good.' leak). A genuine revision is a real list of imperatives.
358
+ if len(plan) <= 1 and _KEEP_RX.search(body):
359
+ return None
360
+ return plan or None
361
+
362
+
363
+ def safe_bash_tool(sandbox, approved_asks: frozenset):
364
+ """A sandbox bash tool gated by the safety classifier for goal mode: block
365
+ tier is always refused (the model must find a reversible path — it's on an
366
+ isolated copy anyway); ask tier runs only if pre-approved for this run."""
367
+ from rockycode.engine.sandbox import _bash as _sandbox_bash
368
+ from rockycode.engine.tools import SCHEMAS, Tool
369
+
370
+ async def bash(command: str) -> str:
371
+ v = classify_command(command)
372
+ if v.action == "block":
373
+ return (f"[blocked] {v.reason}. Goal mode refuses this — find a reversible "
374
+ f"alternative (you're working on an isolated copy of the repo).")
375
+ if v.action == "ask" and v.pattern not in approved_asks:
376
+ return f"[blocked] {v.reason} — not pre-approved for this goal run."
377
+ return await _sandbox_bash(sandbox, command)
378
+
379
+ return Tool(name="bash", schema=SCHEMAS["bash"], fn=bash, risk="risky")
380
+
381
+
382
+ class EngineDriver:
383
+ """Real Driver: plan/verify/review are (non-streaming) model calls; work
384
+ drives the agent Engine one turn per milestone. Usage from every call flows
385
+ into the shared ledger so the budget sees real spend."""
386
+
387
+ def __init__(self, *, engine=None, client, model, reviewer_model, workspace,
388
+ ledger: UsageLedger, currency: str = "usd", network: bool = True,
389
+ verifier=None) -> None:
390
+ self.engine = engine # attached after the sandbox is provisioned (see attach)
391
+ self.client = client
392
+ self.model = model
393
+ self.reviewer_model = reviewer_model
394
+ self.workspace = workspace
395
+ self.ledger = ledger
396
+ self.currency = currency
397
+ self.network = network # False → tell the agent the sandbox is offline
398
+ # Optional grounded verify (explore.make_goal_verifier): a read-only
399
+ # child inspects the tree instead of judging from the worker's summary.
400
+ # None (or any failure) → the original summary judge below.
401
+ self.verifier = verifier
402
+ self._baseline = "" # check output before any work (see capture_baseline)
403
+
404
+ def attach(self, engine, *, network: bool = True) -> None:
405
+ """Wire the sandbox-bound engine after the pre-flight decision. The CLI
406
+ plans before the sandbox exists (so network is decided from the plan),
407
+ then provisions and attaches here."""
408
+ self.engine = engine
409
+ self.network = network
410
+
411
+ async def _call(self, model: str, system: str, user: str, max_tokens: int = 2000) -> str:
412
+ resp = await self.client.chat.completions.create(
413
+ model=model,
414
+ messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
415
+ max_tokens=max_tokens,
416
+ stream=False,
417
+ extra_body={"thinking": {"type": "disabled"}},
418
+ )
419
+ if resp.usage is not None:
420
+ try:
421
+ self.ledger.add(model, resp.usage.model_dump())
422
+ except AttributeError:
423
+ self.ledger.add(model, dict(resp.usage))
424
+ return resp.choices[0].message.content or ""
425
+
426
+ def _workspace_snapshot(self, objective: str = "", max_files: int = 40,
427
+ max_bytes: int = 6000, scan_cap: int = 800) -> str:
428
+ """A compact view of the real project for the planner — the file tree plus
429
+ small-file contents — so the plan targets ACTUAL names, not invented ones.
430
+ When the repo has more than max_files, rank by the OBJECTIVE's keywords
431
+ (filename + content) so a big repo still surfaces the RELEVANT code instead
432
+ of the alphabetical first 40. Bounded for cost (scan_cap files; content
433
+ only for small files)."""
434
+ root = self.workspace.path
435
+ skip = {".git", "node_modules", ".venv", "venv", "env", "__pycache__", "dist", "build"}
436
+ files: list = []
437
+ for p in sorted(root.rglob("*")):
438
+ rel = p.relative_to(root)
439
+ if any(part in skip for part in rel.parts):
440
+ continue
441
+ if p.is_file():
442
+ files.append(p)
443
+ if len(files) >= scan_cap:
444
+ break
445
+ kws = _objective_keywords(objective)
446
+ if kws and len(files) > max_files:
447
+ def _score(p) -> int:
448
+ name = p.name.lower()
449
+ sc = 3 * sum(1 for k in kws if k in name) # filename hit weighs most
450
+ try:
451
+ if p.stat().st_size <= 40_000:
452
+ low = p.read_text(errors="ignore").lower()
453
+ sc += sum(1 for k in kws if k in low)
454
+ except OSError:
455
+ pass
456
+ return sc
457
+ files.sort(key=_score, reverse=True)
458
+ files = files[:max_files]
459
+ files.sort() # back to path order for a readable tree
460
+ else:
461
+ files = files[:max_files]
462
+ lines = ["Project files:"] + [f" {p.relative_to(root)}" for p in files]
463
+ budget = max_bytes
464
+ for p in files:
465
+ if budget <= 0:
466
+ break
467
+ try:
468
+ if p.stat().st_size > 4000:
469
+ continue
470
+ text = p.read_text()
471
+ except (OSError, UnicodeDecodeError):
472
+ continue # binary / unreadable — skip
473
+ chunk = text[:budget]
474
+ budget -= len(chunk)
475
+ lines.append(f"\n--- {p.relative_to(root)} ---\n{chunk}")
476
+ return "\n".join(lines)
477
+
478
+ async def plan(self, objective: str) -> tuple[list[str], str]:
479
+ user = f"Objective:\n{objective}\n\n{self._workspace_snapshot(objective)}"
480
+ return split_plan(await self._call(self.model, _PLAN_SYS, user, max_tokens=2500))
481
+
482
+ async def discuss(self, objective: str, plan: list[str], requires: str,
483
+ message: str) -> tuple[str, list[str], str]:
484
+ """Talk about the plan with the user before it runs — return (reply, plan,
485
+ requires): a short conversational answer plus the plan, revised if they
486
+ asked, unchanged if they only asked a question."""
487
+ plan_txt = "\n".join(f"- {m}" for m in plan)
488
+ req_line = f"\nREQUIRES: {requires}" if requires else ""
489
+ user = (f"Objective:\n{objective}\n\nCurrent plan:\n{plan_txt}{req_line}\n\n"
490
+ f"The user says: {message}")
491
+ out = await self._call(self.model, _DISCUSS_SYS, user, max_tokens=1500)
492
+ chat, sep, plan_part = out.partition("---PLAN---")
493
+ if sep and plan_part.strip():
494
+ new_plan, new_req = split_plan(plan_part)
495
+ if new_plan:
496
+ return chat.strip(), new_plan, new_req
497
+ return (chat or out).strip(), plan, requires # answer only → plan unchanged
498
+
499
+ async def work(self, milestone: str, context: str) -> str:
500
+ from rockycode.engine.events import TextDelta, TurnFinished
501
+ parts: list[str] = []
502
+ offline = "" if self.network else (
503
+ "\nThe sandbox has NO network — do not attempt package installs or "
504
+ "downloads (pip/apt/npm/curl); use only what's already available.")
505
+ prompt = (f"[goal] Work on this milestone: {milestone}\n{context}{offline}\n"
506
+ f"Make the changes, verify locally, then reply with a one-line summary.")
507
+ async for ev in self.engine.run_turn(prompt):
508
+ if isinstance(ev, TextDelta):
509
+ parts.append(ev.text)
510
+ elif isinstance(ev, TurnFinished) and ev.usage:
511
+ self.ledger.add(self.engine.model, ev.usage)
512
+ return "".join(parts).strip()
513
+
514
+ async def capture_baseline(self) -> None:
515
+ """Snapshot the check output before any milestone runs, so verify can
516
+ distinguish a pre-existing problem from a regression this run introduced."""
517
+ from rockycode.engine.checks import run_checks
518
+ self._baseline = await run_checks(self.workspace.path) or ""
519
+
520
+ async def verify(self, milestone: str, result: str) -> tuple[bool, str]:
521
+ from rockycode.engine.checks import run_checks
522
+ checks = await run_checks(self.workspace.path) or "(no linter/type-checker configured)"
523
+ if self.verifier is not None:
524
+ try:
525
+ return await self.verifier(
526
+ milestone=milestone, summary=result,
527
+ baseline=self._baseline, checks=checks,
528
+ )
529
+ except Exception: # noqa: BLE001 — grounded verify degrades to the summary judge
530
+ pass
531
+ user = (
532
+ f"Milestone: {milestone}\n\nAgent summary:\n{result}\n\n"
533
+ f"Baseline checks (before the run began):\n{self._baseline or '(clean — no issues)'}\n\n"
534
+ f"Checks now:\n{checks}"
535
+ )
536
+ return parse_verdict(await self._call(self.model, _VERIFY_SYS, user))
537
+
538
+ async def review(self, objective: str, remaining: list[str], done: list[str], note: str) -> Optional[list[str]]:
539
+ user = (f"Objective:\n{objective}\n\nDone:\n" + "\n".join(f"- {d}" for d in done) +
540
+ "\n\nRemaining:\n" + "\n".join(f"- {r}" for r in remaining) + f"\n\nNote: {note}")
541
+ return parse_review(await self._call(self.reviewer_model, _REVIEW_SYS, user))