rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/engine/goal.py
ADDED
|
@@ -0,0 +1,541 @@
|
|
|
1
|
+
"""Goal mode orchestrator: autonomous, budget-capped, sandbox-isolated runs.
|
|
2
|
+
|
|
3
|
+
Ties the phase-1/2 pieces together: safety (classify each bash command), budget
|
|
4
|
+
(stop on any cap), worktree (run on an isolated COPY of the repo). The loop:
|
|
5
|
+
|
|
6
|
+
plan the objective into milestones
|
|
7
|
+
→ for each: work a turn, then VERIFY (explicit pass/fail via check_code)
|
|
8
|
+
→ every REVIEW_EVERY turns (or after repeated stalls) a milestone REVIEW —
|
|
9
|
+
a (optionally stronger) model judges progress and can rewrite the remaining
|
|
10
|
+
plan to keep the goal moving
|
|
11
|
+
→ finalize gracefully on: plan complete, any budget cap, or a hard stall.
|
|
12
|
+
|
|
13
|
+
The LLM-dependent steps (plan / work / verify / review) live behind the `Driver`
|
|
14
|
+
protocol, so this orchestration is fully testable with a fake driver; the real
|
|
15
|
+
one (EngineDriver) wires them to the agent Engine + models.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from typing import Awaitable, Callable, Optional, Protocol
|
|
22
|
+
|
|
23
|
+
from rockycode.engine.budget import GoalBudget
|
|
24
|
+
from rockycode.engine.safety import Verdict, classify_command, pre_scan
|
|
25
|
+
from rockycode.engine.worktree import GoalWorkspace
|
|
26
|
+
from rockycode.pricing import UsageLedger
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class GoalResult:
|
|
31
|
+
status: str # "done" | "budget" | "stalled" | "aborted"
|
|
32
|
+
reason: str
|
|
33
|
+
milestones_done: int
|
|
34
|
+
milestones_total: int
|
|
35
|
+
diff: str
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Driver(Protocol):
|
|
39
|
+
# Returns (milestones, requires): the plan, plus the planner's optional
|
|
40
|
+
# 'REQUIRES:' declaration text (network/push/sudo/install) for the pre-flight.
|
|
41
|
+
async def plan(self, objective: str) -> tuple[list[str], str]: ...
|
|
42
|
+
async def work(self, milestone: str, context: str) -> str: ...
|
|
43
|
+
# Snapshot the project's check state BEFORE any work — so verify can tell a
|
|
44
|
+
# pre-existing problem (not this milestone's fault) from a real regression.
|
|
45
|
+
async def capture_baseline(self) -> None: ...
|
|
46
|
+
async def verify(self, milestone: str, result: str) -> tuple[bool, str]: ...
|
|
47
|
+
async def review(
|
|
48
|
+
self, objective: str, remaining: list[str], done: list[str], note: str
|
|
49
|
+
) -> Optional[list[str]]: ... # None = plan unchanged; a list = new remaining plan
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class GoalRunner:
|
|
54
|
+
objective: str
|
|
55
|
+
driver: Driver
|
|
56
|
+
budget: GoalBudget
|
|
57
|
+
workspace: GoalWorkspace
|
|
58
|
+
ledger: UsageLedger
|
|
59
|
+
review_every: int = 3
|
|
60
|
+
max_stalls: int = 2
|
|
61
|
+
on_event: Optional[Callable[[str], None]] = None
|
|
62
|
+
# Called with the ask-tier verdicts found at plan time; return True to allow
|
|
63
|
+
# them for the run. Default (None) → deny (fail-safe): a headless run won't
|
|
64
|
+
# silently escalate. Used only on the legacy (non-preplanned) path.
|
|
65
|
+
on_approve: Optional[Callable[[list[Verdict]], Awaitable[bool]]] = None
|
|
66
|
+
# When set, planning + pre-flight already happened upstream (the CLI plans
|
|
67
|
+
# before the sandbox so network/permits are decided from the real plan, then
|
|
68
|
+
# provisions and hands us the plan). We skip straight to execution.
|
|
69
|
+
preplanned: Optional[list[str]] = None
|
|
70
|
+
|
|
71
|
+
async def run(self) -> GoalResult:
|
|
72
|
+
if self.preplanned is not None:
|
|
73
|
+
plan = list(self.preplanned)
|
|
74
|
+
total = len(plan)
|
|
75
|
+
else:
|
|
76
|
+
# Legacy path (tests / headless): plan + pre-flight here. The CLI uses
|
|
77
|
+
# the preplanned path so it can decide network/permits from the plan
|
|
78
|
+
# BEFORE the sandbox exists.
|
|
79
|
+
self._emit(f"planning: {self.objective}")
|
|
80
|
+
plan, _requires = await self.driver.plan(self.objective)
|
|
81
|
+
total = len(plan)
|
|
82
|
+
flags = pre_scan("\n".join(plan))
|
|
83
|
+
blocked = [v for v in flags if v.action == "block"]
|
|
84
|
+
if blocked:
|
|
85
|
+
return GoalResult("aborted", f"plan names a blocked action: {blocked[0].reason}",
|
|
86
|
+
0, total, "")
|
|
87
|
+
asks = [v for v in flags if v.action == "ask"]
|
|
88
|
+
if asks:
|
|
89
|
+
approved = await self.on_approve(asks) if self.on_approve else False
|
|
90
|
+
if not approved:
|
|
91
|
+
names = ", ".join(v.reason for v in asks)
|
|
92
|
+
return GoalResult("aborted", f"needs up-front approval for: {names}", 0, total, "")
|
|
93
|
+
|
|
94
|
+
self.budget.start()
|
|
95
|
+
self._emit(f"budget: {self.budget.preflight_note()}")
|
|
96
|
+
# Snapshot the starting check state — planning doesn't touch files, so the
|
|
97
|
+
# workspace is still pristine here. verify() judges each milestone against
|
|
98
|
+
# this, so a pre-existing lint error a LATER milestone will fix can't fail
|
|
99
|
+
# an earlier one.
|
|
100
|
+
await self.driver.capture_baseline()
|
|
101
|
+
done: list[str] = []
|
|
102
|
+
stalls = 0
|
|
103
|
+
turn = 0
|
|
104
|
+
|
|
105
|
+
while plan:
|
|
106
|
+
over = self.budget.exceeded(self.ledger)
|
|
107
|
+
if over:
|
|
108
|
+
return self._finalize("budget", over, done, plan)
|
|
109
|
+
|
|
110
|
+
milestone = plan[0]
|
|
111
|
+
turn += 1
|
|
112
|
+
self._emit(f"[{turn}] working: {milestone}")
|
|
113
|
+
result = await self.driver.work(milestone, self._context(done))
|
|
114
|
+
ok, why = await self.driver.verify(milestone, result)
|
|
115
|
+
|
|
116
|
+
if ok:
|
|
117
|
+
self._emit(f"[{turn}] verified: {milestone}")
|
|
118
|
+
# Checkpoint the verified work onto the goal branch. Durable +
|
|
119
|
+
# crash-recoverable: a kill mid-run keeps every passed milestone.
|
|
120
|
+
if self.workspace.commit(f"goal: {milestone}"):
|
|
121
|
+
self._emit(f"[{turn}] committed")
|
|
122
|
+
done.append(plan.pop(0))
|
|
123
|
+
stalls = 0
|
|
124
|
+
else:
|
|
125
|
+
stalls += 1
|
|
126
|
+
self._emit(f"[{turn}] verify failed ({stalls}/{self.max_stalls}): {why}")
|
|
127
|
+
if stalls >= self.max_stalls:
|
|
128
|
+
new_plan = await self.driver.review(self.objective, plan, done, why)
|
|
129
|
+
if new_plan is not None:
|
|
130
|
+
plan, total = new_plan, len(done) + len(new_plan)
|
|
131
|
+
stalls = 0
|
|
132
|
+
self._emit("reviewer re-planned after stall")
|
|
133
|
+
else:
|
|
134
|
+
return self._finalize("stalled", f"stuck on '{milestone}': {why}", done, plan)
|
|
135
|
+
|
|
136
|
+
if turn % self.review_every == 0 and plan:
|
|
137
|
+
new_plan = await self.driver.review(self.objective, plan, done, "periodic checkpoint")
|
|
138
|
+
if new_plan is not None:
|
|
139
|
+
plan, total = new_plan, len(done) + len(new_plan)
|
|
140
|
+
self._emit("reviewer adjusted the plan")
|
|
141
|
+
|
|
142
|
+
return self._finalize("done", "all milestones complete", done, plan)
|
|
143
|
+
|
|
144
|
+
def _finalize(self, status: str, reason: str, done: list[str], remaining: list[str]) -> GoalResult:
|
|
145
|
+
self._emit(f"finalizing: {status} — {reason}")
|
|
146
|
+
return GoalResult(status, reason, len(done), len(done) + len(remaining), self.workspace.diff())
|
|
147
|
+
|
|
148
|
+
def _context(self, done: list[str]) -> str:
|
|
149
|
+
return "completed so far: " + "; ".join(done) if done else "(nothing done yet)"
|
|
150
|
+
|
|
151
|
+
def _emit(self, msg: str) -> None:
|
|
152
|
+
if self.on_event:
|
|
153
|
+
self.on_event(msg)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
157
|
+
# EngineDriver — the real Driver: model calls + the agent Engine, safety-gated.
|
|
158
|
+
# The parse_* helpers and safe_bash_tool are pure and unit-tested; the model /
|
|
159
|
+
# Engine wiring needs a live run to validate end-to-end.
|
|
160
|
+
# ─────────────────────────────────────────────────────────────────────────────
|
|
161
|
+
|
|
162
|
+
_PLAN_SYS = (
|
|
163
|
+
"You plan autonomous coding runs. You are given the objective and a snapshot "
|
|
164
|
+
"of the actual project files — plan against what's REALLY there (real file and "
|
|
165
|
+
"symbol names), never invent names. Break the objective into CONCRETE, "
|
|
166
|
+
"individually VERIFIABLE milestones — as FEW as the task genuinely needs "
|
|
167
|
+
"(a one-line change may be a single milestone; use more only for real scope, "
|
|
168
|
+
"up to 8). Don't pad. Do NOT add a separate 'run the linter', 'make it pass', "
|
|
169
|
+
"or 'verify it works/runs' milestone — passing the checks is an acceptance "
|
|
170
|
+
"criterion checked automatically after EVERY milestone, not a step of its own. "
|
|
171
|
+
"Each milestone is a PROSE description of WHAT to accomplish (e.g. 'Create "
|
|
172
|
+
"hello_gui.py with a Tkinter window that shows a Hello World label') — NOT "
|
|
173
|
+
"code. Never output source code, shell commands, here-docs, or file contents "
|
|
174
|
+
"as milestones; the agent writes the code itself. One line = one milestone, so "
|
|
175
|
+
"a single file is ONE milestone, not one per line. "
|
|
176
|
+
"Output the milestones one per line, imperative, no numbering, no preamble. "
|
|
177
|
+
"THEN, only if the plan needs elevated access the sandbox lacks by default, "
|
|
178
|
+
"add ONE final line starting 'REQUIRES:' listing needs and why — from "
|
|
179
|
+
"{network, git push, sudo, package install}. The sandbox is OFFLINE by "
|
|
180
|
+
"default, so anything that installs packages or hits the internet REQUIRES "
|
|
181
|
+
"network. If it needs none, omit the line."
|
|
182
|
+
)
|
|
183
|
+
_VERIFY_SYS = (
|
|
184
|
+
"You verify whether a coding milestone's OBJECTIVE has been achieved. You get "
|
|
185
|
+
"the milestone, the agent's summary, the checks NOW, and the BASELINE checks "
|
|
186
|
+
"(captured before the run). Decide FIRST: the FIRST line must be exactly PASS "
|
|
187
|
+
"or FAIL — no reasoning or hedging before it — then a one-line reason. Judge "
|
|
188
|
+
"the END STATE, not what the agent did:\n"
|
|
189
|
+
"• If the checks NOW are clean (no issues) and the objective is met, PASS. An "
|
|
190
|
+
"error that was in the BASELINE but is GONE now was FIXED — that is SUCCESS, "
|
|
191
|
+
"never an inconsistency or a discrepancy.\n"
|
|
192
|
+
"• 'No issues found' / 'nothing to do' is CORRECT when the state is already "
|
|
193
|
+
"clean (an earlier milestone may have handled it) — never fail an honest "
|
|
194
|
+
"'nothing to do'.\n"
|
|
195
|
+
"• An error present in BOTH baseline and now is pre-existing — it fails THIS "
|
|
196
|
+
"milestone only if fixing it was this milestone's stated objective.\n"
|
|
197
|
+
"• FAIL only if the objective is clearly unmet, or the checks NOW show a NEW "
|
|
198
|
+
"error that is absent from the baseline (a regression this work introduced)."
|
|
199
|
+
)
|
|
200
|
+
_REVIEW_SYS = (
|
|
201
|
+
"You review progress on an autonomous coding goal and keep it on track. Given "
|
|
202
|
+
"the objective, what's done, what remains, and a note, decide: if the plan is "
|
|
203
|
+
"still good, reply with the single word KEEP and nothing else. Otherwise reply "
|
|
204
|
+
"with ONLY the revised remaining milestones, one per line — no heading, no "
|
|
205
|
+
"numbering, no preamble. This replaces the remaining plan."
|
|
206
|
+
)
|
|
207
|
+
_DISCUSS_SYS = (
|
|
208
|
+
"You're refining an autonomous coding plan WITH the user before it runs — talk "
|
|
209
|
+
"to them like a colleague. You get the objective, the current milestone plan, "
|
|
210
|
+
"and the user's message (a QUESTION or a change request). Reply in two parts:\n"
|
|
211
|
+
"1) A SHORT, direct answer to the user (1–3 sentences): answer their question "
|
|
212
|
+
"from the plan/objective, or acknowledge their change.\n"
|
|
213
|
+
"2) Then a line containing exactly '---PLAN---', then the milestone list (one "
|
|
214
|
+
"per line, imperative, no numbering) — REVISED if they asked for a change, "
|
|
215
|
+
"otherwise the SAME plan unchanged. If it needs elevated access "
|
|
216
|
+
"(network / package install / git push / sudo) add a final 'REQUIRES:' line."
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
# A heading / preamble line, not a milestone — e.g. 'REVISED remaining-milestone
|
|
220
|
+
# list', 'Here is the plan', 'Milestones', 'the new plan'. Milestones are
|
|
221
|
+
# imperative ('Add…', 'Run…') and never match this.
|
|
222
|
+
_HEADER_RX = re.compile(
|
|
223
|
+
r"(?i)^\s*("
|
|
224
|
+
r"here\b.*"
|
|
225
|
+
r"|(the|a|an)?\s*(revised|updated|new)?\s*(remaining[-\s]?)?"
|
|
226
|
+
r"(milestone|plan|step)s?[-\s]?(list)?\s*"
|
|
227
|
+
r")$"
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
_FENCE_RX = re.compile(r"^\s*```")
|
|
232
|
+
# A heredoc opener: `<<EOF`, `<<-EOF`, `<< 'EOF'`, `<<"EOF"`. The body up to the
|
|
233
|
+
# delimiter is file CONTENT, not milestones — a planner that dumps
|
|
234
|
+
# `cat <<'EOF' > app.py … EOF` means ONE milestone (write the file), not one per
|
|
235
|
+
# line of the script. (Real bug: a tkinter heredoc became 8 per-line milestones.)
|
|
236
|
+
_HEREDOC_RX = re.compile(r"<<-?\s*(['\"]?)([A-Za-z_]\w*)\1")
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def parse_plan(text: str) -> list[str]:
|
|
240
|
+
"""Pull a milestone list out of a model reply (strips bullets/numbering, skips
|
|
241
|
+
blank lines, trailing-colon headers, and preamble like 'Here is the plan').
|
|
242
|
+
|
|
243
|
+
Code the planner shouldn't have emitted is collapsed, not exploded: a fenced
|
|
244
|
+
```block``` and a here-doc body are kept WITH their owning line as a single
|
|
245
|
+
milestone instead of one milestone per line of code."""
|
|
246
|
+
out: list[str] = []
|
|
247
|
+
lines = text.splitlines()
|
|
248
|
+
i, n = 0, len(lines)
|
|
249
|
+
while i < n and len(out) < 8:
|
|
250
|
+
line = lines[i]
|
|
251
|
+
i += 1
|
|
252
|
+
# A bare code fence: swallow the whole fenced block (it's code, not steps).
|
|
253
|
+
# Attach it to the previous milestone if any; otherwise drop it.
|
|
254
|
+
if _FENCE_RX.match(line):
|
|
255
|
+
block = []
|
|
256
|
+
while i < n and not _FENCE_RX.match(lines[i]):
|
|
257
|
+
block.append(lines[i]); i += 1
|
|
258
|
+
i += 1 # closing fence
|
|
259
|
+
if out and block:
|
|
260
|
+
out[-1] = (out[-1] + "\n" + "\n".join(block)).strip()
|
|
261
|
+
continue
|
|
262
|
+
s = re.sub(r"^\s*(?:[-*•]|\d+[.)])\s*", "", line).strip()
|
|
263
|
+
if not s or s.endswith((":", ":")) or _HEADER_RX.match(s):
|
|
264
|
+
continue
|
|
265
|
+
hd = _HEREDOC_RX.search(s)
|
|
266
|
+
if hd:
|
|
267
|
+
# One milestone spans the whole here-doc, delimiter included.
|
|
268
|
+
delim, body = hd.group(2), [line]
|
|
269
|
+
while i < n and lines[i].strip() != delim:
|
|
270
|
+
body.append(lines[i]); i += 1
|
|
271
|
+
if i < n:
|
|
272
|
+
body.append(lines[i]); i += 1
|
|
273
|
+
out.append("\n".join(body).strip())
|
|
274
|
+
else:
|
|
275
|
+
out.append(s)
|
|
276
|
+
return out[:8]
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
_REQUIRES_RX = re.compile(r"(?i)^\s*(?:[-*•]\s*)?requires\s*[::]\s*(.*)$")
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def split_plan(reply: str) -> tuple[list[str], str]:
|
|
283
|
+
"""Split a plan reply into (milestones, requires-declaration). The planner may
|
|
284
|
+
append one 'REQUIRES: network (why); ...' line — pulled out so it isn't treated
|
|
285
|
+
as a milestone, and returned for the pre-flight approval scan."""
|
|
286
|
+
requires = ""
|
|
287
|
+
kept: list[str] = []
|
|
288
|
+
for line in reply.splitlines():
|
|
289
|
+
m = _REQUIRES_RX.match(line)
|
|
290
|
+
if m:
|
|
291
|
+
requires = m.group(1).strip()
|
|
292
|
+
continue
|
|
293
|
+
kept.append(line)
|
|
294
|
+
return parse_plan("\n".join(kept)), requires
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
_VERDICT_MARK = re.compile(
|
|
298
|
+
r"(?i)\b(?:revised\s+)?(?:judg?ment|verdict|conclusion|decision|final answer)"
|
|
299
|
+
r"\s*[:\-]?\s*\**\s*(pass|fail)\b"
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
_PLAN_STOP = {
|
|
304
|
+
"the", "a", "an", "and", "or", "to", "in", "of", "for", "with", "on", "at",
|
|
305
|
+
"by", "add", "remove", "fix", "update", "make", "create", "run", "then", "it",
|
|
306
|
+
"is", "that", "this", "use", "using", "into", "from", "so", "new", "all",
|
|
307
|
+
"any", "not", "should", "need", "please", "file", "code", "function", "class",
|
|
308
|
+
"test", "tests", "docstring",
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _objective_keywords(text: str) -> list[str]:
|
|
313
|
+
"""Salient words from the objective, for ranking which files to show the
|
|
314
|
+
planner (drops stopwords + short tokens; keeps identifiers like snake_case)."""
|
|
315
|
+
seen: set[str] = set()
|
|
316
|
+
out: list[str] = []
|
|
317
|
+
for w in re.findall(r"[A-Za-z_][A-Za-z0-9_]{2,}", text or ""):
|
|
318
|
+
lw = w.lower()
|
|
319
|
+
if lw in _PLAN_STOP or lw in seen:
|
|
320
|
+
continue
|
|
321
|
+
seen.add(lw)
|
|
322
|
+
out.append(lw)
|
|
323
|
+
return out[:12]
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def parse_verdict(text: str) -> tuple[bool, str]:
|
|
327
|
+
"""Read a verify reply. A clean PASS/FAIL on the first line wins. If the model
|
|
328
|
+
reasoned first and stated its call at the end ('Revised judgment: PASS'), honor
|
|
329
|
+
that explicit marker. Otherwise ambiguity → FAIL (never pass on a guess)."""
|
|
330
|
+
body = text.strip()
|
|
331
|
+
first = (body.splitlines() or [""])[0].strip().upper()
|
|
332
|
+
if first.startswith("PASS") or first == "OK":
|
|
333
|
+
return True, body
|
|
334
|
+
if first.startswith("FAIL"):
|
|
335
|
+
return False, body
|
|
336
|
+
marks = _VERDICT_MARK.findall(body) # 'judgment: PASS' when it reasoned first
|
|
337
|
+
if marks:
|
|
338
|
+
return marks[-1].lower() == "pass", body
|
|
339
|
+
return False, body # conservative: no clean verdict → FAIL
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
_KEEP_RX = re.compile(
|
|
343
|
+
r"(?i)\b(keep|still (good|fine|on ?track|solid)|on track|no changes?|"
|
|
344
|
+
r"unchanged|looks good|plan is (still )?(good|fine|ok|solid|sound))\b"
|
|
345
|
+
)
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def parse_review(text: str) -> Optional[list[str]]:
|
|
349
|
+
"""Read a review reply: KEEP — or a paraphrase like 'the plan is still good'
|
|
350
|
+
— → None (plan unchanged); a real milestone list → the new plan."""
|
|
351
|
+
body = text.strip()
|
|
352
|
+
if not body or re.match(r"(?i)keep\b", body):
|
|
353
|
+
return None
|
|
354
|
+
plan = parse_plan(body)
|
|
355
|
+
# A reply that collapses to ≤1 line AND reads like an affirmation is a KEEP
|
|
356
|
+
# paraphrase, not a one-item revised plan (the live-run '[4] The plan is
|
|
357
|
+
# still good.' leak). A genuine revision is a real list of imperatives.
|
|
358
|
+
if len(plan) <= 1 and _KEEP_RX.search(body):
|
|
359
|
+
return None
|
|
360
|
+
return plan or None
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def safe_bash_tool(sandbox, approved_asks: frozenset):
|
|
364
|
+
"""A sandbox bash tool gated by the safety classifier for goal mode: block
|
|
365
|
+
tier is always refused (the model must find a reversible path — it's on an
|
|
366
|
+
isolated copy anyway); ask tier runs only if pre-approved for this run."""
|
|
367
|
+
from rockycode.engine.sandbox import _bash as _sandbox_bash
|
|
368
|
+
from rockycode.engine.tools import SCHEMAS, Tool
|
|
369
|
+
|
|
370
|
+
async def bash(command: str) -> str:
|
|
371
|
+
v = classify_command(command)
|
|
372
|
+
if v.action == "block":
|
|
373
|
+
return (f"[blocked] {v.reason}. Goal mode refuses this — find a reversible "
|
|
374
|
+
f"alternative (you're working on an isolated copy of the repo).")
|
|
375
|
+
if v.action == "ask" and v.pattern not in approved_asks:
|
|
376
|
+
return f"[blocked] {v.reason} — not pre-approved for this goal run."
|
|
377
|
+
return await _sandbox_bash(sandbox, command)
|
|
378
|
+
|
|
379
|
+
return Tool(name="bash", schema=SCHEMAS["bash"], fn=bash, risk="risky")
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
class EngineDriver:
|
|
383
|
+
"""Real Driver: plan/verify/review are (non-streaming) model calls; work
|
|
384
|
+
drives the agent Engine one turn per milestone. Usage from every call flows
|
|
385
|
+
into the shared ledger so the budget sees real spend."""
|
|
386
|
+
|
|
387
|
+
def __init__(self, *, engine=None, client, model, reviewer_model, workspace,
|
|
388
|
+
ledger: UsageLedger, currency: str = "usd", network: bool = True,
|
|
389
|
+
verifier=None) -> None:
|
|
390
|
+
self.engine = engine # attached after the sandbox is provisioned (see attach)
|
|
391
|
+
self.client = client
|
|
392
|
+
self.model = model
|
|
393
|
+
self.reviewer_model = reviewer_model
|
|
394
|
+
self.workspace = workspace
|
|
395
|
+
self.ledger = ledger
|
|
396
|
+
self.currency = currency
|
|
397
|
+
self.network = network # False → tell the agent the sandbox is offline
|
|
398
|
+
# Optional grounded verify (explore.make_goal_verifier): a read-only
|
|
399
|
+
# child inspects the tree instead of judging from the worker's summary.
|
|
400
|
+
# None (or any failure) → the original summary judge below.
|
|
401
|
+
self.verifier = verifier
|
|
402
|
+
self._baseline = "" # check output before any work (see capture_baseline)
|
|
403
|
+
|
|
404
|
+
def attach(self, engine, *, network: bool = True) -> None:
|
|
405
|
+
"""Wire the sandbox-bound engine after the pre-flight decision. The CLI
|
|
406
|
+
plans before the sandbox exists (so network is decided from the plan),
|
|
407
|
+
then provisions and attaches here."""
|
|
408
|
+
self.engine = engine
|
|
409
|
+
self.network = network
|
|
410
|
+
|
|
411
|
+
async def _call(self, model: str, system: str, user: str, max_tokens: int = 2000) -> str:
|
|
412
|
+
resp = await self.client.chat.completions.create(
|
|
413
|
+
model=model,
|
|
414
|
+
messages=[{"role": "system", "content": system}, {"role": "user", "content": user}],
|
|
415
|
+
max_tokens=max_tokens,
|
|
416
|
+
stream=False,
|
|
417
|
+
extra_body={"thinking": {"type": "disabled"}},
|
|
418
|
+
)
|
|
419
|
+
if resp.usage is not None:
|
|
420
|
+
try:
|
|
421
|
+
self.ledger.add(model, resp.usage.model_dump())
|
|
422
|
+
except AttributeError:
|
|
423
|
+
self.ledger.add(model, dict(resp.usage))
|
|
424
|
+
return resp.choices[0].message.content or ""
|
|
425
|
+
|
|
426
|
+
def _workspace_snapshot(self, objective: str = "", max_files: int = 40,
|
|
427
|
+
max_bytes: int = 6000, scan_cap: int = 800) -> str:
|
|
428
|
+
"""A compact view of the real project for the planner — the file tree plus
|
|
429
|
+
small-file contents — so the plan targets ACTUAL names, not invented ones.
|
|
430
|
+
When the repo has more than max_files, rank by the OBJECTIVE's keywords
|
|
431
|
+
(filename + content) so a big repo still surfaces the RELEVANT code instead
|
|
432
|
+
of the alphabetical first 40. Bounded for cost (scan_cap files; content
|
|
433
|
+
only for small files)."""
|
|
434
|
+
root = self.workspace.path
|
|
435
|
+
skip = {".git", "node_modules", ".venv", "venv", "env", "__pycache__", "dist", "build"}
|
|
436
|
+
files: list = []
|
|
437
|
+
for p in sorted(root.rglob("*")):
|
|
438
|
+
rel = p.relative_to(root)
|
|
439
|
+
if any(part in skip for part in rel.parts):
|
|
440
|
+
continue
|
|
441
|
+
if p.is_file():
|
|
442
|
+
files.append(p)
|
|
443
|
+
if len(files) >= scan_cap:
|
|
444
|
+
break
|
|
445
|
+
kws = _objective_keywords(objective)
|
|
446
|
+
if kws and len(files) > max_files:
|
|
447
|
+
def _score(p) -> int:
|
|
448
|
+
name = p.name.lower()
|
|
449
|
+
sc = 3 * sum(1 for k in kws if k in name) # filename hit weighs most
|
|
450
|
+
try:
|
|
451
|
+
if p.stat().st_size <= 40_000:
|
|
452
|
+
low = p.read_text(errors="ignore").lower()
|
|
453
|
+
sc += sum(1 for k in kws if k in low)
|
|
454
|
+
except OSError:
|
|
455
|
+
pass
|
|
456
|
+
return sc
|
|
457
|
+
files.sort(key=_score, reverse=True)
|
|
458
|
+
files = files[:max_files]
|
|
459
|
+
files.sort() # back to path order for a readable tree
|
|
460
|
+
else:
|
|
461
|
+
files = files[:max_files]
|
|
462
|
+
lines = ["Project files:"] + [f" {p.relative_to(root)}" for p in files]
|
|
463
|
+
budget = max_bytes
|
|
464
|
+
for p in files:
|
|
465
|
+
if budget <= 0:
|
|
466
|
+
break
|
|
467
|
+
try:
|
|
468
|
+
if p.stat().st_size > 4000:
|
|
469
|
+
continue
|
|
470
|
+
text = p.read_text()
|
|
471
|
+
except (OSError, UnicodeDecodeError):
|
|
472
|
+
continue # binary / unreadable — skip
|
|
473
|
+
chunk = text[:budget]
|
|
474
|
+
budget -= len(chunk)
|
|
475
|
+
lines.append(f"\n--- {p.relative_to(root)} ---\n{chunk}")
|
|
476
|
+
return "\n".join(lines)
|
|
477
|
+
|
|
478
|
+
async def plan(self, objective: str) -> tuple[list[str], str]:
|
|
479
|
+
user = f"Objective:\n{objective}\n\n{self._workspace_snapshot(objective)}"
|
|
480
|
+
return split_plan(await self._call(self.model, _PLAN_SYS, user, max_tokens=2500))
|
|
481
|
+
|
|
482
|
+
async def discuss(self, objective: str, plan: list[str], requires: str,
|
|
483
|
+
message: str) -> tuple[str, list[str], str]:
|
|
484
|
+
"""Talk about the plan with the user before it runs — return (reply, plan,
|
|
485
|
+
requires): a short conversational answer plus the plan, revised if they
|
|
486
|
+
asked, unchanged if they only asked a question."""
|
|
487
|
+
plan_txt = "\n".join(f"- {m}" for m in plan)
|
|
488
|
+
req_line = f"\nREQUIRES: {requires}" if requires else ""
|
|
489
|
+
user = (f"Objective:\n{objective}\n\nCurrent plan:\n{plan_txt}{req_line}\n\n"
|
|
490
|
+
f"The user says: {message}")
|
|
491
|
+
out = await self._call(self.model, _DISCUSS_SYS, user, max_tokens=1500)
|
|
492
|
+
chat, sep, plan_part = out.partition("---PLAN---")
|
|
493
|
+
if sep and plan_part.strip():
|
|
494
|
+
new_plan, new_req = split_plan(plan_part)
|
|
495
|
+
if new_plan:
|
|
496
|
+
return chat.strip(), new_plan, new_req
|
|
497
|
+
return (chat or out).strip(), plan, requires # answer only → plan unchanged
|
|
498
|
+
|
|
499
|
+
async def work(self, milestone: str, context: str) -> str:
|
|
500
|
+
from rockycode.engine.events import TextDelta, TurnFinished
|
|
501
|
+
parts: list[str] = []
|
|
502
|
+
offline = "" if self.network else (
|
|
503
|
+
"\nThe sandbox has NO network — do not attempt package installs or "
|
|
504
|
+
"downloads (pip/apt/npm/curl); use only what's already available.")
|
|
505
|
+
prompt = (f"[goal] Work on this milestone: {milestone}\n{context}{offline}\n"
|
|
506
|
+
f"Make the changes, verify locally, then reply with a one-line summary.")
|
|
507
|
+
async for ev in self.engine.run_turn(prompt):
|
|
508
|
+
if isinstance(ev, TextDelta):
|
|
509
|
+
parts.append(ev.text)
|
|
510
|
+
elif isinstance(ev, TurnFinished) and ev.usage:
|
|
511
|
+
self.ledger.add(self.engine.model, ev.usage)
|
|
512
|
+
return "".join(parts).strip()
|
|
513
|
+
|
|
514
|
+
async def capture_baseline(self) -> None:
|
|
515
|
+
"""Snapshot the check output before any milestone runs, so verify can
|
|
516
|
+
distinguish a pre-existing problem from a regression this run introduced."""
|
|
517
|
+
from rockycode.engine.checks import run_checks
|
|
518
|
+
self._baseline = await run_checks(self.workspace.path) or ""
|
|
519
|
+
|
|
520
|
+
async def verify(self, milestone: str, result: str) -> tuple[bool, str]:
|
|
521
|
+
from rockycode.engine.checks import run_checks
|
|
522
|
+
checks = await run_checks(self.workspace.path) or "(no linter/type-checker configured)"
|
|
523
|
+
if self.verifier is not None:
|
|
524
|
+
try:
|
|
525
|
+
return await self.verifier(
|
|
526
|
+
milestone=milestone, summary=result,
|
|
527
|
+
baseline=self._baseline, checks=checks,
|
|
528
|
+
)
|
|
529
|
+
except Exception: # noqa: BLE001 — grounded verify degrades to the summary judge
|
|
530
|
+
pass
|
|
531
|
+
user = (
|
|
532
|
+
f"Milestone: {milestone}\n\nAgent summary:\n{result}\n\n"
|
|
533
|
+
f"Baseline checks (before the run began):\n{self._baseline or '(clean — no issues)'}\n\n"
|
|
534
|
+
f"Checks now:\n{checks}"
|
|
535
|
+
)
|
|
536
|
+
return parse_verdict(await self._call(self.model, _VERIFY_SYS, user))
|
|
537
|
+
|
|
538
|
+
async def review(self, objective: str, remaining: list[str], done: list[str], note: str) -> Optional[list[str]]:
|
|
539
|
+
user = (f"Objective:\n{objective}\n\nDone:\n" + "\n".join(f"- {d}" for d in done) +
|
|
540
|
+
"\n\nRemaining:\n" + "\n".join(f"- {r}" for r in remaining) + f"\n\nNote: {note}")
|
|
541
|
+
return parse_review(await self._call(self.reviewer_model, _REVIEW_SYS, user))
|