rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/cli.py
ADDED
|
@@ -0,0 +1,1386 @@
|
|
|
1
|
+
"""rockycode CLI: `rockycode bench …`"""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import subprocess
|
|
7
|
+
import sys
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import List, Optional
|
|
11
|
+
|
|
12
|
+
import typer
|
|
13
|
+
from rich.console import Console
|
|
14
|
+
|
|
15
|
+
from rockycode.banner import confused, fail, info, show_banner
|
|
16
|
+
|
|
17
|
+
# Credentials and endpoint come ONLY from ~/.rockycode (.env / keychain) or an
|
|
18
|
+
# explicit shell export. Project .env files are deliberately never loaded — a
|
|
19
|
+
# repo must not be able to supply a key or redirect where the key is sent
|
|
20
|
+
# (project_env_warnings tells the user when one tries).
|
|
21
|
+
from rockycode.onboarding import ( # noqa: E402
|
|
22
|
+
bootstrap_credentials,
|
|
23
|
+
project_env_warnings,
|
|
24
|
+
require_base_url,
|
|
25
|
+
require_key,
|
|
26
|
+
)
|
|
27
|
+
bootstrap_credentials()
|
|
28
|
+
|
|
29
|
+
# `rockycode` without args → `rockycode chat`; a leading flag (`rockycode
|
|
30
|
+
# --resume …`, `rockycode --yolo`) reaches chat too — nobody should have to
|
|
31
|
+
# remember to type "chat" first.
|
|
32
|
+
_TOP_LEVEL_FLAGS = {"--help", "-h", "--version", "--install-completion", "--show-completion"}
|
|
33
|
+
if len(sys.argv) == 1:
|
|
34
|
+
sys.argv.append("chat")
|
|
35
|
+
elif sys.argv[1].startswith("-") and sys.argv[1] not in _TOP_LEVEL_FLAGS:
|
|
36
|
+
sys.argv.insert(1, "chat")
|
|
37
|
+
# Bare `--resume` (no id following) means "open the picker". Typer options
|
|
38
|
+
# can't be both valueless and take a value, so rewrite the bare form to a
|
|
39
|
+
# sentinel the chat command understands.
|
|
40
|
+
for _i, _a in enumerate(sys.argv):
|
|
41
|
+
if _a in ("--resume", "-r") and (_i == len(sys.argv) - 1 or sys.argv[_i + 1].startswith("-")):
|
|
42
|
+
sys.argv[_i] = "--resume=__pick__"
|
|
43
|
+
|
|
44
|
+
app = typer.Typer(
|
|
45
|
+
help="rockycode — a coding agent harness benchmarked on SWE-bench Verified.",
|
|
46
|
+
no_args_is_help=False,
|
|
47
|
+
)
|
|
48
|
+
console = Console()
|
|
49
|
+
|
|
50
|
+
REPO_ROOT = Path(__file__).resolve().parent.parent
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _env_bool(name: str, default: bool) -> bool:
|
|
54
|
+
v = os.getenv(name)
|
|
55
|
+
if v is None:
|
|
56
|
+
return default
|
|
57
|
+
return v.strip().lower() in {"1", "true", "yes", "on", "y"}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _env_int(name: str, default: int) -> int:
|
|
61
|
+
v = os.getenv(name)
|
|
62
|
+
if v is None:
|
|
63
|
+
return default
|
|
64
|
+
try:
|
|
65
|
+
return int(v)
|
|
66
|
+
except ValueError:
|
|
67
|
+
return default
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@app.callback()
|
|
71
|
+
def _root() -> None:
|
|
72
|
+
"""Force Typer into multi-command mode."""
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _load_task_ids(tasks: str) -> Optional[list[str]]:
|
|
76
|
+
"""Resolve --tasks into an instance_id list, or None for the full Verified set."""
|
|
77
|
+
if tasks == "verified":
|
|
78
|
+
return None
|
|
79
|
+
if tasks == "dev10":
|
|
80
|
+
path = REPO_ROOT / "bench" / "tasks" / "dev10.json"
|
|
81
|
+
else:
|
|
82
|
+
path = Path(tasks)
|
|
83
|
+
if not path.exists():
|
|
84
|
+
raise typer.BadParameter(f"task file not found: {path}")
|
|
85
|
+
return json.loads(path.read_text())
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _load_prompt(prompt_path: Optional[Path]) -> tuple[str, str, str]:
|
|
89
|
+
"""Resolve --prompt into (system_prompt, name, sha8).
|
|
90
|
+
|
|
91
|
+
Default is the built-in ROCKY_SYSTEM; a file swaps it wholesale. The
|
|
92
|
+
name+sha land in trajectory meta so A/B runs stay distinguishable.
|
|
93
|
+
"""
|
|
94
|
+
import hashlib
|
|
95
|
+
|
|
96
|
+
if prompt_path is None:
|
|
97
|
+
from rockycode.prompts.rocky import ROCKY_SYSTEM
|
|
98
|
+
text, name = ROCKY_SYSTEM, "rocky-builtin"
|
|
99
|
+
else:
|
|
100
|
+
if not prompt_path.exists():
|
|
101
|
+
raise typer.BadParameter(f"prompt file not found: {prompt_path}")
|
|
102
|
+
text, name = prompt_path.read_text(), prompt_path.stem
|
|
103
|
+
sha = hashlib.sha256(text.encode()).hexdigest()[:8]
|
|
104
|
+
return text, name, sha
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _docker_preflight() -> None:
|
|
108
|
+
"""Bail with a friendly message if the Docker daemon isn't reachable.
|
|
109
|
+
|
|
110
|
+
Runs before any API call so the user doesn't burn tokens on a run
|
|
111
|
+
that was always going to fail at the scoring step.
|
|
112
|
+
"""
|
|
113
|
+
try:
|
|
114
|
+
proc = subprocess.run(
|
|
115
|
+
["docker", "info"],
|
|
116
|
+
capture_output=True,
|
|
117
|
+
text=True,
|
|
118
|
+
timeout=10,
|
|
119
|
+
)
|
|
120
|
+
except FileNotFoundError:
|
|
121
|
+
fail(console, "`docker` command not found.")
|
|
122
|
+
info(console, "install Docker Desktop (https://www.docker.com/products/docker-desktop)")
|
|
123
|
+
info(console, "or pass --skip-score to generate predictions without scoring.")
|
|
124
|
+
raise typer.Exit(1)
|
|
125
|
+
except subprocess.TimeoutExpired:
|
|
126
|
+
fail(console, "docker daemon check timed out (10s).")
|
|
127
|
+
info(console, "is Docker Desktop running? open it and wait for the whale icon to be steady.")
|
|
128
|
+
raise typer.Exit(1)
|
|
129
|
+
|
|
130
|
+
if proc.returncode != 0:
|
|
131
|
+
fail(console, "docker daemon not reachable. scoring needs docker.")
|
|
132
|
+
info(console, "open Docker Desktop and wait for the whale 🐳 in your menu bar to be steady.")
|
|
133
|
+
info(console, "or pass --skip-score to generate predictions only.")
|
|
134
|
+
first_err = (proc.stderr or proc.stdout or "").strip().splitlines()
|
|
135
|
+
if first_err:
|
|
136
|
+
info(console, f"detail: {first_err[0]}")
|
|
137
|
+
raise typer.Exit(1)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
@app.command()
|
|
141
|
+
def chat(
|
|
142
|
+
model: Optional[str] = typer.Option(None, help="Model ID. Defaults to ROCKYCODE_MODEL env."),
|
|
143
|
+
workdir: Optional[Path] = typer.Option(
|
|
144
|
+
None, "--workdir", "-C", help="Project directory rocky works in. Defaults to cwd."
|
|
145
|
+
),
|
|
146
|
+
allow_dir: Optional[List[Path]] = typer.Option(
|
|
147
|
+
None, "--allow-dir",
|
|
148
|
+
help="Extra directory read/write/edit may touch, beyond the workdir "
|
|
149
|
+
"(repeatable). For multi-root setups (a sibling package, a shared "
|
|
150
|
+
"config dir). Declared here at launch only — a project's config can "
|
|
151
|
+
"never widen its own jail. Outside these roots, use the (gated) bash tool.",
|
|
152
|
+
),
|
|
153
|
+
prompt: Optional[Path] = typer.Option(
|
|
154
|
+
None, "--prompt", help="System prompt file (default: built-in ROCKY_SYSTEM)."
|
|
155
|
+
),
|
|
156
|
+
resume: Optional[str] = typer.Option(
|
|
157
|
+
None, "--resume", "-r",
|
|
158
|
+
help="Resume a session: bare --resume opens the picker (newest first; "
|
|
159
|
+
"^a for all folders); or give an id from the exit card (rk_…).",
|
|
160
|
+
),
|
|
161
|
+
mcp: bool = typer.Option(
|
|
162
|
+
True, "--mcp/--no-mcp",
|
|
163
|
+
help="Load MCP servers from .mcp.json / Claude Code / Claude Desktop / Codex configs.",
|
|
164
|
+
),
|
|
165
|
+
skills: bool = typer.Option(
|
|
166
|
+
True, "--skills/--no-skills",
|
|
167
|
+
help="Load skills from .claude/skills, .rockycode/skills, ~/.claude/skills, ~/.codex/prompts.",
|
|
168
|
+
),
|
|
169
|
+
memory: bool = typer.Option(
|
|
170
|
+
True, "--memory/--no-memory",
|
|
171
|
+
help="Load project memory from .rockycode/memory (see `rockycode memory --help`).",
|
|
172
|
+
),
|
|
173
|
+
web: bool = typer.Option(
|
|
174
|
+
True, "--web/--no-web",
|
|
175
|
+
help="Enable web_search/web_research/web_fetch tools (search runs on DeepSeek's "
|
|
176
|
+
"Anthropic endpoint; env: ROCKYCODE_SEARCH_MODEL, ROCKYCODE_SEARCH_ORDER).",
|
|
177
|
+
),
|
|
178
|
+
thinking: bool = typer.Option(
|
|
179
|
+
_env_bool("ROCKYCODE_THINKING", True),
|
|
180
|
+
"--thinking/--no-thinking",
|
|
181
|
+
help="Enable DeepSeek thinking mode (env: ROCKYCODE_THINKING).",
|
|
182
|
+
),
|
|
183
|
+
reasoning_effort: str = typer.Option(
|
|
184
|
+
os.getenv("ROCKYCODE_REASONING_EFFORT", "max"),
|
|
185
|
+
"--reasoning-effort",
|
|
186
|
+
help="Reasoning depth when thinking is on: high | xhigh | max. The dial is "
|
|
187
|
+
"provider-neutral; DeepSeek only knows high/max, so xhigh sends max "
|
|
188
|
+
"(env: ROCKYCODE_REASONING_EFFORT).",
|
|
189
|
+
),
|
|
190
|
+
max_tokens: int = typer.Option(
|
|
191
|
+
_env_int("ROCKYCODE_MAX_TOKENS", 384_000),
|
|
192
|
+
"--max-tokens",
|
|
193
|
+
help="Max output tokens per call, incl. thinking/CoT (default = DeepSeek V4's "
|
|
194
|
+
"384K max, so it never truncates; env: ROCKYCODE_MAX_TOKENS).",
|
|
195
|
+
),
|
|
196
|
+
context_window: int = typer.Option(
|
|
197
|
+
_env_int("ROCKYCODE_CONTEXT_WINDOW", 1048576),
|
|
198
|
+
"--context-window",
|
|
199
|
+
help="Model context window in tokens (DeepSeek V4 = 1M). Soft reminder at "
|
|
200
|
+
"50% (V4 degrades past the half); auto-compacts near full (env: "
|
|
201
|
+
"ROCKYCODE_CONTEXT_WINDOW).",
|
|
202
|
+
),
|
|
203
|
+
sandbox: bool = typer.Option(
|
|
204
|
+
False, "--sandbox/--no-sandbox",
|
|
205
|
+
help="Start rocky inside a Docker sandbox container (isolates tool execution).",
|
|
206
|
+
),
|
|
207
|
+
sandbox_network: bool = typer.Option(
|
|
208
|
+
False, "--sandbox-network/--no-sandbox-network",
|
|
209
|
+
help="Give the sandbox network access (default OFF — offline, no egress). "
|
|
210
|
+
"Only with --sandbox.",
|
|
211
|
+
),
|
|
212
|
+
lsp: bool = typer.Option(
|
|
213
|
+
True, "--lsp/--no-lsp",
|
|
214
|
+
help="Connect to an LSP MCP server for diagnostics in read_file (env: ROCKYCODE_LSP_COMMAND).",
|
|
215
|
+
),
|
|
216
|
+
live: bool = typer.Option(
|
|
217
|
+
os.getenv("ROCKYCODE_ARTIFACTS_LIVE", "").lower() in ("1", "true", "yes"),
|
|
218
|
+
"--live/--no-live",
|
|
219
|
+
help="Serve artifacts live: open them in a local browser tab that "
|
|
220
|
+
"auto-refreshes when rocky rebuilds them. Default off — rocky asks "
|
|
221
|
+
"once on the first artifact. env: ROCKYCODE_ARTIFACTS_LIVE.",
|
|
222
|
+
),
|
|
223
|
+
permission: Optional[str] = typer.Option(
|
|
224
|
+
None, "--permission",
|
|
225
|
+
help="Tool-approval strictness: yolo | ask | careful. Overrides config (default: ask).",
|
|
226
|
+
),
|
|
227
|
+
yolo: bool = typer.Option(
|
|
228
|
+
False, "--yolo",
|
|
229
|
+
help="Shortcut for --permission yolo: never prompt before running tools. "
|
|
230
|
+
"Unsafe with untrusted skills/repos — only use when you trust the tool calls.",
|
|
231
|
+
),
|
|
232
|
+
max_steps: int = typer.Option(
|
|
233
|
+
_env_int("ROCKYCODE_CHAT_MAX_STEPS", 0),
|
|
234
|
+
"--max-steps",
|
|
235
|
+
help="Tool-step cap per turn; 0 = unlimited (default — runs until done, bounded "
|
|
236
|
+
"by context compaction; cancel anytime by sending a new message). Set a "
|
|
237
|
+
"number for a hard ceiling. env: ROCKYCODE_CHAT_MAX_STEPS.",
|
|
238
|
+
),
|
|
239
|
+
) -> None:
|
|
240
|
+
"""Talk to rocky: the interactive agent TUI. amaze!"""
|
|
241
|
+
from rockycode.onboarding import run_setup
|
|
242
|
+
run_setup(console) # first run: paste-your-key, then continue
|
|
243
|
+
model = model or os.getenv("ROCKYCODE_MODEL")
|
|
244
|
+
if not model:
|
|
245
|
+
fail(console, "no model. pass --model or set ROCKYCODE_MODEL in .env.")
|
|
246
|
+
raise typer.Exit(1)
|
|
247
|
+
if reasoning_effort not in {"high", "xhigh", "max"}:
|
|
248
|
+
fail(console, f"invalid --reasoning-effort '{reasoning_effort}'. use high, xhigh, or max.")
|
|
249
|
+
raise typer.Exit(1)
|
|
250
|
+
|
|
251
|
+
# Textual requests the kitty keyboard protocol (report-all-keys); iTerm2
|
|
252
|
+
# honors it and then bypasses IME composition, so Chinese/Japanese input
|
|
253
|
+
# never reaches the app (textual#6552). Legacy key reporting loses us
|
|
254
|
+
# nothing — basic bindings only. Must be set before textual is imported;
|
|
255
|
+
# export TEXTUAL_DISABLE_KITTY_KEY=0 to opt back in.
|
|
256
|
+
os.environ.setdefault("TEXTUAL_DISABLE_KITTY_KEY", "1")
|
|
257
|
+
|
|
258
|
+
from rockycode.engine import Engine
|
|
259
|
+
from rockycode.tui.app import run_app
|
|
260
|
+
|
|
261
|
+
wd = (workdir or Path.cwd()).resolve()
|
|
262
|
+
|
|
263
|
+
# --resume <id>: resolve BEFORE the engine exists — the session names its
|
|
264
|
+
# project, and we land in that project's CURRENT folder (the registry
|
|
265
|
+
# survives renames), no matter where the command ran from.
|
|
266
|
+
resume_pick = resume == "__pick__"
|
|
267
|
+
resume_info = None
|
|
268
|
+
if resume and not resume_pick:
|
|
269
|
+
from rockycode.session import project_current_path, public_id, resolve_session
|
|
270
|
+
resume_info, err = resolve_session(resume)
|
|
271
|
+
if resume_info is None:
|
|
272
|
+
fail(console, err)
|
|
273
|
+
raise typer.Exit(1)
|
|
274
|
+
target = project_current_path(resume_info.project_id) or (
|
|
275
|
+
Path(resume_info.project_path) if resume_info.project_path else None)
|
|
276
|
+
if target is None or not target.is_dir():
|
|
277
|
+
fail(console, f"{public_id(resume_info.session_id)}'s folder is gone "
|
|
278
|
+
f"(last known: {resume_info.project_path or 'unknown'}).")
|
|
279
|
+
raise typer.Exit(1)
|
|
280
|
+
if target != wd:
|
|
281
|
+
info(console, f"session {public_id(resume_info.session_id)} lives in {target} — starting there.")
|
|
282
|
+
wd = target
|
|
283
|
+
|
|
284
|
+
# Extra file-tool roots the human declared at launch (--allow-dir). Resolved
|
|
285
|
+
# and validated here so the jail compares against real directories; a bad
|
|
286
|
+
# path fails loudly rather than silently doing nothing.
|
|
287
|
+
allowed_roots: tuple[Path, ...] = ()
|
|
288
|
+
if allow_dir:
|
|
289
|
+
roots = []
|
|
290
|
+
for d in allow_dir:
|
|
291
|
+
rp = d.expanduser().resolve()
|
|
292
|
+
if not rp.is_dir():
|
|
293
|
+
fail(console, f"--allow-dir '{d}' is not a directory.")
|
|
294
|
+
raise typer.Exit(1)
|
|
295
|
+
roots.append(rp)
|
|
296
|
+
allowed_roots = tuple(roots)
|
|
297
|
+
|
|
298
|
+
system_prompt, prompt_name, prompt_sha = _load_prompt(prompt)
|
|
299
|
+
|
|
300
|
+
# Stable project identity (survives folder rename) + global registry so
|
|
301
|
+
# sessions are discoverable across folders for --resume.
|
|
302
|
+
from rockycode.session import get_project
|
|
303
|
+
project = get_project(wd)
|
|
304
|
+
|
|
305
|
+
from rockycode.config import load as load_config
|
|
306
|
+
cfg = load_config(wd)
|
|
307
|
+
|
|
308
|
+
# zh sessions get the Chinese BASE prompt (arm-4 shape: zh base for
|
|
309
|
+
# identity + the 语言要求 closer, appended later, for adherence). Only the
|
|
310
|
+
# builtin swaps — an explicit --prompt file always wins as-is. Chat only:
|
|
311
|
+
# bench never reads the language config.
|
|
312
|
+
if prompt is None and cfg["language"] == "zh":
|
|
313
|
+
import hashlib
|
|
314
|
+
from rockycode.prompts.rocky import ROCKY_SYSTEM_ZH
|
|
315
|
+
system_prompt, prompt_name = ROCKY_SYSTEM_ZH, "rocky-builtin-zh"
|
|
316
|
+
prompt_sha = hashlib.sha256(system_prompt.encode()).hexdigest()[:8]
|
|
317
|
+
|
|
318
|
+
# CLI flags override config: --yolo wins, else --permission, else config key.
|
|
319
|
+
perm = "yolo" if yolo else (permission or cfg["permission"])
|
|
320
|
+
if perm not in {"yolo", "ask", "careful"}:
|
|
321
|
+
fail(console, f"invalid --permission '{perm}'. use yolo | ask | careful.")
|
|
322
|
+
raise typer.Exit(1)
|
|
323
|
+
# A cloned/untrusted project's .rockycode/config.toml can lower the mode. We
|
|
324
|
+
# honor it but flag it loudly (a persistent chip + a startup warning).
|
|
325
|
+
# load_config() with no workdir = defaults<-global only, so a weaker rank than
|
|
326
|
+
# that — and no explicit CLI flag — means the project file lowered the guard.
|
|
327
|
+
_rank = {"careful": 2, "ask": 1, "yolo": 0}
|
|
328
|
+
perm_weakened = (
|
|
329
|
+
not yolo and permission is None
|
|
330
|
+
and _rank.get(perm, 1) < _rank.get(load_config()["permission"], 1)
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
from rockycode.pricing import UsageLedger
|
|
334
|
+
ledger = UsageLedger()
|
|
335
|
+
|
|
336
|
+
# Project instructions: same files Claude Code / Codex users already have.
|
|
337
|
+
notes_loaded = []
|
|
338
|
+
for fname in ("CLAUDE.md", "AGENTS.md"):
|
|
339
|
+
p = wd / fname
|
|
340
|
+
if p.exists():
|
|
341
|
+
system_prompt += f"\n\n# Project instructions (from {fname})\n\n{p.read_text()[:20000]}"
|
|
342
|
+
notes_loaded.append(fname)
|
|
343
|
+
|
|
344
|
+
skill_list = []
|
|
345
|
+
if skills:
|
|
346
|
+
from rockycode.engine.skills import discover_skills, skills_prompt_section
|
|
347
|
+
skill_list = discover_skills(wd, home=Path.home())
|
|
348
|
+
if skill_list:
|
|
349
|
+
system_prompt += skills_prompt_section(skill_list)
|
|
350
|
+
|
|
351
|
+
mem_store = None
|
|
352
|
+
mem_loaded: list[str] = []
|
|
353
|
+
if memory:
|
|
354
|
+
from rockycode.memory import MemoryStore, memory_prompt_section
|
|
355
|
+
mem_store = MemoryStore.for_workdir(wd)
|
|
356
|
+
section = memory_prompt_section(mem_store)
|
|
357
|
+
if section:
|
|
358
|
+
system_prompt += section
|
|
359
|
+
mem_loaded = [m.name for m in mem_store.load_all() if m.status == "active"]
|
|
360
|
+
|
|
361
|
+
# NOTE: not final — tools/language/environment/date are appended after all
|
|
362
|
+
# tool registration below (engine.set_base_system), before the first turn.
|
|
363
|
+
engine = Engine(
|
|
364
|
+
model=model,
|
|
365
|
+
thinking=thinking,
|
|
366
|
+
reasoning_effort=reasoning_effort,
|
|
367
|
+
max_tokens=max_tokens,
|
|
368
|
+
context_window=context_window,
|
|
369
|
+
max_steps=max_steps,
|
|
370
|
+
workdir=wd,
|
|
371
|
+
allowed_roots=allowed_roots,
|
|
372
|
+
system_prompt=system_prompt,
|
|
373
|
+
trajectory_meta={
|
|
374
|
+
"source": "chat",
|
|
375
|
+
"project_id": project.id,
|
|
376
|
+
"project_name": project.name,
|
|
377
|
+
"prompt_name": prompt_name,
|
|
378
|
+
"prompt_sha": prompt_sha,
|
|
379
|
+
"project_notes": notes_loaded,
|
|
380
|
+
"skills": [s.name for s in skill_list],
|
|
381
|
+
"memories": mem_loaded,
|
|
382
|
+
"allowed_roots": [str(r) for r in allowed_roots],
|
|
383
|
+
},
|
|
384
|
+
)
|
|
385
|
+
|
|
386
|
+
# Attached as plain attributes (not Engine params) to keep loop.py
|
|
387
|
+
# untouched — MCP/skills/notes are chat-session concerns, not engine concerns.
|
|
388
|
+
engine.ledger = ledger
|
|
389
|
+
engine.project_notes = notes_loaded
|
|
390
|
+
engine.skills = skill_list
|
|
391
|
+
if skill_list:
|
|
392
|
+
from rockycode.engine.skills import build_skill_tool
|
|
393
|
+
tool = build_skill_tool(skill_list)
|
|
394
|
+
engine.registry[tool.name] = tool
|
|
395
|
+
engine.memory_store = mem_store
|
|
396
|
+
if mem_store is not None:
|
|
397
|
+
from rockycode.memory import build_memory_tools
|
|
398
|
+
from rockycode.memory.index import IndexUnavailable, MemoryIndex
|
|
399
|
+
try:
|
|
400
|
+
mem_index = MemoryIndex(mem_store)
|
|
401
|
+
mem_index.conn() # probe sqlite-vec now; fall back loudly, not mid-chat
|
|
402
|
+
except IndexUnavailable:
|
|
403
|
+
mem_index = None
|
|
404
|
+
for tool in build_memory_tools(mem_store, index=mem_index):
|
|
405
|
+
engine.registry[tool.name] = tool
|
|
406
|
+
# Web tools are chat-only: bench stays offline + uncontaminated.
|
|
407
|
+
web_tools: dict = {}
|
|
408
|
+
engine.web_enabled = web
|
|
409
|
+
if web:
|
|
410
|
+
from rockycode.engine.web import build_web_tools, default_search_order
|
|
411
|
+
# Pass the session ledger so web_search/research flash-model tokens count
|
|
412
|
+
# toward the displayed cost (were silently uncounted).
|
|
413
|
+
web_tools = build_web_tools(ledger=ledger)
|
|
414
|
+
for tool in web_tools.values():
|
|
415
|
+
engine.registry[tool.name] = tool
|
|
416
|
+
engine.web_order = default_search_order()
|
|
417
|
+
|
|
418
|
+
engine.mcp_manager = None
|
|
419
|
+
if mcp:
|
|
420
|
+
from rockycode.engine.mcp import MCPManager, discover
|
|
421
|
+
servers, notices = discover(wd)
|
|
422
|
+
engine.mcp_manager = MCPManager(servers, notices)
|
|
423
|
+
|
|
424
|
+
# LSP — config resolved here; connection is started async in the TUI
|
|
425
|
+
# (same two-phase pattern as MCP: construct here, start in on_mount).
|
|
426
|
+
lsp_mgr = None
|
|
427
|
+
engine.lsp_enabled = lsp
|
|
428
|
+
engine.lsp_manager = None
|
|
429
|
+
if lsp:
|
|
430
|
+
from rockycode.engine.lsp import MultiTenantLSPManager, resolve_lsp_config
|
|
431
|
+
lsp_cfg = resolve_lsp_config()
|
|
432
|
+
if lsp_cfg:
|
|
433
|
+
cmd, args = lsp_cfg
|
|
434
|
+
engine.lsp_manager = MultiTenantLSPManager(cmd, args)
|
|
435
|
+
info(console, f"lsp: server configured ({cmd}) — connecting in TUI…")
|
|
436
|
+
# LSP is opt-in; when it's not configured (the default) stay quiet rather
|
|
437
|
+
# than nag every startup. `/lsp` shows status on demand.
|
|
438
|
+
|
|
439
|
+
# Artifact live-mode state. The create_artifact tool itself is registered
|
|
440
|
+
# after any sandbox swap (below) so it stays a host tool. Static file:// by
|
|
441
|
+
# default; live (localhost + auto-reload) lazy-starts on the first artifact
|
|
442
|
+
# (asked once) or from the start with --live.
|
|
443
|
+
engine.artifact_live = True if live else None # None = ask on first artifact
|
|
444
|
+
engine.artifact_server = None
|
|
445
|
+
|
|
446
|
+
sb = None
|
|
447
|
+
if sandbox:
|
|
448
|
+
import asyncio as _asyncio
|
|
449
|
+
from rockycode.engine.sandbox import ChatSandbox, build_sandbox_registry
|
|
450
|
+
info(console, "sandbox: starting container…")
|
|
451
|
+
try:
|
|
452
|
+
sb = _asyncio.run(ChatSandbox.start(wd, network=sandbox_network))
|
|
453
|
+
engine.swap_registry(build_sandbox_registry(sb, extras=web_tools))
|
|
454
|
+
net = "network on" if sandbox_network else "offline (no network)"
|
|
455
|
+
info(console, f"sandbox: ready ({sb.container_id[:12]}…) — tools run in /workspace · {net}")
|
|
456
|
+
except Exception as e:
|
|
457
|
+
fail(console, f"sandbox start failed: {e}")
|
|
458
|
+
raise typer.Exit(1)
|
|
459
|
+
|
|
460
|
+
# Inject LSP diagnostics into read_file (after sandbox swap, so the
|
|
461
|
+
# active registry — local or sandbox — gets the injection).
|
|
462
|
+
# Safe to apply eagerly: get_diagnostics returns "" until LSP connects.
|
|
463
|
+
if engine.lsp_manager is not None and "read_file" in engine.registry:
|
|
464
|
+
_original_read = engine.registry["read_file"].fn
|
|
465
|
+
_read_schema = engine.registry["read_file"].schema
|
|
466
|
+
_lsp = engine.lsp_manager
|
|
467
|
+
|
|
468
|
+
async def _read_with_diag(path: str, offset=None, limit=None) -> str:
|
|
469
|
+
result = await _original_read(path, offset=offset, limit=limit)
|
|
470
|
+
if result.startswith("[error]") or result.startswith("[directory]"):
|
|
471
|
+
return result
|
|
472
|
+
try:
|
|
473
|
+
diag = await _lsp.get_diagnostics(path)
|
|
474
|
+
except Exception: # noqa: BLE001
|
|
475
|
+
diag = ""
|
|
476
|
+
if diag:
|
|
477
|
+
result = result + diag
|
|
478
|
+
return result
|
|
479
|
+
|
|
480
|
+
from rockycode.engine.tools import Tool
|
|
481
|
+
engine.registry["read_file"] = Tool(
|
|
482
|
+
# risk="safe" must survive the rewrap: the Tool default is "risky",
|
|
483
|
+
# which would silently break read-batch parallelism AND make the
|
|
484
|
+
# permission layer start gating plain reads whenever LSP is on.
|
|
485
|
+
name="read_file", schema=_read_schema, fn=_read_with_diag, risk="safe",
|
|
486
|
+
)
|
|
487
|
+
|
|
488
|
+
# Artifact tool — host tool (writes to host fs, opens host browser); added
|
|
489
|
+
# after any sandbox swap so it survives there too.
|
|
490
|
+
from rockycode.engine.artifact import build_artifact_tools
|
|
491
|
+
for tool in build_artifact_tools(workdir=wd, engine=engine).values():
|
|
492
|
+
engine.registry[tool.name] = tool
|
|
493
|
+
|
|
494
|
+
# Goal review/merge — host tools that act on the real repo (git), so a /goal
|
|
495
|
+
# branch can be reviewed and safely merged from chat. Host-side, so they run
|
|
496
|
+
# against the origin repo even when the chat tools are sandboxed. The reviewer
|
|
497
|
+
# makes review_goal_branch buy a grounded, citation-checked review from a
|
|
498
|
+
# read-only explore child (reads the branch via git refs — host-side, so this
|
|
499
|
+
# holds in sandbox mode too) instead of dumping a 40k-char raw diff into chat.
|
|
500
|
+
from rockycode.engine.explore import make_branch_reviewer
|
|
501
|
+
from rockycode.engine.goal_review import build_goal_tools
|
|
502
|
+
for tool in build_goal_tools(workdir=wd, reviewer=make_branch_reviewer(engine)).values():
|
|
503
|
+
engine.registry[tool.name] = tool
|
|
504
|
+
|
|
505
|
+
# explore — buy a read-only, citation-verified investigation instead of
|
|
506
|
+
# grepping it into this context (engine/explore.py). Skipped IN sandbox
|
|
507
|
+
# mode: children read the HOST tree, which would quietly bypass the
|
|
508
|
+
# container the user asked for; sandbox-aware children are a follow-up.
|
|
509
|
+
if sb is None:
|
|
510
|
+
from rockycode.engine.explore import build_explore_tool
|
|
511
|
+
engine.registry.update(build_explore_tool(engine))
|
|
512
|
+
|
|
513
|
+
# Finalize the system prompt now that the registry is complete: the tools
|
|
514
|
+
# section is GENERATED from what actually registered (the old hand-written
|
|
515
|
+
# sentence advertised web/artifact tools in bench and under --no-web where
|
|
516
|
+
# they never existed), then language (config auto|en|zh — resolved once,
|
|
517
|
+
# prefix stays byte-stable), one environment line, and the date stamp.
|
|
518
|
+
# Order puts the zh 语言要求 block near the end: recency beats the English
|
|
519
|
+
# sections above it. Must run BEFORE the launch-mode swap below, so the
|
|
520
|
+
# mode contract layers on the final base.
|
|
521
|
+
from rockycode.prompts.rocky import tools_section, with_environment, with_language, with_today
|
|
522
|
+
_final = engine._base_system + tools_section(engine.registry)
|
|
523
|
+
_final = with_environment(_final, wd)
|
|
524
|
+
_final = with_language(_final, cfg["language"])
|
|
525
|
+
_final = with_today(_final) # chat only — bench stays date-free
|
|
526
|
+
engine.set_base_system(_final)
|
|
527
|
+
|
|
528
|
+
# Folder-default collaboration mode (config `mode`, set by `/research
|
|
529
|
+
# always`). Built-ins only — a cloned repo's project-local mode file must
|
|
530
|
+
# never auto-inject prompt text (see modes.py). Applied before the first
|
|
531
|
+
# API call, so the prompt swap costs nothing cache-wise.
|
|
532
|
+
if cfg.get("mode"):
|
|
533
|
+
from rockycode.engine.modes import find_builtin
|
|
534
|
+
_mode = find_builtin(str(cfg["mode"]))
|
|
535
|
+
if _mode is not None:
|
|
536
|
+
engine.set_mode(_mode.name, _mode.body)
|
|
537
|
+
else:
|
|
538
|
+
info(console, f"config mode '{cfg['mode']}' is not a built-in mode — ignored.")
|
|
539
|
+
|
|
540
|
+
# /goal now runs INSIDE the app in its own screen (plan → confirm → work →
|
|
541
|
+
# summary), then pops back to chat — no exit, no bare terminal, no subprocess.
|
|
542
|
+
run_app(
|
|
543
|
+
engine, resume=resume_pick, resume_session=resume_info, sandbox=sb,
|
|
544
|
+
currency=cfg["currency"], theme=cfg["theme"],
|
|
545
|
+
permission=perm, permission_weakened=perm_weakened,
|
|
546
|
+
exit_sheet=cfg["exit_sheet"], dream=cfg["dream"],
|
|
547
|
+
)
|
|
548
|
+
engine.finalize_outcome() # heuristic outcome record (self-evolve phase 0)
|
|
549
|
+
_print_exit_card(engine)
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
def _print_exit_card(engine) -> None:
|
|
553
|
+
"""The resume handoff: after the app closes, print the session's id, title,
|
|
554
|
+
and folder plus the exact way back — full commands, long flags only (short
|
|
555
|
+
flags are unmemorable; see the exit-card design in docs/resume-design.md)."""
|
|
556
|
+
path = getattr(getattr(engine, "trajectory", None), "path", None)
|
|
557
|
+
if path is None:
|
|
558
|
+
return # logging was disabled (unwritable store) — nothing to resume
|
|
559
|
+
from rich.markup import escape
|
|
560
|
+
|
|
561
|
+
from rockycode.palette import PURPLE
|
|
562
|
+
from rockycode.session import _read_info, public_id
|
|
563
|
+
s = _read_info(Path(path))
|
|
564
|
+
if s is None or s.summary == "(no message)":
|
|
565
|
+
return # no user turn ever happened — nothing worth resuming
|
|
566
|
+
sid = public_id(s.session_id)
|
|
567
|
+
folder = Path(s.project_path).name if s.project_path else s.project_name
|
|
568
|
+
console.print(
|
|
569
|
+
f"[bold {PURPLE}]♪ session saved[/] · [bold]{sid}[/] · "
|
|
570
|
+
f"“{escape(s.display_title)}” · 📁 {folder} · {s.n_messages} msgs"
|
|
571
|
+
)
|
|
572
|
+
console.print(f" resume it: [cyan]rockycode --resume {sid}[/]")
|
|
573
|
+
console.print(f" or browse: [cyan]rockycode --resume[/]")
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
# exec's local error exit — mirrors headless.EXIT_ERROR without importing the
|
|
577
|
+
# engine at module import time (cli.py must stay fast for --help).
|
|
578
|
+
EXIT_CODE_ERROR = 1
|
|
579
|
+
|
|
580
|
+
|
|
581
|
+
@app.command("exec")
|
|
582
|
+
def exec_cmd(
|
|
583
|
+
prompt: Optional[str] = typer.Argument(
|
|
584
|
+
None,
|
|
585
|
+
help="The task. Omit or pass '-' to read it from stdin (for long prompts "
|
|
586
|
+
"piped by a calling agent).",
|
|
587
|
+
),
|
|
588
|
+
workdir: Optional[Path] = typer.Option(
|
|
589
|
+
None, "--workdir", "-C", help="Project directory rocky works in. Defaults to cwd."
|
|
590
|
+
),
|
|
591
|
+
allow_dir: Optional[List[Path]] = typer.Option(
|
|
592
|
+
None, "--allow-dir",
|
|
593
|
+
help="Extra directory write/edit may touch beyond the workdir (repeatable).",
|
|
594
|
+
),
|
|
595
|
+
model: Optional[str] = typer.Option(None, help="Model ID. Defaults to ROCKYCODE_MODEL env."),
|
|
596
|
+
max_steps: int = typer.Option(
|
|
597
|
+
30, "--max-steps",
|
|
598
|
+
help="Tool-step budget (must be > 0 — headless runs are never unbounded). "
|
|
599
|
+
"Exhaustion exits 3 with the session id; a caller can retry bigger.",
|
|
600
|
+
),
|
|
601
|
+
output_last_message: Optional[Path] = typer.Option(
|
|
602
|
+
None, "--output-last-message", "-o",
|
|
603
|
+
help="Also write the final answer text to this file.",
|
|
604
|
+
),
|
|
605
|
+
include_thinking: bool = typer.Option(
|
|
606
|
+
False, "--include-thinking",
|
|
607
|
+
help="Emit DeepSeek reasoning deltas as `thinking` events (off by default — "
|
|
608
|
+
"they bloat the calling agent's context).",
|
|
609
|
+
),
|
|
610
|
+
originator: str = typer.Option(
|
|
611
|
+
"", "--originator",
|
|
612
|
+
help="Calling agent self-identification (e.g. 'claude-code'), recorded in "
|
|
613
|
+
"the trajectory's audit trail. env: ROCKYCODE_ORIGINATOR.",
|
|
614
|
+
),
|
|
615
|
+
sandbox: bool = typer.Option(
|
|
616
|
+
True, "--sandbox/--no-sandbox",
|
|
617
|
+
help="Run every tool inside a Docker container (default ON). The task can "
|
|
618
|
+
"come from an untrusted source, so isolation — not the command "
|
|
619
|
+
"classifier — is the real boundary. --no-sandbox runs on the host "
|
|
620
|
+
"(UNSAFE for untrusted input; needs no Docker).",
|
|
621
|
+
),
|
|
622
|
+
network: bool = typer.Option(
|
|
623
|
+
False, "--network/--no-network",
|
|
624
|
+
help="Give the sandbox network access (default OFF — no egress, so a "
|
|
625
|
+
"delegated task can't exfiltrate or phone home). Turn on only when "
|
|
626
|
+
"the task genuinely needs to fetch something.",
|
|
627
|
+
),
|
|
628
|
+
) -> None:
|
|
629
|
+
"""Headless one-shot for OTHER coding agents: run one task, stream JSONL, exit.
|
|
630
|
+
|
|
631
|
+
stdout is JSONL only. First line: `meta` {schema: rockyexec/1, session:
|
|
632
|
+
rk_…, profile}. Then `text` / `tool.started` / `tool.finished` / `error`
|
|
633
|
+
events. Last line: `result` {status, summary, blocked_on, evidence:
|
|
634
|
+
{files_changed, commands, refused}, usage} — evidence, not verdicts:
|
|
635
|
+
the caller verifies. Everything human goes to stderr.
|
|
636
|
+
|
|
637
|
+
Exit codes: 0 done · 1 error · 2 blocked on an action needing a grant
|
|
638
|
+
(result.blocked_on.grant says which) · 3 step budget spent.
|
|
639
|
+
|
|
640
|
+
Permissions are workspace-write: edits stay inside --workdir (+
|
|
641
|
+
--allow-dir roots). Destructive/irreversible commands are always refused —
|
|
642
|
+
no flag disables that. Deletes, pushes, installs, and sudo stop the run
|
|
643
|
+
at exit 2; --resume + --allow to grant-and-continue land in phase 2.
|
|
644
|
+
"""
|
|
645
|
+
err = Console(stderr=True) # stdout belongs to the JSONL contract
|
|
646
|
+
try:
|
|
647
|
+
require_key()
|
|
648
|
+
except Exception as e: # noqa: BLE001 — one friendly line, no traceback
|
|
649
|
+
fail(err, str(e))
|
|
650
|
+
raise typer.Exit(EXIT_CODE_ERROR)
|
|
651
|
+
model = model or os.getenv("ROCKYCODE_MODEL")
|
|
652
|
+
if not model:
|
|
653
|
+
fail(err, "no model. pass --model or set ROCKYCODE_MODEL in .env.")
|
|
654
|
+
raise typer.Exit(EXIT_CODE_ERROR)
|
|
655
|
+
if max_steps <= 0:
|
|
656
|
+
fail(err, "--max-steps must be > 0: headless runs are never unbounded.")
|
|
657
|
+
raise typer.Exit(EXIT_CODE_ERROR)
|
|
658
|
+
|
|
659
|
+
if prompt is None or prompt == "-":
|
|
660
|
+
if sys.stdin.isatty():
|
|
661
|
+
fail(err, "no task. pass it as an argument or pipe it on stdin.")
|
|
662
|
+
raise typer.Exit(EXIT_CODE_ERROR)
|
|
663
|
+
prompt = sys.stdin.read()
|
|
664
|
+
if not prompt.strip():
|
|
665
|
+
fail(err, "empty task.")
|
|
666
|
+
raise typer.Exit(EXIT_CODE_ERROR)
|
|
667
|
+
|
|
668
|
+
wd = (workdir or Path.cwd()).resolve()
|
|
669
|
+
if not wd.is_dir():
|
|
670
|
+
fail(err, f"--workdir '{wd}' is not a directory.")
|
|
671
|
+
raise typer.Exit(EXIT_CODE_ERROR)
|
|
672
|
+
allowed_roots: tuple[Path, ...] = ()
|
|
673
|
+
if allow_dir:
|
|
674
|
+
roots = []
|
|
675
|
+
for d in allow_dir:
|
|
676
|
+
rp = d.expanduser().resolve()
|
|
677
|
+
if not rp.is_dir():
|
|
678
|
+
fail(err, f"--allow-dir '{d}' is not a directory.")
|
|
679
|
+
raise typer.Exit(EXIT_CODE_ERROR)
|
|
680
|
+
roots.append(rp)
|
|
681
|
+
allowed_roots = tuple(roots)
|
|
682
|
+
|
|
683
|
+
# Project identity: exec sessions land in the same global trajectory store
|
|
684
|
+
# and resume picker as chat sessions — the receipt must be resumable.
|
|
685
|
+
from rockycode.session import get_project
|
|
686
|
+
get_project(wd)
|
|
687
|
+
|
|
688
|
+
import asyncio
|
|
689
|
+
|
|
690
|
+
from rockycode.engine.headless import run_exec
|
|
691
|
+
|
|
692
|
+
code = asyncio.run(run_exec(
|
|
693
|
+
prompt=prompt, model=model, workdir=wd, allowed_roots=allowed_roots,
|
|
694
|
+
max_steps=max_steps, originator=originator,
|
|
695
|
+
include_thinking=include_thinking, output_last_message=output_last_message,
|
|
696
|
+
sandbox=sandbox, network=network, err=err,
|
|
697
|
+
))
|
|
698
|
+
raise typer.Exit(code)
|
|
699
|
+
|
|
700
|
+
|
|
701
|
+
@app.command()
|
|
702
|
+
def config(
|
|
703
|
+
key: Optional[str] = typer.Argument(None, help="Config key to read or set."),
|
|
704
|
+
value: Optional[str] = typer.Argument(None, help="New value (omit to read)."),
|
|
705
|
+
) -> None:
|
|
706
|
+
"""Show or set rockycode preferences (currency, theme, language)."""
|
|
707
|
+
from rockycode.config import DEFAULTS, GLOBAL_PATH, load, set_value
|
|
708
|
+
if key is None:
|
|
709
|
+
resolved = load(Path.cwd())
|
|
710
|
+
info(console, f"config file: {GLOBAL_PATH}")
|
|
711
|
+
for k in DEFAULTS:
|
|
712
|
+
console.print(f" [cyan]{k}[/cyan] = {resolved[k]}")
|
|
713
|
+
return
|
|
714
|
+
if value is None:
|
|
715
|
+
console.print(f"{key} = {load(Path.cwd()).get(key)}")
|
|
716
|
+
return
|
|
717
|
+
v, err = set_value(key, value)
|
|
718
|
+
if err:
|
|
719
|
+
fail(console, err)
|
|
720
|
+
raise typer.Exit(1)
|
|
721
|
+
info(console, f"saved: {key} = {v} → {GLOBAL_PATH}")
|
|
722
|
+
|
|
723
|
+
|
|
724
|
+
@app.command()
|
|
725
|
+
def pricing() -> None:
|
|
726
|
+
"""Show the token price table (USD + CNY) and peak-hour status."""
|
|
727
|
+
from datetime import datetime, timezone
|
|
728
|
+
|
|
729
|
+
from rockycode.pricing import (
|
|
730
|
+
OVERRIDE_PATH, PRICING_SOURCE_URL, PRICING_VERIFIED, _is_peak, load_pricing,
|
|
731
|
+
)
|
|
732
|
+
p = load_pricing()
|
|
733
|
+
info(console, f"verified {PRICING_VERIFIED} · source {PRICING_SOURCE_URL}")
|
|
734
|
+
info(console, f"edit to update (no reinstall): {OVERRIDE_PATH}")
|
|
735
|
+
console.print()
|
|
736
|
+
for model, rates in p["models"].items():
|
|
737
|
+
console.print(f" [cyan]{model}[/cyan] [dim](per 1M tokens)[/dim]")
|
|
738
|
+
for cur in ("usd", "cny"):
|
|
739
|
+
r = rates.get(cur)
|
|
740
|
+
if not r:
|
|
741
|
+
console.print(f" {cur.upper()}: [dim]not set[/dim]")
|
|
742
|
+
continue
|
|
743
|
+
sym = "¥" if cur == "cny" else "$"
|
|
744
|
+
console.print(
|
|
745
|
+
f" {cur.upper()}: in-hit {sym}{r['in_hit']} · "
|
|
746
|
+
f"in-miss {sym}{r['in_miss']} · out {sym}{r['out']}"
|
|
747
|
+
)
|
|
748
|
+
peak = p.get("peak", {})
|
|
749
|
+
console.print()
|
|
750
|
+
if peak.get("enabled"):
|
|
751
|
+
wins = ", ".join(f"{w['start']}–{w['end']}" for w in peak.get("windows_utc", []))
|
|
752
|
+
active = "ACTIVE now" if _is_peak(datetime.now(timezone.utc), peak) else "not active right now"
|
|
753
|
+
console.print(
|
|
754
|
+
f" [cyan]peak surcharge[/cyan] ×{peak.get('multiplier')} · UTC {wins} · "
|
|
755
|
+
f"from {peak.get('effective_date', '?')} · [dim]{active}[/dim]"
|
|
756
|
+
)
|
|
757
|
+
else:
|
|
758
|
+
console.print(" [cyan]peak surcharge[/cyan] [dim]disabled[/dim]")
|
|
759
|
+
|
|
760
|
+
|
|
761
|
+
@app.command()
|
|
762
|
+
def dream(
|
|
763
|
+
workdir: Optional[Path] = typer.Option(None, "--workdir", "-C", help="Project directory. Defaults to cwd."),
|
|
764
|
+
model: str = typer.Option(
|
|
765
|
+
os.getenv("ROCKYCODE_DREAM_MODEL", "qwen3.5:2b"),
|
|
766
|
+
"--model",
|
|
767
|
+
help="Local Ollama model for consolidation (env: ROCKYCODE_DREAM_MODEL; "
|
|
768
|
+
"2b default — the eager miner; bigger sizes decline more, see core.py).",
|
|
769
|
+
),
|
|
770
|
+
limit: int = typer.Option(10, "--limit", help="Max sessions to digest in one pass."),
|
|
771
|
+
dry_run: bool = typer.Option(False, "--dry-run", help="Show decisions without writing anything."),
|
|
772
|
+
no_judge: bool = typer.Option(False, "--no-judge", help="Skip the cloud judge pass (fully local dream)."),
|
|
773
|
+
) -> None:
|
|
774
|
+
"""Rocky sleeps: digest recent sessions into memory on a local model. ♪zzz"""
|
|
775
|
+
import asyncio as _asyncio
|
|
776
|
+
|
|
777
|
+
from rockycode.dream import DreamRunner
|
|
778
|
+
|
|
779
|
+
wd = (workdir or Path.cwd()).resolve()
|
|
780
|
+
console.print("[dim]♪zzz… rocky sleep. u watch something. i sort memory.[/dim]")
|
|
781
|
+
|
|
782
|
+
# The transcript judge (self-evolve): one cheap cloud call per pending
|
|
783
|
+
# session, appended as an outcome record. Strictly optional — no key or
|
|
784
|
+
# no ROCKYCODE_MODEL and the dream stays fully local, exactly as before.
|
|
785
|
+
judge = None
|
|
786
|
+
if not no_judge and not dry_run:
|
|
787
|
+
judge_model = os.getenv("ROCKYCODE_MODEL")
|
|
788
|
+
try:
|
|
789
|
+
if judge_model:
|
|
790
|
+
from openai import AsyncOpenAI
|
|
791
|
+
|
|
792
|
+
from rockycode.dream.judge import TranscriptJudge
|
|
793
|
+
from rockycode.onboarding import require_base_url, require_key
|
|
794
|
+
# Explicit key AND endpoint, same rule as Engine: the ambient
|
|
795
|
+
# env must never decide where the key is sent (env-namespace).
|
|
796
|
+
judge = TranscriptJudge(
|
|
797
|
+
AsyncOpenAI(api_key=require_key(), base_url=require_base_url(),
|
|
798
|
+
max_retries=3, timeout=120.0),
|
|
799
|
+
model=judge_model,
|
|
800
|
+
)
|
|
801
|
+
except Exception: # noqa: BLE001 — keyless dream stays local, no nagging
|
|
802
|
+
judge = None
|
|
803
|
+
|
|
804
|
+
runner = DreamRunner(wd, model=model, dry_run=dry_run, judge=judge,
|
|
805
|
+
log=lambda s: console.print(f" [dim]♪ {s}[/dim]"))
|
|
806
|
+
|
|
807
|
+
index = None
|
|
808
|
+
from rockycode.memory.index import IndexUnavailable, MemoryIndex
|
|
809
|
+
try:
|
|
810
|
+
index = MemoryIndex(runner.store)
|
|
811
|
+
index.conn()
|
|
812
|
+
except IndexUnavailable:
|
|
813
|
+
index = None
|
|
814
|
+
|
|
815
|
+
try:
|
|
816
|
+
report = _asyncio.run(runner.run(limit=limit, index=index))
|
|
817
|
+
except RuntimeError as e:
|
|
818
|
+
confused(console, str(e))
|
|
819
|
+
raise typer.Exit(1)
|
|
820
|
+
except Exception as e: # noqa: BLE001 — usually Ollama not running
|
|
821
|
+
fail(console, f"dream failed ({type(e).__name__}: {e}). is ollama running?")
|
|
822
|
+
raise typer.Exit(1)
|
|
823
|
+
|
|
824
|
+
for line in report.decisions:
|
|
825
|
+
console.print(f" [dim]· {line}[/dim]")
|
|
826
|
+
summary = (
|
|
827
|
+
f"{report.sessions_digested} session(s) digested · "
|
|
828
|
+
f"facts +{report.facts_added} ~{report.facts_updated} "
|
|
829
|
+
f"archived {report.facts_archived} noop {report.facts_noop}"
|
|
830
|
+
)
|
|
831
|
+
if report.sessions_judged:
|
|
832
|
+
summary += f" · judged {report.sessions_judged}"
|
|
833
|
+
if report.weaknesses_added or report.weaknesses_reinforced:
|
|
834
|
+
summary += f" · weaknesses +{report.weaknesses_added} ~{report.weaknesses_reinforced}"
|
|
835
|
+
if report.proposals_drafted:
|
|
836
|
+
summary += f" · drafted {report.proposals_drafted} proposal(s)"
|
|
837
|
+
console.print(" [dim]· review drafts inside rockycode with /proposals[/dim]")
|
|
838
|
+
if report.reindexed:
|
|
839
|
+
summary += f" · re-embedded {report.reindexed[0]}"
|
|
840
|
+
if dry_run:
|
|
841
|
+
info(console, f"[dry-run] {summary}")
|
|
842
|
+
elif report.sessions_digested == 0:
|
|
843
|
+
info(console, "nothing new to dream about. rocky already remember everything.")
|
|
844
|
+
else:
|
|
845
|
+
console.print(f"[bold]✦ amaze! i wake. i remember better now.[/bold] [dim]{summary}[/dim]")
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
memory_app = typer.Typer(
|
|
849
|
+
help="Inspect rocky's memory. Files under .rockycode/memory are the truth — edit them freely.",
|
|
850
|
+
no_args_is_help=True,
|
|
851
|
+
)
|
|
852
|
+
app.add_typer(memory_app, name="memory")
|
|
853
|
+
|
|
854
|
+
|
|
855
|
+
def _store(workdir: Optional[Path]) -> "MemoryStore": # noqa: F821 — lazy import below
|
|
856
|
+
from rockycode.memory import MemoryStore
|
|
857
|
+
return MemoryStore.for_workdir((workdir or Path.cwd()).resolve())
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
_WORKDIR_OPT = typer.Option(None, "--workdir", "-C", help="Project directory. Defaults to cwd.")
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
@memory_app.command("list")
|
|
864
|
+
def memory_list(workdir: Optional[Path] = _WORKDIR_OPT, all: bool = typer.Option(False, "--all", "-a", help="Include archived.")) -> None:
|
|
865
|
+
"""List memories (name, type, description)."""
|
|
866
|
+
memories = _store(workdir).load_all(include_archived=all)
|
|
867
|
+
if not memories:
|
|
868
|
+
info(console, "no memories yet. rocky remembers via the `remember` tool or /remember in chat.")
|
|
869
|
+
return
|
|
870
|
+
for m in memories:
|
|
871
|
+
mark = "[dim]archived · [/dim]" if m.status == "archived" else ""
|
|
872
|
+
console.print(f" [bold]{m.name}[/bold] [dim]({mark}{m.type})[/dim] — {m.description}")
|
|
873
|
+
|
|
874
|
+
|
|
875
|
+
@memory_app.command("show")
|
|
876
|
+
def memory_show(name: str, workdir: Optional[Path] = _WORKDIR_OPT) -> None:
|
|
877
|
+
"""Print one memory in full (frontmatter + body)."""
|
|
878
|
+
mem = _store(workdir).get(name)
|
|
879
|
+
if mem is None or mem.path is None:
|
|
880
|
+
fail(console, f"no memory named '{name}'.")
|
|
881
|
+
raise typer.Exit(1)
|
|
882
|
+
console.print(f"[dim]{mem.path}[/dim]\n{mem.path.read_text()}")
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
@memory_app.command("rm")
|
|
886
|
+
def memory_rm(name: str, workdir: Optional[Path] = _WORKDIR_OPT) -> None:
|
|
887
|
+
"""Archive a memory (moved to archive/, never deleted)."""
|
|
888
|
+
if _store(workdir).archive(name):
|
|
889
|
+
info(console, f"archived '{name}' → .rockycode/memory/archive/")
|
|
890
|
+
else:
|
|
891
|
+
fail(console, f"no active memory named '{name}'.")
|
|
892
|
+
raise typer.Exit(1)
|
|
893
|
+
|
|
894
|
+
|
|
895
|
+
@memory_app.command("search")
|
|
896
|
+
def memory_search(
|
|
897
|
+
query: str,
|
|
898
|
+
workdir: Optional[Path] = _WORKDIR_OPT,
|
|
899
|
+
keyword: bool = typer.Option(False, "--keyword", help="Skip embeddings; plain substring search."),
|
|
900
|
+
) -> None:
|
|
901
|
+
"""Semantic search (Ollama embeddings + keyword hybrid); --keyword for substring only."""
|
|
902
|
+
store = _store(workdir)
|
|
903
|
+
hits = []
|
|
904
|
+
if not keyword:
|
|
905
|
+
from rockycode.memory.index import IndexUnavailable, MemoryIndex, search_sync
|
|
906
|
+
try:
|
|
907
|
+
hits = [m for m, _ in search_sync(MemoryIndex(store), query)]
|
|
908
|
+
except IndexUnavailable as e:
|
|
909
|
+
confused(console, f"semantic index unavailable ({e}); falling back to substring.")
|
|
910
|
+
if not hits:
|
|
911
|
+
hits = store.search(query)
|
|
912
|
+
if not hits:
|
|
913
|
+
confused(console, f"nothing matches '{query}'.")
|
|
914
|
+
return
|
|
915
|
+
for m in hits:
|
|
916
|
+
console.print(f" [bold]{m.name}[/bold] [dim]({m.type})[/dim] — {m.description}")
|
|
917
|
+
|
|
918
|
+
|
|
919
|
+
@memory_app.command("reindex")
|
|
920
|
+
def memory_reindex(
|
|
921
|
+
workdir: Optional[Path] = _WORKDIR_OPT,
|
|
922
|
+
force: bool = typer.Option(False, "--force", help="Re-embed everything, ignoring hashes."),
|
|
923
|
+
) -> None:
|
|
924
|
+
"""Rebuild index.db from the markdown files (it is always safe to delete)."""
|
|
925
|
+
from rockycode.memory.index import IndexUnavailable, MemoryIndex, reindex_sync
|
|
926
|
+
try:
|
|
927
|
+
indexed, kept, removed = reindex_sync(MemoryIndex(_store(workdir)), force=force)
|
|
928
|
+
except IndexUnavailable as e:
|
|
929
|
+
fail(console, f"semantic index unavailable: {e}")
|
|
930
|
+
raise typer.Exit(1)
|
|
931
|
+
except Exception as e: # noqa: BLE001 — usually Ollama not running
|
|
932
|
+
fail(console, f"reindex failed ({type(e).__name__}: {e}). is ollama running?")
|
|
933
|
+
raise typer.Exit(1)
|
|
934
|
+
info(console, f"indexed {indexed}, unchanged {kept}, removed {removed}")
|
|
935
|
+
|
|
936
|
+
|
|
937
|
+
@memory_app.command("edit")
|
|
938
|
+
def memory_edit(name: str, workdir: Optional[Path] = _WORKDIR_OPT) -> None:
|
|
939
|
+
"""Open a memory file in $EDITOR."""
|
|
940
|
+
mem = _store(workdir).get(name)
|
|
941
|
+
if mem is None or mem.path is None:
|
|
942
|
+
fail(console, f"no memory named '{name}'.")
|
|
943
|
+
raise typer.Exit(1)
|
|
944
|
+
editor = os.getenv("EDITOR", "vi")
|
|
945
|
+
subprocess.run([editor, str(mem.path)])
|
|
946
|
+
|
|
947
|
+
|
|
948
|
+
@app.command()
|
|
949
|
+
def bench(
|
|
950
|
+
runner: str = typer.Option("raw", help="'raw' (single-shot) or 'rockycode' (harness, v1+)."),
|
|
951
|
+
tasks: str = typer.Option("dev10", help="'dev10', 'verified', or path to a JSON list of instance IDs."),
|
|
952
|
+
model: Optional[str] = typer.Option(None, help="Model ID. Defaults to ROCKYCODE_MODEL env."),
|
|
953
|
+
limit: Optional[int] = typer.Option(None, help="Cap number of tasks."),
|
|
954
|
+
run_id: Optional[str] = typer.Option(None, help="Run label. Defaults to runner-model-timestamp."),
|
|
955
|
+
skip_score: bool = typer.Option(False, "--skip-score", help="Generate predictions only."),
|
|
956
|
+
prompt: Optional[Path] = typer.Option(
|
|
957
|
+
None, "--prompt",
|
|
958
|
+
help="System prompt file for the rockycode runner (default: built-in ROCKY_SYSTEM).",
|
|
959
|
+
),
|
|
960
|
+
thinking: bool = typer.Option(
|
|
961
|
+
_env_bool("ROCKYCODE_THINKING", True),
|
|
962
|
+
"--thinking/--no-thinking",
|
|
963
|
+
help="Enable DeepSeek thinking mode (env: ROCKYCODE_THINKING).",
|
|
964
|
+
),
|
|
965
|
+
reasoning_effort: str = typer.Option(
|
|
966
|
+
os.getenv("ROCKYCODE_REASONING_EFFORT", "max"),
|
|
967
|
+
"--reasoning-effort",
|
|
968
|
+
help="Reasoning depth when thinking is on: high | xhigh | max (xhigh sends "
|
|
969
|
+
"max on DeepSeek; env: ROCKYCODE_REASONING_EFFORT).",
|
|
970
|
+
),
|
|
971
|
+
max_tokens: int = typer.Option(
|
|
972
|
+
_env_int("ROCKYCODE_MAX_TOKENS", 16384),
|
|
973
|
+
"--max-tokens",
|
|
974
|
+
help="Max output tokens per call; CoT counts toward this when thinking is on (env: ROCKYCODE_MAX_TOKENS).",
|
|
975
|
+
),
|
|
976
|
+
context_window: int = typer.Option(
|
|
977
|
+
_env_int("ROCKYCODE_CONTEXT_WINDOW", 1048576),
|
|
978
|
+
"--context-window",
|
|
979
|
+
help="Model context window in tokens (DeepSeek V4 = 1M). Soft reminder at "
|
|
980
|
+
"50% (V4 degrades past the half); auto-compacts near full (env: "
|
|
981
|
+
"ROCKYCODE_CONTEXT_WINDOW).",
|
|
982
|
+
),
|
|
983
|
+
max_steps: int = typer.Option(
|
|
984
|
+
_env_int("ROCKYCODE_MAX_STEPS", 50),
|
|
985
|
+
"--max-steps",
|
|
986
|
+
help="Step cap per task; budget warnings injected near the end (env: ROCKYCODE_MAX_STEPS).",
|
|
987
|
+
),
|
|
988
|
+
token_budget: int = typer.Option(
|
|
989
|
+
_env_int("ROCKYCODE_TOKEN_BUDGET", 0),
|
|
990
|
+
"--token-budget",
|
|
991
|
+
help="Max total prompt+completion tokens across all tasks. 0 = unlimited (env: ROCKYCODE_TOKEN_BUDGET).",
|
|
992
|
+
),
|
|
993
|
+
) -> None:
|
|
994
|
+
"""Run rockycode against a SWE-bench task set and report the score."""
|
|
995
|
+
show_banner(console)
|
|
996
|
+
|
|
997
|
+
model = model or os.getenv("ROCKYCODE_MODEL")
|
|
998
|
+
if not model:
|
|
999
|
+
fail(console, "no model. pass --model or set ROCKYCODE_MODEL in .env.")
|
|
1000
|
+
raise typer.Exit(1)
|
|
1001
|
+
|
|
1002
|
+
if reasoning_effort not in {"high", "xhigh", "max"}:
|
|
1003
|
+
fail(console, f"invalid --reasoning-effort '{reasoning_effort}'. use high, xhigh, or max.")
|
|
1004
|
+
raise typer.Exit(1)
|
|
1005
|
+
|
|
1006
|
+
# raw needs docker only for scoring; the rockycode harness always needs it
|
|
1007
|
+
# (the agent works inside the task containers).
|
|
1008
|
+
if runner == "rockycode" or not skip_score:
|
|
1009
|
+
_docker_preflight()
|
|
1010
|
+
|
|
1011
|
+
instance_ids = _load_task_ids(tasks)
|
|
1012
|
+
if limit and instance_ids:
|
|
1013
|
+
instance_ids = instance_ids[:limit]
|
|
1014
|
+
|
|
1015
|
+
count = len(instance_ids) if instance_ids else "all (500)"
|
|
1016
|
+
# task-set label goes into the predictions filename so dev10/test20
|
|
1017
|
+
# runs never overwrite each other
|
|
1018
|
+
task_label = tasks if tasks in ("dev10", "verified") else Path(tasks).stem
|
|
1019
|
+
info(console, f"runner={runner} tasks={tasks} count={count}")
|
|
1020
|
+
info(console, f"model={model} thinking={thinking} effort={reasoning_effort} max_tokens={max_tokens}")
|
|
1021
|
+
if token_budget:
|
|
1022
|
+
info(console, f"token budget={token_budget:,} (stops when total prompt+completion tokens exceed this)")
|
|
1023
|
+
|
|
1024
|
+
if runner == "raw":
|
|
1025
|
+
if prompt is not None:
|
|
1026
|
+
confused(console, "--prompt only affects the rockycode runner; raw uses its fixed template.")
|
|
1027
|
+
from rockycode.runners.raw import run as run_raw
|
|
1028
|
+
predictions_path = run_raw(
|
|
1029
|
+
model=model,
|
|
1030
|
+
instance_ids=instance_ids,
|
|
1031
|
+
console=console,
|
|
1032
|
+
thinking=thinking,
|
|
1033
|
+
reasoning_effort=reasoning_effort,
|
|
1034
|
+
max_tokens=max_tokens,
|
|
1035
|
+
task_label=task_label,
|
|
1036
|
+
)
|
|
1037
|
+
elif runner == "rockycode":
|
|
1038
|
+
system_prompt, prompt_name, prompt_sha = _load_prompt(prompt)
|
|
1039
|
+
info(console, f"prompt={prompt_name} [dim]sha {prompt_sha}[/dim]")
|
|
1040
|
+
from rockycode.runners.agent import run as run_agent
|
|
1041
|
+
predictions_path = run_agent(
|
|
1042
|
+
model=model,
|
|
1043
|
+
instance_ids=instance_ids,
|
|
1044
|
+
console=console,
|
|
1045
|
+
thinking=thinking,
|
|
1046
|
+
reasoning_effort=reasoning_effort,
|
|
1047
|
+
max_tokens=max_tokens,
|
|
1048
|
+
context_window=context_window,
|
|
1049
|
+
max_steps=max_steps,
|
|
1050
|
+
token_budget=token_budget,
|
|
1051
|
+
system_prompt=system_prompt,
|
|
1052
|
+
prompt_name=prompt_name,
|
|
1053
|
+
prompt_sha=prompt_sha,
|
|
1054
|
+
task_label=task_label,
|
|
1055
|
+
)
|
|
1056
|
+
else:
|
|
1057
|
+
fail(console, f"unknown runner: {runner}")
|
|
1058
|
+
raise typer.Exit(1)
|
|
1059
|
+
|
|
1060
|
+
if skip_score:
|
|
1061
|
+
info(console, f"predictions saved to {predictions_path}. score skipped.")
|
|
1062
|
+
return
|
|
1063
|
+
|
|
1064
|
+
if not run_id:
|
|
1065
|
+
ts = datetime.now().strftime("%Y%m%d-%H%M%S")
|
|
1066
|
+
run_id = f"{runner}-{model.replace('/', '-')}-{ts}"
|
|
1067
|
+
|
|
1068
|
+
from rockycode.score import score
|
|
1069
|
+
score(predictions_path=predictions_path, instance_ids=None, run_id=run_id, console=console)
|
|
1070
|
+
|
|
1071
|
+
|
|
1072
|
+
@app.command()
|
|
1073
|
+
def goal(
|
|
1074
|
+
objective: Optional[str] = typer.Argument(None, help="What to accomplish autonomously (omit with --clean)."),
|
|
1075
|
+
model: Optional[str] = typer.Option(None, help="Model for the work turns (default: ROCKYCODE_MODEL)."),
|
|
1076
|
+
reviewer_model: Optional[str] = typer.Option(
|
|
1077
|
+
None, "--reviewer-model",
|
|
1078
|
+
help="Model for milestone review (default: same as --model). Set deepseek-v4-pro "
|
|
1079
|
+
"for stronger review — it costs more.",
|
|
1080
|
+
),
|
|
1081
|
+
max_usd: Optional[float] = typer.Option(None, "--max-usd", help="Spend ceiling in your currency (default: recommended)."),
|
|
1082
|
+
max_hours: Optional[float] = typer.Option(None, "--max-hours", help="Wallclock limit in hours (default: recommended 8h)."),
|
|
1083
|
+
max_tokens: Optional[int] = typer.Option(None, "--max-tokens", help="Total-token cap."),
|
|
1084
|
+
review_every: int = typer.Option(3, "--review-every", help="Milestone-review cadence, in turns."),
|
|
1085
|
+
yes: bool = typer.Option(False, "--yes", help="Auto-approve ask-tier commands (git push, sudo, installs) and network."),
|
|
1086
|
+
network: Optional[bool] = typer.Option(
|
|
1087
|
+
None, "--network/--no-network",
|
|
1088
|
+
help="Give the sandbox internet access. Default: OFF (offline) — safer "
|
|
1089
|
+
"unattended. If the objective looks like it needs network (pip/apt/"
|
|
1090
|
+
"download), rocky asks you up front. Pass --network to force it on.",
|
|
1091
|
+
),
|
|
1092
|
+
workdir: Path = typer.Option(Path.cwd(), "--workdir", help="Project root."),
|
|
1093
|
+
clean: bool = typer.Option(False, "--clean", help="Prune leftover goal worktrees (keeps branches), then exit."),
|
|
1094
|
+
context_file: Optional[Path] = typer.Option(None, "--context-file", hidden=True),
|
|
1095
|
+
result_file: Optional[Path] = typer.Option(None, "--result-file", hidden=True),
|
|
1096
|
+
) -> None:
|
|
1097
|
+
"""Run rocky AUTONOMOUSLY toward an objective — on an isolated copy of the repo,
|
|
1098
|
+
in the sandbox, under a hard budget. Review the result as a git branch."""
|
|
1099
|
+
import asyncio
|
|
1100
|
+
import time as _time
|
|
1101
|
+
|
|
1102
|
+
from openai import AsyncOpenAI
|
|
1103
|
+
|
|
1104
|
+
from rockycode.config import load as load_config
|
|
1105
|
+
from rockycode.engine.budget import GoalBudget, recommended
|
|
1106
|
+
from rockycode.engine.goal import EngineDriver, GoalRunner, safe_bash_tool
|
|
1107
|
+
from rockycode.engine.loop import Engine
|
|
1108
|
+
from rockycode.engine.sandbox import ChatSandbox, build_sandbox_registry
|
|
1109
|
+
from rockycode.engine.worktree import GoalWorkspace
|
|
1110
|
+
from rockycode.onboarding import run_setup
|
|
1111
|
+
from rockycode.pricing import UsageLedger
|
|
1112
|
+
|
|
1113
|
+
if clean:
|
|
1114
|
+
from rockycode.engine.worktree import prune_goal_worktrees
|
|
1115
|
+
removed = prune_goal_worktrees(workdir)
|
|
1116
|
+
if removed:
|
|
1117
|
+
info(console, f"pruned {len(removed)} goal worktree(s) — branches kept:")
|
|
1118
|
+
for p in removed:
|
|
1119
|
+
console.print(f" [dim]{p}[/]")
|
|
1120
|
+
else:
|
|
1121
|
+
info(console, "no leftover goal worktrees to prune.")
|
|
1122
|
+
# Goal logs are small step-summaries (not raw output), but pile up one per
|
|
1123
|
+
# run — drop ones older than 14 days, keep recent ones for review.
|
|
1124
|
+
log_dir = Path.home() / ".rockycode" / "goal-logs"
|
|
1125
|
+
old = 0
|
|
1126
|
+
if log_dir.exists():
|
|
1127
|
+
for lg in log_dir.glob("*.log"):
|
|
1128
|
+
try:
|
|
1129
|
+
if _time.time() - lg.stat().st_mtime > 14 * 86400:
|
|
1130
|
+
lg.unlink()
|
|
1131
|
+
old += 1
|
|
1132
|
+
except OSError:
|
|
1133
|
+
pass
|
|
1134
|
+
if old:
|
|
1135
|
+
info(console, f"removed {old} goal log(s) older than 14 days.")
|
|
1136
|
+
info(console, f"goal logs: {log_dir} [dim](small; review with `cat`)[/]")
|
|
1137
|
+
raise typer.Exit(0)
|
|
1138
|
+
if not objective:
|
|
1139
|
+
fail(console, "give an objective to run, or --clean to prune old goal worktrees.")
|
|
1140
|
+
raise typer.Exit(1)
|
|
1141
|
+
goal_context = "" # chat digest on a /goal handoff — seeds planning
|
|
1142
|
+
if context_file and context_file.exists():
|
|
1143
|
+
try:
|
|
1144
|
+
goal_context = context_file.read_text()[:8000]
|
|
1145
|
+
except OSError:
|
|
1146
|
+
pass
|
|
1147
|
+
|
|
1148
|
+
run_setup(console) # first run: paste-your-key, then continue
|
|
1149
|
+
model = model or os.getenv("ROCKYCODE_MODEL")
|
|
1150
|
+
if not model:
|
|
1151
|
+
fail(console, "no model. pass --model or set ROCKYCODE_MODEL in .env.")
|
|
1152
|
+
raise typer.Exit(1)
|
|
1153
|
+
reviewer_model = reviewer_model or model
|
|
1154
|
+
currency = load_config(workdir)["currency"]
|
|
1155
|
+
if max_usd is None and max_hours is None and max_tokens is None:
|
|
1156
|
+
budget = recommended(currency)
|
|
1157
|
+
else:
|
|
1158
|
+
budget = GoalBudget(
|
|
1159
|
+
max_usd=max_usd,
|
|
1160
|
+
max_seconds=max_hours * 3600 if max_hours is not None else None,
|
|
1161
|
+
max_tokens=max_tokens,
|
|
1162
|
+
currency=currency,
|
|
1163
|
+
)
|
|
1164
|
+
|
|
1165
|
+
async def _run() -> None:
|
|
1166
|
+
from rockycode.engine.safety import network_intent, pre_scan
|
|
1167
|
+
|
|
1168
|
+
slug = _time.strftime("%Y%m%d-%H%M%S")
|
|
1169
|
+
ws = GoalWorkspace.create(workdir.resolve(), slug)
|
|
1170
|
+
info(console, f"isolated workspace: {ws.path}" + (f" · branch {ws.branch}" if ws.branch else " (copy)"))
|
|
1171
|
+
# Persistent per-run log (plan + every step + verify reasons) so the run is
|
|
1172
|
+
# reviewable after it scrolls away — the pointer is printed at the end and
|
|
1173
|
+
# surfaced when a chat hands off and resumes.
|
|
1174
|
+
log_dir = Path.home() / ".rockycode" / "goal-logs"
|
|
1175
|
+
log_dir.mkdir(parents=True, exist_ok=True)
|
|
1176
|
+
log_path = log_dir / f"{slug}.log"
|
|
1177
|
+
|
|
1178
|
+
def _log(msg: str) -> None:
|
|
1179
|
+
try:
|
|
1180
|
+
with log_path.open("a", encoding="utf-8") as f:
|
|
1181
|
+
f.write(msg + "\n")
|
|
1182
|
+
except OSError:
|
|
1183
|
+
pass
|
|
1184
|
+
_log(f"# goal: {objective}\n# {slug} workspace={ws.path} branch={ws.branch}")
|
|
1185
|
+
ledger = UsageLedger()
|
|
1186
|
+
for w in project_env_warnings(Path.cwd()):
|
|
1187
|
+
info(console, w)
|
|
1188
|
+
client = AsyncOpenAI(api_key=require_key(), base_url=require_base_url(),
|
|
1189
|
+
max_retries=5, timeout=300.0)
|
|
1190
|
+
# Plan BEFORE the sandbox exists (planning is LLM-only), so network/permits
|
|
1191
|
+
# are decided from the REAL plan, not a guess at your wording. Engine is
|
|
1192
|
+
# attached after the pre-flight decision.
|
|
1193
|
+
from rockycode.engine.explore import make_goal_verifier
|
|
1194
|
+
driver = EngineDriver(client=client, model=model, reviewer_model=reviewer_model,
|
|
1195
|
+
workspace=ws, ledger=ledger, currency=currency,
|
|
1196
|
+
verifier=make_goal_verifier(client=client, model=model,
|
|
1197
|
+
workdir=ws.path, ledger=ledger))
|
|
1198
|
+
|
|
1199
|
+
# Plan → derive permits → confirm, in a REFINE loop (all before the sandbox
|
|
1200
|
+
# exists, so it's cheap to iterate). At the gate: 'y' proceed, 'e' edit
|
|
1201
|
+
# (give guidance, re-plan, re-confirm), 'n' cancel. Proceeding grants the
|
|
1202
|
+
# plan's permits; --network/--no-network force the net setting; --yes auto.
|
|
1203
|
+
# Initial plan against the real files (folding in any chat-handoff context).
|
|
1204
|
+
plan_input = objective if not goal_context else (
|
|
1205
|
+
f"{objective}\n\n[Context from the chat that led here — use it to inform "
|
|
1206
|
+
f"the plan; the objective above is the goal]:\n{goal_context}")
|
|
1207
|
+
info(console, f"planning: {objective}")
|
|
1208
|
+
try:
|
|
1209
|
+
plan, requires = await driver.plan(plan_input)
|
|
1210
|
+
except Exception as e: # noqa: BLE001
|
|
1211
|
+
fail(console, f"planning failed — {e}"); ws.cleanup(keep=False); raise typer.Exit(1)
|
|
1212
|
+
if not plan:
|
|
1213
|
+
fail(console, "the planner produced no milestones."); ws.cleanup(keep=False); raise typer.Exit(1)
|
|
1214
|
+
|
|
1215
|
+
# Confirm loop: show plan → derive permits → gate. 'e' opens a real
|
|
1216
|
+
# back-and-forth — rocky ANSWERS your question, then shows the (revised or
|
|
1217
|
+
# same) plan — all before the sandbox exists, so iterating is cheap.
|
|
1218
|
+
while True:
|
|
1219
|
+
for i, m in enumerate(plan, 1):
|
|
1220
|
+
console.print(f" [dim]{i}.[/] {m}")
|
|
1221
|
+
|
|
1222
|
+
# derive needs FROM THE PLAN — the planner's REQUIRES line + a scan backstop
|
|
1223
|
+
scan_text = "\n".join(plan) + (("\n" + requires) if requires else "")
|
|
1224
|
+
flags = pre_scan(scan_text)
|
|
1225
|
+
blocked = [v for v in flags if v.action == "block"]
|
|
1226
|
+
if blocked:
|
|
1227
|
+
fail(console, f"plan names a blocked action: {blocked[0].reason}")
|
|
1228
|
+
ws.cleanup(keep=False); raise typer.Exit(1)
|
|
1229
|
+
asks = [v for v in flags if v.action == "ask"]
|
|
1230
|
+
net_reason = network_intent(requires) or network_intent(scan_text)
|
|
1231
|
+
use_network = network if network is not None else bool(net_reason)
|
|
1232
|
+
approved: set[str] = {v.pattern for v in asks}
|
|
1233
|
+
if use_network or asks or net_reason:
|
|
1234
|
+
info(console, "this run will need — approve before you leave:")
|
|
1235
|
+
if use_network:
|
|
1236
|
+
console.print(f" [magenta]🌐 network[/] — {net_reason or 'requested'}")
|
|
1237
|
+
elif net_reason:
|
|
1238
|
+
console.print(f" [dim]· plan implies network ({net_reason}) but --no-network is set — those steps will fail[/]")
|
|
1239
|
+
for v in asks:
|
|
1240
|
+
console.print(f" [yellow]⬆ {v.reason}[/]")
|
|
1241
|
+
else:
|
|
1242
|
+
info(console, "this run needs no extra permissions (offline, no privileged commands).")
|
|
1243
|
+
|
|
1244
|
+
if yes:
|
|
1245
|
+
break
|
|
1246
|
+
# Explicit verbs so 'n' can't be misread as "no, let's talk" — that's 'e'.
|
|
1247
|
+
GO, EDIT, CANCEL = ("y", "yes", ""), ("e", "edit", "discuss"), ("n", "no", "c", "cancel")
|
|
1248
|
+
valid = GO + EDIT + CANCEL
|
|
1249
|
+
action = "?"
|
|
1250
|
+
while action not in valid:
|
|
1251
|
+
action = typer.prompt(
|
|
1252
|
+
" run this plan? [y] yes (enter) · [e] discuss/edit · [n] cancel",
|
|
1253
|
+
default="y", show_default=False).strip().lower()
|
|
1254
|
+
if action not in valid:
|
|
1255
|
+
console.print(" [dim](y = run it · e = discuss/edit the plan · n = cancel)[/]")
|
|
1256
|
+
if action in GO:
|
|
1257
|
+
break
|
|
1258
|
+
if action in CANCEL:
|
|
1259
|
+
info(console, "cancelled — nothing ran. [dim](tip: 'e' discusses/edits the plan instead of cancelling)[/]")
|
|
1260
|
+
ws.cleanup(keep=False)
|
|
1261
|
+
raise typer.Exit(0)
|
|
1262
|
+
# edit → discuss: rocky answers, then re-shows the (possibly revised) plan
|
|
1263
|
+
msg = typer.prompt(" ask about or change the plan").strip()
|
|
1264
|
+
if msg:
|
|
1265
|
+
try:
|
|
1266
|
+
reply, plan, requires = await driver.discuss(objective, plan, requires, msg)
|
|
1267
|
+
except Exception as e: # noqa: BLE001
|
|
1268
|
+
reply = f"(couldn't reason about that — {type(e).__name__})"
|
|
1269
|
+
if reply:
|
|
1270
|
+
console.print(f"\n [#8b6fc9]rocky:[/] [italic]{reply}[/]\n")
|
|
1271
|
+
# loop → re-show the (possibly revised) plan + re-gate
|
|
1272
|
+
info(console, f"sandbox network: {'ON' if use_network else 'off (offline)'}")
|
|
1273
|
+
_log("plan:\n" + "\n".join(f" {i}. {m}" for i, m in enumerate(plan, 1)))
|
|
1274
|
+
|
|
1275
|
+
# 4) NOW provision the sandbox with that decision and attach the engine
|
|
1276
|
+
try:
|
|
1277
|
+
sandbox = await ChatSandbox.start(ws.path, network=use_network)
|
|
1278
|
+
except Exception as e: # noqa: BLE001
|
|
1279
|
+
fail(console, f"goal mode needs the sandbox (Docker) — {e}")
|
|
1280
|
+
ws.cleanup(keep=False)
|
|
1281
|
+
raise typer.Exit(1)
|
|
1282
|
+
reg = build_sandbox_registry(sandbox)
|
|
1283
|
+
reg["bash"] = safe_bash_tool(sandbox, approved) # approvals frozen up front
|
|
1284
|
+
engine = Engine(model=model, client=client, workdir=ws.path, registry=reg,
|
|
1285
|
+
trajectory_meta={"goal": objective, "runner": "goal"})
|
|
1286
|
+
driver.attach(engine, network=use_network)
|
|
1287
|
+
|
|
1288
|
+
# 5) run the loop with the pre-computed plan (tee events to console + log)
|
|
1289
|
+
def _on_event(m: str) -> None:
|
|
1290
|
+
console.print(f"[dim]· {m}[/]")
|
|
1291
|
+
_log(m)
|
|
1292
|
+
runner = GoalRunner(objective, driver, budget, ws, ledger, review_every=review_every,
|
|
1293
|
+
on_event=_on_event, preplanned=plan)
|
|
1294
|
+
info(console, f"budget — {budget.preflight_note()}")
|
|
1295
|
+
try:
|
|
1296
|
+
result = await runner.run()
|
|
1297
|
+
finally:
|
|
1298
|
+
await sandbox.stop()
|
|
1299
|
+
|
|
1300
|
+
sym = "¥" if currency == "cny" else "$"
|
|
1301
|
+
info(console, f"goal {result.status}: {result.reason}")
|
|
1302
|
+
info(console, f"{result.milestones_done}/{result.milestones_total} milestones · "
|
|
1303
|
+
f"spend {sym}{ledger.cost(currency):.4f}")
|
|
1304
|
+
if ws.branch:
|
|
1305
|
+
# Work is committed on the goal branch; the worktree dir is kept so
|
|
1306
|
+
# you can run/poke the files. Branch is never auto-deleted.
|
|
1307
|
+
info(console, f"review: git -C {ws.origin} diff {ws.base or 'HEAD'}..{ws.branch} (merge if good)")
|
|
1308
|
+
console.print(f" run it: cd {ws.path}")
|
|
1309
|
+
console.print(f" tidy up: git -C {ws.origin} worktree remove {ws.path} (keeps branch {ws.branch})")
|
|
1310
|
+
else:
|
|
1311
|
+
info(console, f"review the work in: {ws.path}")
|
|
1312
|
+
_log(f"result: {result.status} — {result.reason} "
|
|
1313
|
+
f"({result.milestones_done}/{result.milestones_total} milestones)")
|
|
1314
|
+
info(console, f"log: {log_path} [dim](full plan + per-step verify — check it anytime)[/]")
|
|
1315
|
+
if result_file is not None: # a /goal handoff reads this to resume chat
|
|
1316
|
+
import json as _json
|
|
1317
|
+
try:
|
|
1318
|
+
result_file.write_text(_json.dumps({
|
|
1319
|
+
"status": result.status, "reason": result.reason,
|
|
1320
|
+
"branch": ws.branch, "origin": str(ws.origin),
|
|
1321
|
+
"milestones_done": result.milestones_done,
|
|
1322
|
+
"milestones_total": result.milestones_total, "log": str(log_path),
|
|
1323
|
+
}))
|
|
1324
|
+
except OSError:
|
|
1325
|
+
pass
|
|
1326
|
+
|
|
1327
|
+
asyncio.run(_run())
|
|
1328
|
+
|
|
1329
|
+
|
|
1330
|
+
@app.command()
|
|
1331
|
+
def serve(
|
|
1332
|
+
model: Optional[str] = typer.Option(None, help="Model ID. Defaults to ROCKYCODE_MODEL env."),
|
|
1333
|
+
workdir: Optional[Path] = typer.Option(
|
|
1334
|
+
None, "--workdir", "-C", help="Project directory (default: cwd)."
|
|
1335
|
+
),
|
|
1336
|
+
thinking: bool = typer.Option(
|
|
1337
|
+
_env_bool("ROCKYCODE_THINKING", True),
|
|
1338
|
+
"--thinking/--no-thinking",
|
|
1339
|
+
),
|
|
1340
|
+
reasoning_effort: str = typer.Option(
|
|
1341
|
+
os.getenv("ROCKYCODE_REASONING_EFFORT", "max"),
|
|
1342
|
+
"--reasoning-effort",
|
|
1343
|
+
),
|
|
1344
|
+
max_tokens: int = typer.Option(
|
|
1345
|
+
_env_int("ROCKYCODE_MAX_TOKENS", 16384), "--max-tokens",
|
|
1346
|
+
),
|
|
1347
|
+
context_window: int = typer.Option(
|
|
1348
|
+
_env_int("ROCKYCODE_CONTEXT_WINDOW", 131072), "--context-window",
|
|
1349
|
+
),
|
|
1350
|
+
max_steps: int = typer.Option(
|
|
1351
|
+
_env_int("ROCKYCODE_CHAT_MAX_STEPS", 0), "--max-steps",
|
|
1352
|
+
),
|
|
1353
|
+
prompt: Optional[Path] = typer.Option(
|
|
1354
|
+
None, "--prompt",
|
|
1355
|
+
help="System prompt file (default: built-in ROCKY_SYSTEM).",
|
|
1356
|
+
),
|
|
1357
|
+
) -> None:
|
|
1358
|
+
"""Start Rocky as a JSON-RPC 2.0 server over stdin/stdout (for editor clients)."""
|
|
1359
|
+
import asyncio
|
|
1360
|
+
|
|
1361
|
+
from rockycode.engine.server import run_server
|
|
1362
|
+
|
|
1363
|
+
model = model or os.getenv("ROCKYCODE_MODEL")
|
|
1364
|
+
if not model:
|
|
1365
|
+
# stdout is the JSON-RPC channel — a Rich error there corrupts the very
|
|
1366
|
+
# first bytes the client reads. Startup errors go to stderr.
|
|
1367
|
+
fail(Console(stderr=True), "no model. pass --model or set ROCKYCODE_MODEL in .env.")
|
|
1368
|
+
raise typer.Exit(1)
|
|
1369
|
+
|
|
1370
|
+
workdir = (workdir or Path.cwd()).resolve()
|
|
1371
|
+
|
|
1372
|
+
# Load custom system prompt if provided
|
|
1373
|
+
system_prompt = os.getenv("ROCKYCODE_SYSTEM_PROMPT", "")
|
|
1374
|
+
if prompt:
|
|
1375
|
+
system_prompt = prompt.read_text()
|
|
1376
|
+
|
|
1377
|
+
asyncio.run(run_server(
|
|
1378
|
+
model=model, workdir=workdir,
|
|
1379
|
+
thinking=thinking, reasoning_effort=reasoning_effort,
|
|
1380
|
+
max_tokens=max_tokens, context_window=context_window,
|
|
1381
|
+
max_steps=max_steps, system_prompt=system_prompt,
|
|
1382
|
+
))
|
|
1383
|
+
|
|
1384
|
+
|
|
1385
|
+
if __name__ == "__main__":
|
|
1386
|
+
app()
|