rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,488 @@
1
+ Metadata-Version: 2.4
2
+ Name: rockycode
3
+ Version: 0.1.0
4
+ Summary: A coding agent harness, benchmarked on SWE-bench Verified. amaze!
5
+ Author: rockycode contributors
6
+ License: MIT
7
+ License-File: LICENSE
8
+ Requires-Python: >=3.11
9
+ Requires-Dist: aiohttp==3.13.5
10
+ Requires-Dist: beautifulsoup4==4.14.3
11
+ Requires-Dist: httpx==0.28.1
12
+ Requires-Dist: mcp==1.27.2
13
+ Requires-Dist: openai==2.36.0
14
+ Requires-Dist: pyflakes==3.2.0
15
+ Requires-Dist: python-dotenv==1.2.2
16
+ Requires-Dist: rich==15.0.0
17
+ Requires-Dist: sqlite-vec==0.1.9
18
+ Requires-Dist: textual==8.2.7
19
+ Requires-Dist: typer==0.25.1
20
+ Provides-Extra: bench
21
+ Requires-Dist: datasets==4.8.5; extra == 'bench'
22
+ Requires-Dist: docker==7.1.0; extra == 'bench'
23
+ Requires-Dist: swebench==4.1.0; extra == 'bench'
24
+ Provides-Extra: data-analysis
25
+ Requires-Dist: numpy>=1.26; extra == 'data-analysis'
26
+ Requires-Dist: pandas>=2.2; extra == 'data-analysis'
27
+ Provides-Extra: keyring
28
+ Requires-Dist: keyring>=24; extra == 'keyring'
29
+ Provides-Extra: pdf
30
+ Requires-Dist: pdfplumber>=0.11; extra == 'pdf'
31
+ Requires-Dist: pypdf>=4.0; extra == 'pdf'
32
+ Description-Content-Type: text/markdown
33
+
34
+ <div align="center">
35
+
36
+ <img src="brand/rockycode-wordmark.svg" alt="rockycode" width="420">
37
+
38
+ <br>
39
+
40
+ **A coding agent engine you can talk to**<br>
41
+ Built for the DeepSeek V4 series, with a unique research mode, bench-tested, and self-evolving features under development.
42
+
43
+ [English](README.md) · [简体中文](README.zh-CN.md)
44
+
45
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-~80%25_100--task_slice-7d5cc6)
46
+ ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
47
+ ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
48
+
49
+ </div>
50
+
51
+ ---
52
+
53
+ rockycode is a coding-agent harness, adapted for the DeepSeek V4 series and
54
+ able to run on any OpenAI-compatible endpoint. Leaning on DeepSeek V4's broad
55
+ world knowledge, it adds a search-augmented **Research mode**; it keeps the
56
+ core lean while exploring how to wire the harness onto Docker-based benchmarks,
57
+ so that every change to the framework produces a change in the score — a
58
+ number, not an impression; and it wraps a Docker sandbox around the risky
59
+ operations `goal` and `exec` modes might run. We also ship a batch of
60
+ still-rough **experimental features** exploring the harness's self-evolution —
61
+ and how the trajectories it produces feed better post-training, so model and
62
+ framework improve together.
63
+
64
+ **A single engine powers three entry points:**
65
+
66
+ | Entry point | Command | What it does |
67
+ |---|---|---|
68
+ | **Interactive agent** | `rockycode` | A terminal UI where Rocky reads, edits, and runs code in your project with native tool calls, streaming his reasoning as he works. |
69
+ | **Autonomous runner** | `rockycode goal` | Executes an objective unattended — on an isolated git-worktree **copy** of your repository, inside a Docker sandbox, under a hard budget cap, driving a plan → verify → review loop. The result is a git branch for you to review. |
70
+ | **Measurement rig** | `rockycode bench` | Drops the same agent loop onto SWE-bench Verified and scores it with the official harness, so every prompt or loop change is measured rather than felt. |
71
+
72
+ Every session — interactive or benchmarked — is logged as a training-ready
73
+ trajectory. The harness therefore doubles as an **RL environment**: the
74
+ groundwork for fine-tuning small models (tool use, compaction, memory roles)
75
+ against it.
76
+
77
+ ## Results — SWE-bench Verified
78
+
79
+ On SWE-bench Verified, a **randomly-chosen 100-task slice** (resources are
80
+ limited — no repeated runs averaged over the full 500):
81
+
82
+ **Result: ≈80% on a 100-task SWE-bench Verified slice** with `deepseek-v4-pro`,
83
+ no tuning against those tasks — the mean of a four-arm configuration sweep.
84
+
85
+ Read it with its limits. It's a representative **random 100-task slice**, not
86
+ the official 500-task Verified set: its harder 20-task core scored 60–70% while
87
+ the other 80 scored >80%. Treat it as an honest internal measurement, not a
88
+ leaderboard entry; for the full breakdown, follow our X
89
+ ([@rockycode_ai](https://x.com/rockycode_ai)).
90
+
91
+ We plan to add **DeepSWE-bench** support as well — currently in progress.
92
+
93
+ ## Getting started
94
+
95
+ Requirements: Python 3.11+ via [uv](https://docs.astral.sh/uv/) and an
96
+ OpenAI-compatible API key (DeepSeek is the default provider). That is the
97
+ entire install for chat and the research/learn modes — no Docker.
98
+
99
+ **Docker Desktop** is required only for the modes that isolate tool execution
100
+ in a container: `goal` (autonomous runs), `exec` (headless delegation),
101
+ `bench` (SWE-bench scoring), and the optional `/sandbox` in chat. Those modes
102
+ run offline in the sandbox by design, so a delegated or unattended task cannot
103
+ touch your host or reach the network.
104
+
105
+ ```bash
106
+ git clone https://github.com/cicialgo/rockycode.git && cd rockycode
107
+ uv tool install . # puts the `rockycode` command on your PATH
108
+ rockycode # the first run walks you through API-key setup
109
+ ```
110
+
111
+ On first launch you paste your API key once. It is stored in the OS keychain
112
+ (with the `[keyring]` extra) or in a private `0600` file at
113
+ `~/.rockycode/.env` — never in your project and never in your shell profile.
114
+ rockycode never reads a project `.env`: a cloned repository must not be able
115
+ to supply a key or redirect the endpoint, so credential-shaped variables in
116
+ one are warned about by name with their values left unread.
117
+
118
+ Run it in any project:
119
+
120
+ ```bash
121
+ rockycode # current directory
122
+ rockycode -C ~/code/myproject # any other project
123
+ rockycode -r # browse past sessions and pick one (or -r <id> for a specific one)
124
+ ```
125
+
126
+ Working on rockycode itself? `uv sync && uv run rockycode` runs straight from
127
+ the clone, no install.
128
+
129
+ ### Terminal setup
130
+
131
+ The TUI is fully mouse-driven: wheel-scroll the history, click file links,
132
+ dock a document beside the chat, and drag over any text to copy it (release =
133
+ copied, a toast confirms).
134
+
135
+ | Terminal | Setup |
136
+ |---|---|
137
+ | **ghostty** | None — mouse and clipboard work out of the box. |
138
+ | **iTerm2** | Enable **Settings → Profiles → Terminal → "Enable mouse reporting"**; without it, scrolling, clicks, and drag-copy never reach the app. `⌥`-drag keeps iTerm2's native selection. The clipboard needs no settings (copy goes through `pbcopy`). |
139
+ | **VS Code terminal** | None — mouse events are on by default. |
140
+
141
+ Over SSH the clipboard rides OSC 52 — enable "applications may access
142
+ clipboard" (or your terminal's equivalent) on the local end.
143
+
144
+ ## Interactive use
145
+
146
+ ### Slash commands
147
+
148
+ | Command | Description |
149
+ |---|---|
150
+ | `/help` | List all commands |
151
+ | `/plan [topic\|off]` | Plan mode: read-only exploration into a plan you approve, then build it here or hand it to `goal` |
152
+ | `/goal [objective]` | Go autonomous in its own screen (requires Docker) |
153
+ | `/research` | Research modes: deep-research · paper-reading · whiteboard · prove |
154
+ | `/learn` | Tutor mode — your understanding is the goal, not the diff |
155
+ | `/model` | Switch provider and model (see below) |
156
+ | `/effort off\|high\|xhigh\|max` | Reasoning depth, adjustable live per session |
157
+ | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
158
+ | `/sandbox on\|off\|status` | Isolate tool execution in a container |
159
+ | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
160
+ | `/artifact live on\|off` | Auto-refresh HTML artifacts in the browser |
161
+ | `/prompt` | Inspect the live system prompt |
162
+ | `/mcp` | Connected MCP servers and their tools |
163
+ | `/skills` | Installed skills |
164
+ | `/memory` · `/remember <note>` | Inspect memory · save a note |
165
+ | `/proposals` | Review skills drafted by the dream pass (approve or archive) |
166
+ | `/routines` | Run or lease recurring routines |
167
+ | `/config [key] [value]` | Show or set preferences |
168
+ | `/clear` · `/exit` | Session control |
169
+ | `! <cmd>` | Run a shell command directly; the output lands in Rocky's context |
170
+
171
+ ### Modes
172
+
173
+ - **Plan mode** (`/plan`) — the session becomes read-only except for one plan
174
+ file. Rocky explores the codebase, drafts a plan, and nothing is built until
175
+ you approve it — at which point you can execute it in-session or hand it to
176
+ goal mode.
177
+ - **Research modes** (`/research`) — a picker of prompt contracts:
178
+ **deep-research** (multi-source, fact-checked reports), **paper-reading**,
179
+ **whiteboard** (thinking out loud together), and **prove** — which turns an
180
+ informal mathematical claim into a Lean 4 compiler-certified verdict via the
181
+ built-in `lean-prover` skill (Mathlib and TorchLean). *(prove is
182
+ [experimental](#experimental).)*
183
+ - **Learn mode** (`/learn`) — a tutor posture: explanations and checks of your
184
+ understanding instead of code dumps.
185
+
186
+ ### Models and providers
187
+
188
+ DeepSeek is the home model, but providers are data, not code: each is a
189
+ base URL, a model list, a key variable, and a reasoning shape over the
190
+ OpenAI-compatible API.
191
+
192
+ | Provider | Models |
193
+ |---|---|
194
+ | **deepseek** (default) | `deepseek-v4-pro`, `deepseek-v4-flash` |
195
+ | **minimax** | `minimax-m3` |
196
+ | **kimi** | `kimi-k3` |
197
+ | **glm** | `glm-5.2` |
198
+
199
+ Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
200
+ and custom providers — including local vLLM/SGLang servers — go in
201
+ `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
202
+ keys are actually configured. Only DeepSeek is verified on the harness; the
203
+ others are [experimental](#experimental).
204
+
205
+ The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
206
+ provider maps it to its own reasoning tiers at the wire (DeepSeek, for
207
+ example, only distinguishes `high|max`, so `xhigh` clamps to `max`).
208
+
209
+ ## Autonomous use
210
+
211
+ ### Goal mode
212
+
213
+ `rockycode goal "<objective>"` runs Rocky unattended and hands you a git
214
+ branch to review. Safety is structural, not hopeful:
215
+
216
+ - The run happens on a git-worktree **copy** of your repository — nothing it
217
+ does touches your working tree.
218
+ - Tool execution is confined to the Docker sandbox, offline by design.
219
+ - Every bash command is screened by a classifier: destructive commands
220
+ (`rm -rf` of a root, `mkfs`, …) are refused outright; risky-but-legitimate
221
+ ones (`git push`, `sudo`, installs) require one up-front approval.
222
+ - A **budget cap** — spend, wallclock, and tokens, at real DeepSeek prices
223
+ including the peak-hour surcharge — stops the run gracefully, and the
224
+ worst-case spend is printed *before* the run starts.
225
+
226
+ The loop plans the objective into milestones, verifies each one with your own
227
+ linters (`check_code`), and a periodic reviewer re-plans to keep it on track.
228
+ Start small and cheap:
229
+
230
+ ```bash
231
+ rockycode goal "add a docstring to <fn> and run the linter" --max-usd 0.50 --max-hours 1
232
+ ```
233
+
234
+ ### Headless delegation: `exec`
235
+
236
+ `rockycode exec "<task>"` is the single-shot, non-interactive entry point,
237
+ designed to be called by *other* agents and scripts. Events stream as JSONL on
238
+ stdout; the Docker sandbox is **on by default** (the command classifier is
239
+ defense-in-depth, not the boundary); budgets are always enforced; and exit
240
+ codes distinguish success, failure, needs-approval, and budget-stop — so a
241
+ calling agent can grant an approval and resume instead of guessing.
242
+
243
+ ### Editor integration: `serve` and the VS Code extension
244
+
245
+ `rockycode serve` exposes the engine as JSON-RPC 2.0 over stdio, keeping it
246
+ UI-agnostic. The bundled VS Code extension (`rockycode-vscode/`) builds on it:
247
+ a sidebar chat with streaming reasoning, inline tool-approval cards, and diff
248
+ previews — with the API key kept in VS Code's encrypted Secret Storage.
249
+
250
+ ## Memory *(experimental)*
251
+
252
+ Rocky remembers across sessions in plain markdown files under
253
+ `.rockycode/memory/` (facts / skills / episodes / feedback) — the files are
254
+ the truth; edit them freely. `MEMORY.md` and user feedback load into every
255
+ session; everything else gets a one-line index and is fetched on demand via
256
+ the `recall_memory` tool — by exact name or by meaning. Semantic search runs
257
+ on local Ollama embeddings (`nomic-embed-text` for English,
258
+ `qwen3-embedding:0.6b` for Chinese and cross-lingual) over a self-rebuilding
259
+ sqlite-vec + FTS5 index; without Ollama it degrades cleanly to keyword search.
260
+ Removal archives, never deletes. Inspect from the shell with
261
+ `rockycode memory list|show|search|reindex|edit|rm`; disable with
262
+ `--no-memory`.
263
+
264
+ > ⚠️ In testing we found `/memory` lets remembered context *dominate* every
265
+ > session in a project — it can bend a perfectly normal task to fit old
266
+ > patterns. We don't recommend it for everyday use yet.
267
+
268
+ ## Experimental
269
+
270
+ These features work today but are early — opt-in, and their surface may still
271
+ change. Anything that could act on its own is **off by default**.
272
+
273
+ - **Dream** *(early, lightly tested).* `rockycode dream` consolidates recent
274
+ sessions while you rest: a local Ollama model (default `qwen3.5:2b`, zero API
275
+ tokens) digests each trajectory into an episode note, reconciles new facts
276
+ against old ones (contradictions archived, never deleted), rewrites the
277
+ dream-owned section of `MEMORY.md`, and re-embeds the index. `--dry-run`
278
+ previews every decision. Still early — not recommended for everyday use yet.
279
+ - **Self-improvement** *(default off).* On top of consolidation, the dream pass
280
+ judges each session into an outcome record, mines recurring failures into
281
+ weakness notes, and drafts candidate skills — and, from tasks you repeat by
282
+ hand, candidate **routines** — into a proposals inbox. Nothing self-installs:
283
+ you approve or archive via `/proposals`; an approved **routine** (`/routines`)
284
+ runs pre-approved and budgeted, on a bounded lease that expires back to
285
+ click-to-run. Enable it with `exit_sheet` / `dream` in config; it stays
286
+ invisible without a local Ollama stack regardless.
287
+ - **Formal proof** — `/research prove` and the built-in `lean-prover` skill turn
288
+ an informal math or model claim into a Lean 4 compiler-certified verdict
289
+ (green / amber / red), over Mathlib and TorchLean. The compiler is the judge,
290
+ so "proved" always means a real green build. Tested but still being polished —
291
+ and enabling it pulls a large Lean 4 toolchain download.
292
+ - **`explore` — read-only delegation.** Chat can buy a bounded, read-only
293
+ investigation from a fresh-context child that returns only a cited,
294
+ mechanically-verified report; the search noise never enters your session. It
295
+ also grounds goal mode's branch review and milestone verification.
296
+ - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
297
+ OpenAI-compatible profiles (`/model`), but only DeepSeek is verified on the
298
+ harness — treat the others as untested until they carry a bench number.
299
+
300
+ ## Works with your existing setup
301
+
302
+ Chat reads what other agents already use; there is no migration step:
303
+
304
+ - **MCP servers** from the project's `.mcp.json`, Claude Code's user config,
305
+ Claude Desktop's config, and Codex's `~/.codex/config.toml` (stdio servers;
306
+ first-defined name wins, project first). Their tools join Rocky's as
307
+ `mcp__<server>__<tool>`. Disable with `--no-mcp`.
308
+ - **Skills** from `.claude/skills/`, `.rockycode/skills/`,
309
+ `~/.claude/skills/` (SKILL.md folders) and `~/.codex/prompts/` (`*.md`).
310
+ Only name and description enter the context; the full instructions load on
311
+ demand via the `skill` tool. Disable with `--no-skills`.
312
+ - **Project instructions** in `CLAUDE.md` or `AGENTS.md`, folded into the
313
+ system prompt automatically.
314
+
315
+ None of this — memory included — loads in `bench`: published scores measure
316
+ the harness, not your plugins, and cross-task memory would contaminate
317
+ SWE-bench results.
318
+
319
+ ## Security model
320
+
321
+ rockycode assumes a repository you just cloned might be hostile:
322
+
323
+ - A project `.mcp.json` is **not** auto-started — a cloned repo cannot run
324
+ code or exfiltrate keys on launch (opt in with
325
+ `ROCKYCODE_TRUST_PROJECT_MCP=1`). MCP tool descriptions are scanned for
326
+ prompt injection.
327
+ - `read_file` refuses `.env`, credentials, and private keys; reads outside
328
+ the working directory require approval; a path jail applies regardless of
329
+ permission mode.
330
+ - Secrets are redacted from tool output before it reaches the model or the
331
+ trajectory log.
332
+ - An untrusted project config can *tighten* the tool-approval mode but never
333
+ weaken it.
334
+ - Permission modes (`yolo|ask|careful`) compose with per-command
335
+ classification: block-tier commands are refused even in yolo, and session
336
+ grants are scoped to a single binary.
337
+ - Goal mode adds worktree-copy isolation and the bash classifier on top;
338
+ `exec` keeps the sandbox on by default.
339
+
340
+ Reporting: see [SECURITY.md](SECURITY.md).
341
+
342
+ ## Benchmarking
343
+
344
+ Benchmarking runs from the clone and needs the `bench` extra (the SWE-bench
345
+ harness and the Docker SDK): `uv tool install '.[bench]'` — or
346
+ `uv sync --extra bench` and prefix commands with `uv run`.
347
+
348
+ ```bash
349
+ # raw single-shot baseline (Docker is only needed for scoring)
350
+ rockycode bench --runner raw --tasks dev10
351
+
352
+ # the harness: Rocky works inside each task's official SWE-bench container
353
+ rockycode bench --runner rockycode --tasks dev10
354
+
355
+ # fast sanity check
356
+ rockycode bench --runner rockycode --tasks dev10 --limit 1
357
+ ```
358
+
359
+ Useful flags: `--model`, `--limit`, `--skip-score`, `--run-id`,
360
+ `--thinking/--no-thinking`, `--reasoning-effort high|xhigh|max`,
361
+ `--max-tokens`, `--context-window` (compaction trigger point), `--max-steps`,
362
+ and `--prompt <file>` for system-prompt A/B runs (see `prompts/README.md`).
363
+
364
+ First runs are slow: the HF dataset downloads once (hundreds of MB), and each
365
+ task pulls its official image from Docker Hub (~1 GB each, cached forever
366
+ after). On Apple Silicon, enable *"Use Rosetta for x86_64/amd64 emulation"*
367
+ in Docker Desktop — the images are x86.
368
+
369
+ ## Architecture
370
+
371
+ - **Engine** (`rockycode/engine/`) — a model- and UI-agnostic ReAct loop. It
372
+ streams the provider with native tool calling, executes tools, and repeats
373
+ until the model answers without them, emitting a typed event stream — the
374
+ TUI, the bench console, the JSON-RPC server, and the trajectory logger are
375
+ all just subscribers. A configurable step cap (`--max-steps`) with budget
376
+ warnings near the end nudges the agent to commit to a fix instead of
377
+ exploring to exhaustion.
378
+ - **Compaction** (`engine/compaction.py`) — before every API call the engine
379
+ projects the next prompt size (the last real `prompt_tokens` plus
380
+ conservative estimates for newer messages). At 50% of the context window a
381
+ one-time nudge appears; at 90% it auto-compacts: first stubbing old tool
382
+ outputs (free and deterministic), then — if that is not enough — one API
383
+ call folds the older history into a dense state document and the context is
384
+ rebuilt as `[system, state, recent tail]`. Compactions are events and
385
+ trajectory records, so long tasks survive the window and the rewrite stays
386
+ visible in the training data.
387
+ - **Tools** — `bash`, `read_file`, `write_file`, `edit_file`, `grep`, `glob`,
388
+ and `check_code` (the project's own ruff/pyright, or a bundled pyflakes
389
+ fallback, for grounded lint and type feedback); plus `explore` (a read-only,
390
+ citation-verified sub-investigation), web tools, memory tools, and
391
+ goal-branch review tools. The same schemas serve chat and bench; only the
392
+ execution target differs (local directory vs. `docker exec` into the task
393
+ container). Read-only tool batches run concurrently; anything that writes
394
+ stays serial. Tool outputs are written to teach recovery: errors come back
395
+ as readable text, never exceptions.
396
+ - **Bench runner** — per task: pull the official image → start a container →
397
+ the agent works at `/testbed` → `git add -A && git diff --cached` is the
398
+ prediction → scored by `swebench.harness.run_evaluation`. Agent and scorer
399
+ share the same images.
400
+ - **Trajectories** — every session (chat and bench) appends to
401
+ `.rockycode/trajectories/*.jsonl`: metadata (model, prompt name and sha,
402
+ instance id), every message in OpenAI shape, per-call usage (including
403
+ DeepSeek cache hit/miss), and an outcome record. SFT/RL-ready by design.
404
+
405
+ ### Layout
406
+
407
+ ```
408
+ rockycode/
409
+ ├── rockycode/
410
+ │ ├── cli.py # subcommands: chat/exec/goal/bench/serve/dream/memory/config/pricing
411
+ │ ├── engine/
412
+ │ │ ├── loop.py # the ReAct loop (events out, history in)
413
+ │ │ ├── providers.py # provider registry (DeepSeek, MiniMax, Kimi, GLM, …)
414
+ │ │ ├── effort.py # the off/high/xhigh/max dial → provider tiers
415
+ │ │ ├── compaction.py # context compaction (prune → state summary)
416
+ │ │ ├── tools.py # tool schemas + local execution + path jail
417
+ │ │ ├── permission.py # yolo/ask/careful × risk tiers → allow/ask/block
418
+ │ │ ├── planmode.py # the read-only plan-mode gate
419
+ │ │ ├── modes.py # research/learn mode contracts
420
+ │ │ ├── goal.py # autonomous planner + runner
421
+ │ │ ├── headless.py # `exec`: sandboxed one-shot for other agents
422
+ │ │ ├── mcp.py # MCP client (stdio servers → extra tools)
423
+ │ │ ├── skills.py # skill discovery + progressive disclosure
424
+ │ │ ├── web.py # web_search / web_research / web_fetch
425
+ │ │ ├── container.py # docker-exec execution + patch extraction
426
+ │ │ ├── events.py # the event contract all UIs subscribe to
427
+ │ │ └── trajectory.py # training-ready session logs
428
+ │ ├── memory/ # files-as-truth memory store + semantic recall
429
+ │ ├── dream/ # consolidation, session judge, mining, proposals
430
+ │ ├── routines.py # recurring pre-approved work (leased auto-runs)
431
+ │ ├── modes/ # research/learn mode contracts (markdown)
432
+ │ ├── skills/ # built-in skills (lean-prover)
433
+ │ ├── tui/ # Textual chat app (rocky theme)
434
+ │ ├── runners/ # raw baseline · agent-on-SWE-bench · shared data
435
+ │ ├── prompts/rocky.py # built-in system + task prompts
436
+ │ └── score.py # wraps the official swebench eval
437
+ ├── rockycode-vscode/ # VS Code extension (chat panel over `rockycode serve`)
438
+ ├── prompts/ # prompt lab (A/B variants)
439
+ ├── bench/tasks/dev10.json # the fast iteration subset
440
+ └── tests/ # docker-free smoke tests (fake model streams)
441
+ ```
442
+
443
+ ## Prompt lab
444
+
445
+ System prompts are swappable files. Copy `prompts/rocky-v1.txt`, change one
446
+ thing, run dev10 with `--prompt`, and compare score, steps, and tokens. Names
447
+ and hashes are recorded everywhere, and per-variant prediction files never
448
+ clobber each other. Details in `prompts/README.md`.
449
+
450
+ ## Development
451
+
452
+ The smoke suite needs no API key and no Docker — a scripted fake model stream
453
+ drives the engine end to end:
454
+
455
+ ```bash
456
+ uv run python tests/run_all.py # the whole gate (what CI runs)
457
+ uv run python tests/run_all.py --all # include the Docker-dependent tests
458
+ uv run python tests/smoke_engine.py # or any single piece by name
459
+ ```
460
+
461
+ See [CONTRIBUTING.md](CONTRIBUTING.md) for conventions and
462
+ [CHANGELOG.md](CHANGELOG.md) for release history.
463
+
464
+ ## About the name
465
+
466
+ > *"i learn traditional physics. i no know e=mc^2 yet. but we fix bug. amaze!"*
467
+
468
+ - **rockycode** — Rocky from *Project Hail Mary*: enthusiastic, curious,
469
+ occasionally wrong, gets there anyway. He sings ♪♫ while he thinks.
470
+ - **dev10** — the fast 10-task iteration subset; real runs use full Verified.
471
+ - **amaze** — what Rocky says when the tests pass.
472
+
473
+ ## License
474
+
475
+ [MIT](LICENSE)
476
+
477
+ ## Contribution
478
+
479
+ Contributions are welcome — but the project is still early and has plenty of
480
+ rough edges, so we'd rather talk through feature and architecture design with
481
+ you before diving in. We're not trying to make it big-and-comprehensive: we
482
+ picture it as a modified N-1 starfighter — pushed to the limit in a few places,
483
+ and that's enough.
484
+
485
+ The initial commit comes from these developers:
486
+ [@cicialgo](https://github.com/cicialgo) — LLM algorithm engineer; overall design and adding the stranger experimental features.
487
+ [@dy2012](https://github.com/dy2012) — LLM engineer and architect; the permission, security, and Docker-based protections, the VS Code extension, coding-capability improvements, and many fixes.
488
+ [@codingmiu](https://github.com/codingmiu) — ML researcher; the standout research-mode designs.
@@ -0,0 +1,83 @@
1
+ rockycode/__init__.py,sha256=kUR5RAFc7HCeiqdlX36dZOHkUI5wI6V_43RpEcD8b-0,22
2
+ rockycode/banner.py,sha256=IQYuw2ky1ZJbOoqXcYF1vD1oi82ICl1CVIbAtAGTTvI,1331
3
+ rockycode/cli.py,sha256=freSFWDb06fC1zyBeFH1oLCNRVAc-xGx-VhNhPTO5Do,62704
4
+ rockycode/config.py,sha256=nwXojfkE-O1UmRlOH2O1bGMSiqmbCCgJSgNa3vzK-oU,7411
5
+ rockycode/onboarding.py,sha256=nlCiXRleUl6cVXinZJAaNG633EPB_Qpeb4uOZQIhtmo,13526
6
+ rockycode/palette.py,sha256=-P1QBdma9iNTgkvQvN85PIaxeNe2OPo_27HdAxMvuWg,751
7
+ rockycode/pricing.py,sha256=MDm3QBKgz5GN2neQCKyLp6EXRywP2fSAsp7XhpIv64g,7832
8
+ rockycode/routines.py,sha256=eNNP76Kob35FXVXvR6TCV-A1v29VlzBXxJoneMHy7nU,12662
9
+ rockycode/score.py,sha256=rYjPVDRSfyQOeLcdxdwz6Y57MKWSuNZ324xEoSnzUy0,4246
10
+ rockycode/session.py,sha256=A4GRb7-bMdTwzpMogwiKDUu27ftNKLv2biIFlABNhCk,10757
11
+ rockycode/dream/__init__.py,sha256=Echnrpg7bzUvxqQGSXGeX2OeJvT2beEaw1D6Zix8dgs,367
12
+ rockycode/dream/core.py,sha256=QgXSARqpIY-uQwJmfh1hNB5MohTZL4FgFwJ7ZYWGvH8,21659
13
+ rockycode/dream/judge.py,sha256=JA7zFWdwZegnpPXRReeVly1SgWVFyeKdtwWt9rjW4J8,5831
14
+ rockycode/dream/mining.py,sha256=BcwZIedieEZspYE1lLtqNYh3J2ObnrOcQm9SAXT7S6c,6478
15
+ rockycode/dream/proposals.py,sha256=jjFmx2FWoDV0dZraXg-gVPoZnqNxXatXvjDqQFmJWZ8,17253
16
+ rockycode/engine/__init__.py,sha256=TSXjK5f3K9t-hbWXP0uLF7oNYju4Ygr562S_fRu9XYo,420
17
+ rockycode/engine/artifact.py,sha256=pXjD_pCaO16Ex4cCThi8prlzBpgVtwL7LEO1WOwDUCc,15880
18
+ rockycode/engine/budget.py,sha256=nNhjHCWqkK1uNz2Det9Pargp1Idpqmos36tw2VpSyVY,3689
19
+ rockycode/engine/checks.py,sha256=AjSDNyE0laMxfhqpcAjUzEBtogCmkZVu3ad-h4IsePo,6817
20
+ rockycode/engine/compaction.py,sha256=pEeysOD0UpjhY2tOYtdX8MiYG6SiMwAlcMg5oCsTiZY,6897
21
+ rockycode/engine/container.py,sha256=pgcEBCbPCvhO7OnOjjqE81KSPlkfjjgPn2tNvZ-o7rw,8641
22
+ rockycode/engine/effort.py,sha256=4WNjyUwDmx1dCnV4KNigU7P2LDMs6ig2Ql-KzKxAeAE,2147
23
+ rockycode/engine/events.py,sha256=hrA8jFwy1RhZnhmisbCZSOO6Tj0iTbI1mZNDhTp-pPY,2205
24
+ rockycode/engine/explore.py,sha256=6xbtGNZXFNXhz3Bx1JSazRGBPFjaVIEJxLJlLHsw-2o,27311
25
+ rockycode/engine/goal.py,sha256=uPw0HpIvKOGd80jSceePHnHjKv-bnOJsCC9TI-46hrY,26476
26
+ rockycode/engine/goal_review.py,sha256=ZXp_6jJCwDIFQn8muYpUBTGczr0u5j2pTvvbT06m7c0,8646
27
+ rockycode/engine/goal_session.py,sha256=Bx6I0Oyh_xNQ8cQxoAOxj3WJm7oMzYoitdj26kpX2Uw,11370
28
+ rockycode/engine/headless.py,sha256=cwD3MC-BgDLIep1Y_7LOIx23ZyFg3S6LDwgW0MLnbxs,19200
29
+ rockycode/engine/loop.py,sha256=XRe9xzP79JGc6bdRjdUssFJn_z_kBH06vrNFAnBdA0o,35537
30
+ rockycode/engine/lsp.py,sha256=552u56sqWijq5j0u2dTsI9WODvw3YO3ATbZdu2Zsi1w,19333
31
+ rockycode/engine/mcp.py,sha256=46uR-GBlvv0kekdzYZBCdVoGRv8YCFQNgdbwiojkAOg,16084
32
+ rockycode/engine/modes.py,sha256=4-qPI8z2FmmBAaGkX6YTC1tu8B2xjCSCgIzU2JXuNQw,4724
33
+ rockycode/engine/outcome.py,sha256=evkRspeoG9se26HDSnhnBV8drsnK3dpycLUEKCHd4Nk,3648
34
+ rockycode/engine/permission.py,sha256=GYByHzqa9QmgUxCxd-I86J7vRFED1Q6viL7ttXpDK1M,8545
35
+ rockycode/engine/planmode.py,sha256=WM0YLHEZF_pEjpkeElxWRo0n4tD40qJZwQ_Y9R7WwNQ,10958
36
+ rockycode/engine/providers.py,sha256=bLcobfXpFnNk8I3X-TriM7myu_6tIizRjguiQxZZtGY,8117
37
+ rockycode/engine/redact.py,sha256=hKBuv1gOZJOkTNwjcEqbZ957mv55psIqY3Vg_MjpjQE,4342
38
+ rockycode/engine/safety.py,sha256=3sfcnBAUVzxtxyUnEqllMiJlYWEozYuE7KfVZmh-w7Q,6474
39
+ rockycode/engine/sandbox.py,sha256=UOy4VHQD6xXEy1bs2bHyTxTX3zCkLauanAjN4KX--UM,8952
40
+ rockycode/engine/server.py,sha256=qjx4xCJGmRugRoQ9Rm6DfSWBacKKe4KMXDf4CKbb9p0,17397
41
+ rockycode/engine/skills.py,sha256=hZxK9xyd40mxWcMLoWrszfhfG9wI1WAnY_eNyY_M6sI,6917
42
+ rockycode/engine/titler.py,sha256=azs572Qm-7yMfhk3mXpPinlwMD2R1ygz0EipDBIzovk,1725
43
+ rockycode/engine/tools.py,sha256=1xUIW9qeZ258JGEw2VhO5-vbl8-Q3VAwnLr9YORjsN0,19750
44
+ rockycode/engine/trajectory.py,sha256=hshsKY6KawJxCBtrc5jBKGuzyFXMh-_GD7oFXAF8488,5867
45
+ rockycode/engine/web.py,sha256=gfhGBlUt8Z8W8Uq-CAMk_7S0DOlIIMfhuOJC9uOaeVk,18326
46
+ rockycode/engine/worktree.py,sha256=0DORUWXKVQVhLWbNPnWX4GyMdudJbtC3nrPwnlfFpzM,6108
47
+ rockycode/memory/__init__.py,sha256=CU3jgm-RJz-Qqd9LU5hqW5eEByP7XRjRnVhZxvZEcn4,334
48
+ rockycode/memory/index.py,sha256=KtSphyyFAtABNgiHtWR8mMxSwfPlsm5_W29HOdzuzsA,10682
49
+ rockycode/memory/store.py,sha256=dOshMI0Vhw96WKTvg9D4FEG3UeB0OqRnL0S9gEfvpi8,12652
50
+ rockycode/modes/learn/learn.md,sha256=slEQbZz_mHXQ-3Z32TCrmcUU6I4lZa9eKBFcBjOlEEU,2036
51
+ rockycode/modes/research/deep-research.md,sha256=UFzLKV0imaOtpcr7J_Jrbq2fGsMI2r0FTncphNvL4PM,2391
52
+ rockycode/modes/research/paper-reading.md,sha256=EsR8eKj9qQ0FxjXGagDPsizY1GgIsG1iUOLPcZfk1_o,2081
53
+ rockycode/modes/research/prove.md,sha256=4J01eP7cc5HOlBfLpt8r0AnfWo9UMboGn9a9I42UzuQ,2925
54
+ rockycode/modes/research/whiteboard.md,sha256=b7XBpBuaDa7xXXGYeqqjZt_3gugQt7Px1n6e9KtOdPQ,3075
55
+ rockycode/prompts/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
56
+ rockycode/prompts/rocky.py,sha256=IKw3AveZh11mL8WSuXBEvw455HqToHZc3jwYx8-FXUc,11215
57
+ rockycode/runners/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
58
+ rockycode/runners/agent.py,sha256=noD_6GB3ex6D_ak949zT8sJUL3Y-dnxOUgQXaf3KNJM,9776
59
+ rockycode/runners/data.py,sha256=AjWGrDBCeJ7XbI79-kfr_CbmxaWj0HLPohEanJ6JMLM,2138
60
+ rockycode/runners/raw.py,sha256=rVgAvIOG2JNxi-YYAyYOt9eeT2m2UkAC5Z_ln54GoDY,6169
61
+ rockycode/skills/architecture-viz/SKILL.md,sha256=rM9qeveaW9RpL7mMUHN9gcLvmZxAuUseQPuRBaa0E3M,3936
62
+ rockycode/skills/architecture-viz/template.html,sha256=AHumvdAUMmYiyYSdIkGnh9iGsf5gELYX8zkRUB2pbEQ,8982
63
+ rockycode/skills/lean-prover/SKILL.md,sha256=mysN1MnpGYgiDUvabUQHdTWcj8ek3xxDUPRx27nbl08,8418
64
+ rockycode/skills/lean-prover/torchlean-api.md,sha256=ca9MXTAB5S98E1ITOyVrZ7HXMt9xKZ5wo4an4fDQM5o,3112
65
+ rockycode/tui/__init__.py,sha256=8nwemrzhEYp9Tg9yOM_ACu_1p7YJrVbq1izTzc6e9ig,71
66
+ rockycode/tui/app.py,sha256=QVQgRPTVlWyqSLmEjaaBOdPI3MfEbW4gTZeNWA169PE,117618
67
+ rockycode/tui/exitsheet.py,sha256=zThXu1X14H2MEUEGdmOK7u5NLTM6motqGOUv7tavmhA,6714
68
+ rockycode/tui/goal_screen.py,sha256=xD1AwKGsNTT0wPW1dKNQm8wrqjLA8XdCtnML5gUVlbk,14585
69
+ rockycode/tui/mdterm.py,sha256=aXiWkKMf4Pba3CghHabA_oLFT2tbF4PT0fKJbUA_5IU,9210
70
+ rockycode/tui/mdview.py,sha256=CdIcfRY_gp-NFXsNBmC9j9uCzFJF1gXfMJ-Efsgoxkg,3935
71
+ rockycode/tui/modepicker.py,sha256=qQlBwYjgXyI2LUL8WoHbMYsDTVII1ssgAgaUpI_cxYQ,3887
72
+ rockycode/tui/permission.py,sha256=fgcNhz-D2Hm9OB2uWpS9hQ3OJSfPu1L22gCWtfskuQg,6264
73
+ rockycode/tui/plangate.py,sha256=63bEdJJPXdpTq_j0JqmWh4jD5FSVGBi9thnFAbXj7lw,4139
74
+ rockycode/tui/prompt_history.py,sha256=Cv5PmDsyBZiha0TUy7QXJvBBCl0kExD97adfLzMij6Y,2747
75
+ rockycode/tui/proposalcard.py,sha256=YjDRPTFPaWV2s_f-y3QTj6kPeJr5YMZ5NMiTdtV8ReM,4719
76
+ rockycode/tui/resume.py,sha256=VTf8m791GJA_mC57ulSxZiJZjbNM64yly2OXBt87-do,5094
77
+ rockycode/tui/rocky_pet.py,sha256=8NLTVna_tWz6-pJMowJeoepf9m5cpc79UCqY9wgX6hc,3998
78
+ rockycode/tui/routinecard.py,sha256=5-dpfmsm-P7L6IQC5cTFO6FJtU7dk2zCLuPZYMH-ccs,4263
79
+ rockycode-0.1.0.dist-info/METADATA,sha256=P4FgmZiFbGbA1DcQiptLWXdRmRlzYu6Th0zlOSQBjog,24462
80
+ rockycode-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
81
+ rockycode-0.1.0.dist-info/entry_points.txt,sha256=uUAlfvOXOXEezJiRFienGClmIHwinRmgsHp8wqlsB2o,48
82
+ rockycode-0.1.0.dist-info/licenses/LICENSE,sha256=moz7aidSOsHgw4BV0xEmO-YQj_nKYyPxLtjfsOBUvEQ,1079
83
+ rockycode-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.31.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ rockycode = rockycode.cli:app
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 rockycode contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.