rockycode 0.1.0__tar.gz → 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (217) hide show
  1. {rockycode-0.1.0 → rockycode-0.1.1}/.github/workflows/ci.yml +3 -0
  2. rockycode-0.1.1/CHANGELOG.md +42 -0
  3. {rockycode-0.1.0 → rockycode-0.1.1}/CONTRIBUTING.md +1 -1
  4. {rockycode-0.1.0 → rockycode-0.1.1}/PKG-INFO +66 -21
  5. {rockycode-0.1.0 → rockycode-0.1.1}/README.md +65 -20
  6. {rockycode-0.1.0 → rockycode-0.1.1}/README.zh-CN.md +45 -12
  7. {rockycode-0.1.0 → rockycode-0.1.1}/pyproject.toml +19 -1
  8. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/cli.py +75 -7
  9. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/config.py +25 -2
  10. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/artifact.py +211 -27
  11. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/compaction.py +27 -1
  12. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/container.py +4 -1
  13. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/goal.py +6 -3
  14. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/goal_session.py +1 -1
  15. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/headless.py +10 -2
  16. rockycode-0.1.1/rockycode/engine/images.py +201 -0
  17. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/loop.py +42 -30
  18. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/providers.py +16 -3
  19. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/sandbox.py +98 -3
  20. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/server.py +58 -2
  21. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/tools.py +46 -0
  22. rockycode-0.1.1/rockycode/engine/vision.py +179 -0
  23. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/onboarding.py +2 -2
  24. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/routines.py +2 -2
  25. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/session.py +81 -1
  26. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/app.py +479 -67
  27. rockycode-0.1.1/rockycode/tui/clipboard.py +129 -0
  28. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/goal_screen.py +4 -4
  29. rockycode-0.1.1/rockycode/tui/modelpicker.py +116 -0
  30. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/resume.py +1 -2
  31. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/media/chat.html +13 -0
  32. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/package-lock.json +2 -2
  33. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/package.json +34 -1
  34. rockycode-0.1.1/rockycode-vscode/src/artifactTree.ts +168 -0
  35. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/extension.ts +40 -1
  36. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/protocol.ts +28 -0
  37. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/rockyConnection.ts +33 -3
  38. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/rockyProvider.ts +8 -2
  39. {rockycode-0.1.0 → rockycode-0.1.1}/tests/run_all.py +7 -6
  40. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_artifact.py +163 -1
  41. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_checks.py +0 -1
  42. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_container.py +5 -1
  43. rockycode-0.1.1/tests/smoke_docker_sandbox_cancel.py +63 -0
  44. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_engine.py +1 -1
  45. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_exec.py +97 -0
  46. rockycode-0.1.1/tests/smoke_hardkill_resume.py +134 -0
  47. rockycode-0.1.1/tests/smoke_images.py +249 -0
  48. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_mdview.py +17 -1
  49. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_routines.py +1 -2
  50. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_serve.py +17 -4
  51. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_artifact_modal.py +25 -3
  52. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_model.py +22 -11
  53. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_proposals.py +1 -2
  54. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_read_grant.py +0 -1
  55. {rockycode-0.1.0 → rockycode-0.1.1}/uv.lock +34 -1
  56. rockycode-0.1.0/CHANGELOG.md +0 -20
  57. {rockycode-0.1.0 → rockycode-0.1.1}/.dockerignore +0 -0
  58. {rockycode-0.1.0 → rockycode-0.1.1}/.github/workflows/release.yml +0 -0
  59. {rockycode-0.1.0 → rockycode-0.1.1}/.gitignore +0 -0
  60. {rockycode-0.1.0 → rockycode-0.1.1}/Dockerfile +0 -0
  61. {rockycode-0.1.0 → rockycode-0.1.1}/Dockerfile.sandbox +0 -0
  62. {rockycode-0.1.0 → rockycode-0.1.1}/LICENSE +0 -0
  63. {rockycode-0.1.0 → rockycode-0.1.1}/SECURITY.md +0 -0
  64. {rockycode-0.1.0 → rockycode-0.1.1}/bench/tasks/dev10.json +0 -0
  65. {rockycode-0.1.0 → rockycode-0.1.1}/bench/tasks/test20.json +0 -0
  66. {rockycode-0.1.0 → rockycode-0.1.1}/brand/rockycode-note.svg +0 -0
  67. {rockycode-0.1.0 → rockycode-0.1.1}/brand/rockycode-wordmark.svg +0 -0
  68. {rockycode-0.1.0 → rockycode-0.1.1}/docker-compose.yml +0 -0
  69. {rockycode-0.1.0 → rockycode-0.1.1}/prompts/README.md +0 -0
  70. {rockycode-0.1.0 → rockycode-0.1.1}/prompts/rocky-v1.txt +0 -0
  71. {rockycode-0.1.0 → rockycode-0.1.1}/prompts/rocky-v2-search-first.txt +0 -0
  72. {rockycode-0.1.0 → rockycode-0.1.1}/prompts/rocky-v3-decisive.txt +0 -0
  73. {rockycode-0.1.0 → rockycode-0.1.1}/prompts/rocky-zh-closer.txt +0 -0
  74. {rockycode-0.1.0 → rockycode-0.1.1}/prompts/rocky-zh-full-closer.txt +0 -0
  75. {rockycode-0.1.0 → rockycode-0.1.1}/prompts/rocky-zh-full.txt +0 -0
  76. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/__init__.py +0 -0
  77. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/banner.py +0 -0
  78. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/dream/__init__.py +0 -0
  79. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/dream/core.py +0 -0
  80. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/dream/judge.py +0 -0
  81. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/dream/mining.py +0 -0
  82. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/dream/proposals.py +0 -0
  83. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/__init__.py +0 -0
  84. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/budget.py +0 -0
  85. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/checks.py +0 -0
  86. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/effort.py +0 -0
  87. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/events.py +0 -0
  88. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/explore.py +0 -0
  89. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/goal_review.py +0 -0
  90. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/lsp.py +0 -0
  91. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/mcp.py +0 -0
  92. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/modes.py +0 -0
  93. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/outcome.py +0 -0
  94. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/permission.py +0 -0
  95. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/planmode.py +0 -0
  96. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/redact.py +0 -0
  97. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/safety.py +0 -0
  98. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/skills.py +0 -0
  99. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/titler.py +0 -0
  100. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/trajectory.py +0 -0
  101. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/web.py +0 -0
  102. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/engine/worktree.py +0 -0
  103. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/memory/__init__.py +0 -0
  104. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/memory/index.py +0 -0
  105. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/memory/store.py +0 -0
  106. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/modes/learn/learn.md +0 -0
  107. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/modes/research/deep-research.md +0 -0
  108. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/modes/research/paper-reading.md +0 -0
  109. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/modes/research/prove.md +0 -0
  110. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/modes/research/whiteboard.md +0 -0
  111. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/palette.py +0 -0
  112. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/pricing.py +0 -0
  113. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/prompts/__init__.py +0 -0
  114. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/prompts/rocky.py +0 -0
  115. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/runners/__init__.py +0 -0
  116. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/runners/agent.py +0 -0
  117. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/runners/data.py +0 -0
  118. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/runners/raw.py +0 -0
  119. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/score.py +0 -0
  120. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/skills/architecture-viz/SKILL.md +0 -0
  121. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/skills/architecture-viz/template.html +0 -0
  122. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/skills/lean-prover/SKILL.md +0 -0
  123. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/skills/lean-prover/torchlean-api.md +0 -0
  124. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/__init__.py +0 -0
  125. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/exitsheet.py +0 -0
  126. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/mdterm.py +0 -0
  127. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/mdview.py +0 -0
  128. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/modepicker.py +0 -0
  129. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/permission.py +0 -0
  130. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/plangate.py +0 -0
  131. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/prompt_history.py +0 -0
  132. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/proposalcard.py +0 -0
  133. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/rocky_pet.py +0 -0
  134. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode/tui/routinecard.py +0 -0
  135. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/.gitignore +0 -0
  136. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/.vscode/launch.json +0 -0
  137. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/.vscode/tasks.json +0 -0
  138. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/.vscodeignore +0 -0
  139. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/CHANGELOG.md +0 -0
  140. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/LICENSE +0 -0
  141. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/README.md +0 -0
  142. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/esbuild.config.mjs +0 -0
  143. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/media/icon-marketplace.svg +0 -0
  144. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/media/icon.png +0 -0
  145. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/media/marked.js +0 -0
  146. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/media/rocky-icon.svg +0 -0
  147. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/diffManager.ts +0 -0
  148. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/editorContext.ts +0 -0
  149. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/permissionManager.ts +0 -0
  150. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/statusBar.ts +0 -0
  151. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/src/webview-highlight.ts +0 -0
  152. {rockycode-0.1.0 → rockycode-0.1.1}/rockycode-vscode/tsconfig.json +0 -0
  153. {rockycode-0.1.0 → rockycode-0.1.1}/tests/fake_lsp_server.py +0 -0
  154. {rockycode-0.1.0 → rockycode-0.1.1}/tests/fake_mcp_server.py +0 -0
  155. {rockycode-0.1.0 → rockycode-0.1.1}/tests/invariants.py +0 -0
  156. {rockycode-0.1.0 → rockycode-0.1.1}/tests/real_planmode.py +0 -0
  157. {rockycode-0.1.0 → rockycode-0.1.1}/tests/real_reasoning_roundtrip.py +0 -0
  158. {rockycode-0.1.0 → rockycode-0.1.1}/tests/real_tool_contract.py +0 -0
  159. {rockycode-0.1.0 → rockycode-0.1.1}/tests/run_real.py +0 -0
  160. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_approver.py +0 -0
  161. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_bash.py +0 -0
  162. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_bash_grant.py +0 -0
  163. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_bilang_prompt.py +0 -0
  164. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_budget.py +0 -0
  165. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_compaction.py +0 -0
  166. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_credentials.py +0 -0
  167. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_dream.py +0 -0
  168. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_dream_trigger.py +0 -0
  169. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_effort.py +0 -0
  170. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_explore.py +0 -0
  171. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_goal.py +0 -0
  172. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_goal_driver.py +0 -0
  173. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_goal_review.py +0 -0
  174. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_goal_runner.py +0 -0
  175. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_interrupt.py +0 -0
  176. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_judge.py +0 -0
  177. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_lsp.py +0 -0
  178. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_mcp.py +0 -0
  179. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_mdterm.py +0 -0
  180. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_memory.py +0 -0
  181. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_memory_index.py +0 -0
  182. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_mining.py +0 -0
  183. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_modes.py +0 -0
  184. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_onboarding.py +0 -0
  185. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_outcome.py +0 -0
  186. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_parallel.py +0 -0
  187. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_permission.py +0 -0
  188. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_planmode.py +0 -0
  189. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_pricing.py +0 -0
  190. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_prompt_history.py +0 -0
  191. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_proposals.py +0 -0
  192. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_providers.py +0 -0
  193. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_redact.py +0 -0
  194. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_resume_handoff.py +0 -0
  195. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_routine_run.py +0 -0
  196. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_safety.py +0 -0
  197. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_score.py +0 -0
  198. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_session.py +0 -0
  199. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_skills.py +0 -0
  200. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_submit_race.py +0 -0
  201. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_today.py +0 -0
  202. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tools.py +0 -0
  203. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tools_jail.py +0 -0
  204. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_trajectory.py +0 -0
  205. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_bash_gate.py +0 -0
  206. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_copy.py +0 -0
  207. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_envwarn.py +0 -0
  208. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_exitsheet.py +0 -0
  209. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_goal_screen.py +0 -0
  210. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_input_nav.py +0 -0
  211. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_permission.py +0 -0
  212. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_plan.py +0 -0
  213. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_routines.py +0 -0
  214. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_scroll.py +0 -0
  215. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_shell.py +0 -0
  216. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_tui_toggle.py +0 -0
  217. {rockycode-0.1.0 → rockycode-0.1.1}/tests/smoke_web.py +0 -0
@@ -23,5 +23,8 @@ jobs:
23
23
  - name: Sync deps from the lockfile
24
24
  run: uv sync --frozen
25
25
 
26
+ - name: Ruff (conservative correctness gate)
27
+ run: uv run ruff check .
28
+
26
29
  - name: Smoke tests (the CORE gate)
27
30
  run: uv run python tests/run_all.py
@@ -0,0 +1,42 @@
1
+ # Changelog
2
+
3
+ All notable changes to rockycode are recorded here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/), and the project follows
5
+ [Semantic Versioning](https://semver.org/) — pre-1.0, so the surface may still
6
+ change between minor versions.
7
+
8
+ ## [Unreleased]
9
+
10
+ _Nothing yet._
11
+
12
+ ## [0.1.1] — session artifacts, images in chat, real sandbox cancel
13
+
14
+ ### Added
15
+ - Images in chat: paste (`ctrl+v` / `/paste`) or drag an image in. Vision models
16
+ see it raw; no-vision models route through a provider sidecar, your own image
17
+ CLI, or a `view_image` tool — picked once, remembered. Known CLIs set up with
18
+ one word (`/config image_cli mmx` — rocky knows the invocation).
19
+ - Session artifact inventory: `/artifact list · open <n> · stop · live on|off`,
20
+ a footer badge with open-tab counts, and an Artifacts tree in the VS Code
21
+ extension fed live by `rockycode serve`.
22
+ - Bare `/model` opens a live provider + model picker.
23
+
24
+ ### Fixed
25
+ - Live artifacts no longer drop and reconnect every 30 s; the artifact server
26
+ stops/restarts cleanly and rebinds saved live pages to the new port.
27
+ - Sandbox cancel/timeout kills the in-container process group, not just the
28
+ host-side docker client; images without `python3` fall back to plain `bash -c`.
29
+ - Resume self-heals after a hard kill; closing the doc dock no longer wedges the TUI.
30
+
31
+ ### Internal
32
+ - CI: conservative ruff correctness gate (pinned `0.16.1`) ahead of the smoke suite.
33
+
34
+ ## [0.1.0] — first public release
35
+
36
+ Initial public release. One repo, one engine, three ways to use it — interactive
37
+ `chat`, autonomous `goal`, and the `bench` measurement rig — running on DeepSeek
38
+ or any OpenAI-compatible model.
39
+
40
+ Some capabilities ship as **experimental and default-off** (self-improvement,
41
+ `prove` / `lean-prover`, `explore`, and providers other than DeepSeek); see the
42
+ README's Experimental section for what they are and how to enable them.
@@ -40,7 +40,7 @@ rockycode goal "…" # autonomous run (sandboxed worktree copy)
40
40
  ## Before you open a PR
41
41
 
42
42
  1. **Run the test suite** — `python tests/run_all.py`. It should be green.
43
- 2. **Lint** — `ruff check .` (config is in `pyproject.toml`).
43
+ 2. **Lint** — `uv run ruff check .` (pinned version and conservative rules are in `pyproject.toml`).
44
44
  3. **Match the surrounding code.** Read the file you're editing and mirror its
45
45
  naming, comment density, and idioms. New code should be indistinguishable in
46
46
  style from what's around it.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rockycode
3
- Version: 0.1.0
3
+ Version: 0.1.1
4
4
  Summary: A coding agent harness, benchmarked on SWE-bench Verified. amaze!
5
5
  Author: rockycode contributors
6
6
  License: MIT
@@ -42,7 +42,8 @@ Built for the DeepSeek V4 series, with a unique research mode, bench-tested, and
42
42
 
43
43
  [English](README.md) · [简体中文](README.zh-CN.md)
44
44
 
45
- ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-~80%25_100--task_slice-7d5cc6)
45
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-79.8%25_V4--flash-7d5cc6)
46
+ ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-81.8%25_pass@3-8d6cd0)
46
47
  ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
47
48
  ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
48
49
 
@@ -76,17 +77,30 @@ against it.
76
77
 
77
78
  ## Results — SWE-bench Verified
78
79
 
79
- On SWE-bench Verified, a **randomly-chosen 100-task slice** (resources are
80
- limited — no repeated runs averaged over the full 500):
81
-
82
- **Result: ≈80% on a 100-task SWE-bench Verified slice** with `deepseek-v4-pro`,
83
- no tuning against those tasks — the mean of a four-arm configuration sweep.
84
-
85
- Read it with its limits. It's a representative **random 100-task slice**, not
86
- the official 500-task Verified set: its harder 20-task core scored 60–70% while
87
- the other 80 scored >80%. Treat it as an honest internal measurement, not a
88
- leaderboard entry; for the full breakdown, follow our X
89
- ([@rockycode_ai](https://x.com/rockycode_ai)).
80
+ Full-set numbers: **independent full-500 runs** (three each for
81
+ `deepseek-v4-pro` and `minimax-m3`, one so far for `deepseek-v4-flash`), same
82
+ harness and config for all (100-step cap, 32,768 max output tokens, reasoning
83
+ effort `max`, thinking on), scored with the official SWE-bench harness. No
84
+ tuning against the tasks.
85
+
86
+ | run | `deepseek-v4-pro` | `deepseek-v4-flash` | `minimax-m3` |
87
+ |---|---|---|---|
88
+ | round 1 | 75.6% (378/500) | 79.8% (399/500) | 72.8% (364/500) |
89
+ | round 2 | 74.8% (374/500) | — | 71.2% (356/500) |
90
+ | round 3 | 74.4% (372/500) | — | 70.0% (350/500) |
91
+ | **average** | **74.9%** | **79.8%** *(1 run)* | **71.3%** |
92
+ | union of runs (pass@3) | 81.8% (409/500) | — | 83.6% (418/500) |
93
+
94
+ Read the two summary rows differently. The **average** is the
95
+ leaderboard-comparable number — each round is an independent single-pass run
96
+ over the full 500. The **union** is pass@3: tasks solved by at least one
97
+ round. The gap between them (~7 points for DeepSeek, ~12 for MiniMax) is
98
+ run-to-run variance, not capability — the models already reach these tasks
99
+ under this harness, they just don't hold them every run. Closing that gap
100
+ (verify-before-finish gating and run selection, not more prompting) is the
101
+ current line of work. For reference, DeepSeek reports 80.6% with its own
102
+ scaffold; the three-run union crosses that mark. Per-round breakdowns:
103
+ [@rockycode_ai](https://x.com/rockycode_ai).
90
104
 
91
105
  We plan to add **DeepSWE-bench** support as well — currently in progress.
92
106
 
@@ -102,10 +116,38 @@ in a container: `goal` (autonomous runs), `exec` (headless delegation),
102
116
  run offline in the sandbox by design, so a delegated or unattended task cannot
103
117
  touch your host or reach the network.
104
118
 
119
+ ```bash
120
+ uv tool install rockycode # recommended — puts the `rockycode` command on your PATH
121
+ rockycode # the first run walks you through API-key setup
122
+ ```
123
+
124
+ Don't have uv yet? One command installs it:
125
+ `curl -LsSf https://astral.sh/uv/install.sh | sh` — Windows and other options
126
+ in the [uv install docs](https://docs.astral.sh/uv/getting-started/installation/).
127
+
128
+ Three ways to install — they look similar but land in different places:
129
+
130
+ - **`uv tool install rockycode`** (recommended) — gives the CLI its own
131
+ isolated environment and puts `rockycode` on your PATH; if your system
132
+ Python is older than 3.11, uv fetches a matching interpreter by itself.
133
+ The "install it like an app" path.
134
+ - **`uv pip install rockycode`** — installs into the **currently active
135
+ virtual environment** only: the `rockycode` command exists inside that
136
+ venv, so a new shell won't find it unless the venv is active (or run it
137
+ as `uv run rockycode`).
138
+ - **`pip install rockycode`** — same venv caveat as above, and it needs
139
+ Python 3.11+. On an older Python it fails with the misleading
140
+ `ERROR: No matching distribution found for rockycode`. Why: pip only
141
+ offers releases whose `requires-python` matches your interpreter, so on an
142
+ old Python it sees no installable version at all and reports that as a
143
+ missing package. If you hit this, don't fight it — use
144
+ `uv tool install rockycode` above; uv brings its own Python 3.11+.
145
+
146
+ Or install from source:
147
+
105
148
  ```bash
106
149
  git clone https://github.com/cicialgo/rockycode.git && cd rockycode
107
- uv tool install . # puts the `rockycode` command on your PATH
108
- rockycode # the first run walks you through API-key setup
150
+ uv tool install .
109
151
  ```
110
152
 
111
153
  On first launch you paste your API key once. It is stored in the OS keychain
@@ -157,7 +199,8 @@ clipboard" (or your terminal's equivalent) on the local end.
157
199
  | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
158
200
  | `/sandbox on\|off\|status` | Isolate tool execution in a container |
159
201
  | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
160
- | `/artifact live on\|off` | Auto-refresh HTML artifacts in the browser |
202
+ | `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
203
+ | `/paste` | Attach a clipboard image (or `ctrl+v`); no-vision models pick a route |
161
204
  | `/prompt` | Inspect the live system prompt |
162
205
  | `/mcp` | Connected MCP servers and their tools |
163
206
  | `/skills` | Installed skills |
@@ -191,7 +234,7 @@ OpenAI-compatible API.
191
234
 
192
235
  | Provider | Models |
193
236
  |---|---|
194
- | **deepseek** (default) | `deepseek-v4-pro`, `deepseek-v4-flash` |
237
+ | **deepseek** (default) | `deepseek-v4-flash` (default), `deepseek-v4-pro` (preview) |
195
238
  | **minimax** | `minimax-m3` |
196
239
  | **kimi** | `kimi-k3` |
197
240
  | **glm** | `glm-5.2` |
@@ -199,8 +242,9 @@ OpenAI-compatible API.
199
242
  Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
200
243
  and custom providers — including local vLLM/SGLang servers — go in
201
244
  `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
202
- keys are actually configured. Only DeepSeek is verified on the harness; the
203
- others are [experimental](#experimental).
245
+ keys are actually configured. DeepSeek and MiniMax both carry full-500 bench
246
+ numbers (see [Results](#results--swe-bench-verified)); Kimi and GLM are
247
+ [experimental](#experimental).
204
248
 
205
249
  The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
206
250
  provider maps it to its own reasoning tiers at the wire (DeepSeek, for
@@ -294,8 +338,9 @@ change. Anything that could act on its own is **off by default**.
294
338
  mechanically-verified report; the search noise never enters your session. It
295
339
  also grounds goal mode's branch review and milestone verification.
296
340
  - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
297
- OpenAI-compatible profiles (`/model`), but only DeepSeek is verified on the
298
- harness — treat the others as untested until they carry a bench number.
341
+ OpenAI-compatible profiles (`/model`). DeepSeek and MiniMax carry full
342
+ bench numbers (see Results); treat GLM and Kimi as untested until they do
343
+ too.
299
344
 
300
345
  ## Works with your existing setup
301
346
 
@@ -9,7 +9,8 @@ Built for the DeepSeek V4 series, with a unique research mode, bench-tested, and
9
9
 
10
10
  [English](README.md) · [简体中文](README.zh-CN.md)
11
11
 
12
- ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-~80%25_100--task_slice-7d5cc6)
12
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-79.8%25_V4--flash-7d5cc6)
13
+ ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-81.8%25_pass@3-8d6cd0)
13
14
  ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
14
15
  ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
15
16
 
@@ -43,17 +44,30 @@ against it.
43
44
 
44
45
  ## Results — SWE-bench Verified
45
46
 
46
- On SWE-bench Verified, a **randomly-chosen 100-task slice** (resources are
47
- limited — no repeated runs averaged over the full 500):
48
-
49
- **Result: ≈80% on a 100-task SWE-bench Verified slice** with `deepseek-v4-pro`,
50
- no tuning against those tasks — the mean of a four-arm configuration sweep.
51
-
52
- Read it with its limits. It's a representative **random 100-task slice**, not
53
- the official 500-task Verified set: its harder 20-task core scored 60–70% while
54
- the other 80 scored >80%. Treat it as an honest internal measurement, not a
55
- leaderboard entry; for the full breakdown, follow our X
56
- ([@rockycode_ai](https://x.com/rockycode_ai)).
47
+ Full-set numbers: **independent full-500 runs** (three each for
48
+ `deepseek-v4-pro` and `minimax-m3`, one so far for `deepseek-v4-flash`), same
49
+ harness and config for all (100-step cap, 32,768 max output tokens, reasoning
50
+ effort `max`, thinking on), scored with the official SWE-bench harness. No
51
+ tuning against the tasks.
52
+
53
+ | run | `deepseek-v4-pro` | `deepseek-v4-flash` | `minimax-m3` |
54
+ |---|---|---|---|
55
+ | round 1 | 75.6% (378/500) | 79.8% (399/500) | 72.8% (364/500) |
56
+ | round 2 | 74.8% (374/500) | — | 71.2% (356/500) |
57
+ | round 3 | 74.4% (372/500) | — | 70.0% (350/500) |
58
+ | **average** | **74.9%** | **79.8%** *(1 run)* | **71.3%** |
59
+ | union of runs (pass@3) | 81.8% (409/500) | — | 83.6% (418/500) |
60
+
61
+ Read the two summary rows differently. The **average** is the
62
+ leaderboard-comparable number — each round is an independent single-pass run
63
+ over the full 500. The **union** is pass@3: tasks solved by at least one
64
+ round. The gap between them (~7 points for DeepSeek, ~12 for MiniMax) is
65
+ run-to-run variance, not capability — the models already reach these tasks
66
+ under this harness, they just don't hold them every run. Closing that gap
67
+ (verify-before-finish gating and run selection, not more prompting) is the
68
+ current line of work. For reference, DeepSeek reports 80.6% with its own
69
+ scaffold; the three-run union crosses that mark. Per-round breakdowns:
70
+ [@rockycode_ai](https://x.com/rockycode_ai).
57
71
 
58
72
  We plan to add **DeepSWE-bench** support as well — currently in progress.
59
73
 
@@ -69,10 +83,38 @@ in a container: `goal` (autonomous runs), `exec` (headless delegation),
69
83
  run offline in the sandbox by design, so a delegated or unattended task cannot
70
84
  touch your host or reach the network.
71
85
 
86
+ ```bash
87
+ uv tool install rockycode # recommended — puts the `rockycode` command on your PATH
88
+ rockycode # the first run walks you through API-key setup
89
+ ```
90
+
91
+ Don't have uv yet? One command installs it:
92
+ `curl -LsSf https://astral.sh/uv/install.sh | sh` — Windows and other options
93
+ in the [uv install docs](https://docs.astral.sh/uv/getting-started/installation/).
94
+
95
+ Three ways to install — they look similar but land in different places:
96
+
97
+ - **`uv tool install rockycode`** (recommended) — gives the CLI its own
98
+ isolated environment and puts `rockycode` on your PATH; if your system
99
+ Python is older than 3.11, uv fetches a matching interpreter by itself.
100
+ The "install it like an app" path.
101
+ - **`uv pip install rockycode`** — installs into the **currently active
102
+ virtual environment** only: the `rockycode` command exists inside that
103
+ venv, so a new shell won't find it unless the venv is active (or run it
104
+ as `uv run rockycode`).
105
+ - **`pip install rockycode`** — same venv caveat as above, and it needs
106
+ Python 3.11+. On an older Python it fails with the misleading
107
+ `ERROR: No matching distribution found for rockycode`. Why: pip only
108
+ offers releases whose `requires-python` matches your interpreter, so on an
109
+ old Python it sees no installable version at all and reports that as a
110
+ missing package. If you hit this, don't fight it — use
111
+ `uv tool install rockycode` above; uv brings its own Python 3.11+.
112
+
113
+ Or install from source:
114
+
72
115
  ```bash
73
116
  git clone https://github.com/cicialgo/rockycode.git && cd rockycode
74
- uv tool install . # puts the `rockycode` command on your PATH
75
- rockycode # the first run walks you through API-key setup
117
+ uv tool install .
76
118
  ```
77
119
 
78
120
  On first launch you paste your API key once. It is stored in the OS keychain
@@ -124,7 +166,8 @@ clipboard" (or your terminal's equivalent) on the local end.
124
166
  | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
125
167
  | `/sandbox on\|off\|status` | Isolate tool execution in a container |
126
168
  | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
127
- | `/artifact live on\|off` | Auto-refresh HTML artifacts in the browser |
169
+ | `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
170
+ | `/paste` | Attach a clipboard image (or `ctrl+v`); no-vision models pick a route |
128
171
  | `/prompt` | Inspect the live system prompt |
129
172
  | `/mcp` | Connected MCP servers and their tools |
130
173
  | `/skills` | Installed skills |
@@ -158,7 +201,7 @@ OpenAI-compatible API.
158
201
 
159
202
  | Provider | Models |
160
203
  |---|---|
161
- | **deepseek** (default) | `deepseek-v4-pro`, `deepseek-v4-flash` |
204
+ | **deepseek** (default) | `deepseek-v4-flash` (default), `deepseek-v4-pro` (preview) |
162
205
  | **minimax** | `minimax-m3` |
163
206
  | **kimi** | `kimi-k3` |
164
207
  | **glm** | `glm-5.2` |
@@ -166,8 +209,9 @@ OpenAI-compatible API.
166
209
  Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
167
210
  and custom providers — including local vLLM/SGLang servers — go in
168
211
  `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
169
- keys are actually configured. Only DeepSeek is verified on the harness; the
170
- others are [experimental](#experimental).
212
+ keys are actually configured. DeepSeek and MiniMax both carry full-500 bench
213
+ numbers (see [Results](#results--swe-bench-verified)); Kimi and GLM are
214
+ [experimental](#experimental).
171
215
 
172
216
  The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
173
217
  provider maps it to its own reasoning tiers at the wire (DeepSeek, for
@@ -261,8 +305,9 @@ change. Anything that could act on its own is **off by default**.
261
305
  mechanically-verified report; the search noise never enters your session. It
262
306
  also grounds goal mode's branch review and milestone verification.
263
307
  - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
264
- OpenAI-compatible profiles (`/model`), but only DeepSeek is verified on the
265
- harness — treat the others as untested until they carry a bench number.
308
+ OpenAI-compatible profiles (`/model`). DeepSeek and MiniMax carry full
309
+ bench numbers (see Results); treat GLM and Kimi as untested until they do
310
+ too.
266
311
 
267
312
  ## Works with your existing setup
268
313
 
@@ -9,7 +9,8 @@
9
9
 
10
10
  [English](README.md) · [简体中文](README.zh-CN.md)
11
11
 
12
- ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-~80%25_100--task_slice-7d5cc6)
12
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-79.8%25_V4--flash-7d5cc6)
13
+ ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-81.8%25_pass@3-8d6cd0)
13
14
  ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
14
15
  ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
15
16
 
@@ -36,11 +37,17 @@ rockycode 是一个编程智能体 harness,为 DeepSeek V4 系列适配,也
36
37
 
37
38
  ## 能力量化:SWE bench
38
39
 
39
- 在 SWE-bench Verified 上,**随机抽选 100 题**测试(资源所限,没有在完整 500 题上做重复取平均):
40
+ 完整 500 题的成绩:**独立的完整 500 题运行**(`deepseek-v4-pro` 与 `minimax-m3` 各 3 轮,`deepseek-v4-flash` 目前 1 轮),所有模型使用完全相同的 harness 与配置(步数上限 100、单次最大输出 32,768 token、推理力度 `max`、thinking 开启),由官方 SWE-bench harness 打分,未针对任务做任何调优。
40
41
 
41
- **结果:在这个 100 题切片上约 80%** —— 用 `deepseek-v4-pro`,未针对这些任务做调优,是一次四臂配置扫描的均值。
42
+ | 轮次 | `deepseek-v4-pro` | `deepseek-v4-flash` | `minimax-m3` |
43
+ |---|---|---|---|
44
+ | 第 1 轮 | 75.6%(378/500) | 79.8%(399/500) | 72.8%(364/500) |
45
+ | 第 2 轮 | 74.8%(374/500) | — | 71.2%(356/500) |
46
+ | 第 3 轮 | 74.4%(372/500) | — | 70.0%(350/500) |
47
+ | **平均** | **74.9%** | **79.8%**(1 轮) | **71.3%** |
48
+ | 多轮并集(pass@3) | 81.8%(409/500) | — | 83.6%(418/500) |
42
49
 
43
- 请连同它的边界一起看:这是一个**随机抽取的代表性切片**,不是官方 500 题的完整Verified 集;其中较难的 20 题核心得分 60–70%,另外 80 题得分 >80%。请把它当作诚实的内部测量,而非排行榜成绩;完整拆解请关注我们的X账号(@rockycode_ai)。
50
+ 两行汇总要分开读。**平均值**是可以与排行榜对比的数字 —— 每一轮都是独立的单次完整 500 题。**并集**是 pass@3:至少被某一轮解出的任务。两者之间的差距(DeepSeek 约 7 个点、MiniMax 约 12 个点)是轮次间方差,不是能力上限 —— 模型在这套 harness 下已经"够得着"这些任务,只是无法每一轮都稳住。当前的工作重心就是收掉这个差距:finish 前的验证门控与多轮选择,而不是继续改提示词。作为参照:DeepSeek 用自家 scaffold 报告 80.6%,三轮并集已越过这个数字。逐轮拆解请关注我们的X账号(@rockycode_ai)。
44
51
 
45
52
  我们计划支持DeepSWE bench,目前还在调试中。
46
53
 
@@ -50,10 +57,35 @@ rockycode 是一个编程智能体 harness,为 DeepSeek V4 系列适配,也
50
57
 
51
58
  **Docker Desktop** 仅在需要容器隔离工具执行的模式下才必需:`goal`(自主运行)、`exec`(自动化委托)、`bench`(SWE-bench 打分),以及 chat 里可选的`/sandbox`。这些模式在沙箱内默认离线运行,被委托或无人值守的任务因此无法触碰你的主机、无法访问网络。
52
59
 
60
+ ```bash
61
+ uv tool install rockycode # 推荐 —— `rockycode` 命令直接进 PATH
62
+ rockycode # 首次运行会引导你完成 API key 设置
63
+ ```
64
+
65
+ 还没有 uv?一条命令安装:
66
+ `curl -LsSf https://astral.sh/uv/install.sh | sh`(Windows 及其他方式见
67
+ [uv 安装文档](https://docs.astral.sh/uv/getting-started/installation/))。
68
+
69
+ 三种装法看着像,落点完全不同:
70
+
71
+ - **`uv tool install rockycode`**(推荐)—— 给 CLI 一个独立的隔离环境,并把
72
+ `rockycode` 命令放上 PATH;系统 Python 低于 3.11 时,uv 会自动拉取一个匹配
73
+ 的解释器。想「当应用装」就用它。
74
+ - **`uv pip install rockycode`** —— 只装进**当前激活的虚拟环境**:`rockycode`
75
+ 命令只存在于那个 venv 里,新开一个 shell 会找不到它(要么先激活 venv,
76
+ 要么用 `uv run rockycode` 运行)。
77
+ - **`pip install rockycode`** —— 同样只进当前环境,且要求 Python 3.11+。在更老
78
+ 的 Python 上会报一个很有误导性的
79
+ `ERROR: No matching distribution found for rockycode`。原因:pip 只会提供
80
+ `requires-python` 与你解释器匹配的版本,老 Python 下它一个可装的版本都看
81
+ 不到,于是把「版本不满足」报成了「包不存在」。遇到这个错不用纠结 —— 直接用
82
+ 上面的 `uv tool install rockycode`,uv 自带 3.11+ 的 Python。
83
+
84
+ 或从源码安装:
85
+
53
86
  ```bash
54
87
  git clone https://github.com/cicialgo/rockycode.git && cd rockycode
55
- uv tool install . # 把 `rockycode` 命令装到 PATH
56
- rockycode # 首次运行会引导你完成 API key 设置
88
+ uv tool install .
57
89
  ```
58
90
 
59
91
  首次启动只需粘贴一次 API key。它被存入操作系统钥匙串(安装 `[keyring]`扩展时)或 `~/.rockycode/.env` 私有文件(权限 `0600`)—— 不进你的项目或者shell 配置。rockycode 从不读取项目内的 `.env`:克隆来的仓库不应有能力注入 key 或改写 endpoint,因此其中形似凭据的变量只会按名字给出警告,值不会被读取。
@@ -98,7 +130,8 @@ SSH 远程会话下剪贴板走 OSC 52 —— 在本地端开启「允许应用
98
130
  | `/permission yolo\|ask\|careful` | 本次会话的工具审批严格度 |
99
131
  | `/sandbox on\|off\|status` | 把工具执行隔离进容器 |
100
132
  | `/lsp` | 语言服务器状态;诊断信息随 `read_file` 一并返回 |
101
- | `/artifact live on\|off` | 浏览器中自动刷新 HTML artifact |
133
+ | `/artifact` | 本会话的 artifact:`list` · `open <n>` · `stop` · `live on\|off` |
134
+ | `/paste` | 粘贴剪贴板图片(或 `ctrl+v`);无视觉模型自选识图路由 |
102
135
  | `/prompt` | 查看当前生效的系统提示词 |
103
136
  | `/mcp` | 已连接的 MCP 服务及其工具 |
104
137
  | `/skills` | 已安装的技能 |
@@ -128,15 +161,15 @@ DeepSeek 是主场模型,但提供商是数据而非代码:每个提供商
128
161
 
129
162
  | 提供商 | 模型 |
130
163
  |---|---|
131
- | **deepseek**(默认) | `deepseek-v4-pro`、`deepseek-v4-flash` |
164
+ | **deepseek**(默认) | `deepseek-v4-flash`(默认)、`deepseek-v4-pro`(preview) |
132
165
  | **minimax** | `minimax-m3` |
133
166
  | **kimi** | `kimi-k3` |
134
167
  | **glm** | `glm-5.2` |
135
168
 
136
169
  区域端点写作 `<提供商>-<区域>`(如 `kimi-cn`);自定义提供商 —— 包括本地
137
170
  vLLM/SGLang 服务 —— 写进 `~/.rockycode/providers.toml`。`/model` 选择器
138
- 只展示已配置好 key 的提供商。只有 DeepSeek 在 harness 上验证过,其余为
139
- [实验性功能](#实验性功能)。
171
+ 只展示已配置好 key 的提供商。DeepSeek 与 MiniMax 都有完整 500 题的 bench
172
+ 成绩(见上方「能力量化」);Kimi 与 GLM 属于[实验性功能](#实验性功能)。
140
173
 
141
174
  推理深度旋钮(`/effort off|high|xhigh|max`)与提供商无关;各提供商在请求层
142
175
  把它映射到自己的档位(例如 DeepSeek 只区分 `high|max`,`xhigh` 会收敛为
@@ -220,8 +253,8 @@ sqlite-vec + FTS5 索引;没有 Ollama 则平滑退化为关键词检索。删
220
253
  有界的只读调查,只拿回带引用、经机械校验的报告;搜索噪声绝不进入你的会话。
221
254
  它同样为 goal 模式的分支评审与里程碑验证提供依据。
222
255
  - **DeepSeek 以外的提供商。** MiniMax、GLM / z.ai、Kimi 都以 OpenAI 兼容的
223
- profile 接入(`/model`),但只有 DeepSeek 在 harness 上验证过 —— 其余在拿到
224
- bench 分数前,请当作未验证。
256
+ profile 接入(`/model`)。DeepSeek 与 MiniMax 已有完整 bench 成绩(见
257
+ 「能力量化」);GLM 与 Kimi 在拿到 bench 分数前,请当作未验证。
225
258
 
226
259
  ## 复用你已有的配置
227
260
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "rockycode"
3
- version = "0.1.0"
3
+ version = "0.1.1"
4
4
  description = "A coding agent harness, benchmarked on SWE-bench Verified. amaze!"
5
5
  requires-python = ">=3.11"
6
6
  readme = "README.md"
@@ -68,6 +68,13 @@ pdf = [
68
68
  # uv pip install 'rockycode[keyring]'
69
69
  keyring = ["keyring>=24"]
70
70
 
71
+ [dependency-groups]
72
+ # Development-only quality gate. Pin it so contributors and CI enforce the
73
+ # exact same rules; Ruff is not shipped in the rockycode runtime wheel.
74
+ dev = [
75
+ "ruff==0.16.1",
76
+ ]
77
+
71
78
  [project.scripts]
72
79
  rockycode = "rockycode.cli:app"
73
80
 
@@ -81,3 +88,14 @@ packages = ["rockycode"]
81
88
  [tool.ruff]
82
89
  line-length = 110
83
90
  target-version = "py311"
91
+
92
+ [tool.ruff.lint]
93
+ # Conservative correctness baseline: import/syntax errors, undefined names,
94
+ # unused imports/variables, and a small set of unambiguous PEP 8 errors.
95
+ select = ["E4", "E7", "E9", "F"]
96
+
97
+ [tool.ruff.lint.per-file-ignores]
98
+ # Smoke tests are executable scripts: many intentionally set cwd/env before
99
+ # importing rockycode, and some use compact procedural assertions. Keep F/E9
100
+ # correctness checks active without forcing a wholesale test rewrite.
101
+ "tests/*.py" = ["E402", "E702", "E741"]
@@ -423,7 +423,6 @@ def chat(
423
423
 
424
424
  # LSP — config resolved here; connection is started async in the TUI
425
425
  # (same two-phase pattern as MCP: construct here, start in on_mount).
426
- lsp_mgr = None
427
426
  engine.lsp_enabled = lsp
428
427
  engine.lsp_manager = None
429
428
  if lsp:
@@ -570,7 +569,7 @@ def _print_exit_card(engine) -> None:
570
569
  f"“{escape(s.display_title)}” · 📁 {folder} · {s.n_messages} msgs"
571
570
  )
572
571
  console.print(f" resume it: [cyan]rockycode --resume {sid}[/]")
573
- console.print(f" or browse: [cyan]rockycode --resume[/]")
572
+ console.print(" or browse: [cyan]rockycode --resume[/]")
574
573
 
575
574
 
576
575
  # exec's local error exit — mirrors headless.EXIT_ERROR without importing the
@@ -593,6 +592,12 @@ def exec_cmd(
593
592
  help="Extra directory write/edit may touch beyond the workdir (repeatable).",
594
593
  ),
595
594
  model: Optional[str] = typer.Option(None, help="Model ID. Defaults to ROCKYCODE_MODEL env."),
595
+ image: Optional[List[Path]] = typer.Option(
596
+ None, "--image",
597
+ help="Image file to attach to the task (repeatable). Requires a "
598
+ "vision-capable model/endpoint (stepfun · minimax-m3 · kimi-k3); "
599
+ "sent as a base64 data URL, the one form every provider accepts.",
600
+ ),
596
601
  max_steps: int = typer.Option(
597
602
  30, "--max-steps",
598
603
  help="Tool-step budget (must be > 0 — headless runs are never unbounded). "
@@ -680,6 +685,20 @@ def exec_cmd(
680
685
  roots.append(rp)
681
686
  allowed_roots = tuple(roots)
682
687
 
688
+ images: list[Path] = []
689
+ if image:
690
+ from rockycode.engine.images import IMAGE_EXTS
691
+ for ip in image:
692
+ p = ip.expanduser().resolve()
693
+ if not p.is_file():
694
+ fail(err, f"--image '{ip}' is not a file.")
695
+ raise typer.Exit(EXIT_CODE_ERROR)
696
+ if p.suffix.lower() not in IMAGE_EXTS:
697
+ fail(err, f"--image '{ip}': unsupported format. "
698
+ f"use {' / '.join(sorted(IMAGE_EXTS))}.")
699
+ raise typer.Exit(EXIT_CODE_ERROR)
700
+ images.append(p)
701
+
683
702
  # Project identity: exec sessions land in the same global trajectory store
684
703
  # and resume picker as chat sessions — the receipt must be resumable.
685
704
  from rockycode.session import get_project
@@ -690,7 +709,8 @@ def exec_cmd(
690
709
  from rockycode.engine.headless import run_exec
691
710
 
692
711
  code = asyncio.run(run_exec(
693
- prompt=prompt, model=model, workdir=wd, allowed_roots=allowed_roots,
712
+ prompt=prompt, model=model, workdir=wd, images=images or None,
713
+ allowed_roots=allowed_roots,
694
714
  max_steps=max_steps, originator=originator,
695
715
  include_thinking=include_thinking, output_last_message=output_last_message,
696
716
  sandbox=sandbox, network=network, err=err,
@@ -945,8 +965,13 @@ def memory_edit(name: str, workdir: Optional[Path] = _WORKDIR_OPT) -> None:
945
965
  subprocess.run([editor, str(mem.path)])
946
966
 
947
967
 
948
- @app.command()
968
+ bench_app = typer.Typer()
969
+ app.add_typer(bench_app, name="bench")
970
+
971
+
972
+ @bench_app.callback(invoke_without_command=True)
949
973
  def bench(
974
+ ctx: typer.Context,
950
975
  runner: str = typer.Option("raw", help="'raw' (single-shot) or 'rockycode' (harness, v1+)."),
951
976
  tasks: str = typer.Option("dev10", help="'dev10', 'verified', or path to a JSON list of instance IDs."),
952
977
  model: Optional[str] = typer.Option(None, help="Model ID. Defaults to ROCKYCODE_MODEL env."),
@@ -992,6 +1017,8 @@ def bench(
992
1017
  ),
993
1018
  ) -> None:
994
1019
  """Run rockycode against a SWE-bench task set and report the score."""
1020
+ if ctx.invoked_subcommand:
1021
+ return # a subcommand (`bench score`) runs instead of a bench run
995
1022
  show_banner(console)
996
1023
 
997
1024
  model = model or os.getenv("ROCKYCODE_MODEL")
@@ -1069,6 +1096,42 @@ def bench(
1069
1096
  score(predictions_path=predictions_path, instance_ids=None, run_id=run_id, console=console)
1070
1097
 
1071
1098
 
1099
+ @bench_app.command("score")
1100
+ def bench_score(
1101
+ predictions: Path = typer.Argument(
1102
+ ..., help="Predictions JSONL from a bench run (results/predictions/…)."
1103
+ ),
1104
+ run_id: Optional[str] = typer.Option(
1105
+ None,
1106
+ help="Eval run id. Defaults to the predictions filename, so re-running "
1107
+ "the SAME command resumes a crashed eval (already-scored instances "
1108
+ "are skipped by the swebench harness).",
1109
+ ),
1110
+ max_workers: int = typer.Option(4, help="Parallel eval containers."),
1111
+ timeout: int = typer.Option(1800, help="Per-instance eval timeout in seconds."),
1112
+ ) -> None:
1113
+ """Score an existing predictions file — or resume a crashed scoring run.
1114
+
1115
+ The agent phase and the scoring phase are separable: if scoring dies
1116
+ (network, docker), the predictions file still holds the full run. This
1117
+ re-runs ONLY the eval, with a run_id stable across retries by default.
1118
+ """
1119
+ show_banner(console)
1120
+ if not predictions.exists():
1121
+ fail(console, f"predictions file not found: {predictions}")
1122
+ raise typer.Exit(1)
1123
+ _docker_preflight()
1124
+ from rockycode.score import score
1125
+ score(
1126
+ predictions_path=predictions,
1127
+ run_id=run_id or predictions.stem,
1128
+ instance_ids=None,
1129
+ console=console,
1130
+ max_workers=max_workers,
1131
+ timeout=timeout,
1132
+ )
1133
+
1134
+
1072
1135
  @app.command()
1073
1136
  def goal(
1074
1137
  objective: Optional[str] = typer.Argument(None, help="What to accomplish autonomously (omit with --clean)."),
@@ -1208,9 +1271,13 @@ def goal(
1208
1271
  try:
1209
1272
  plan, requires = await driver.plan(plan_input)
1210
1273
  except Exception as e: # noqa: BLE001
1211
- fail(console, f"planning failed — {e}"); ws.cleanup(keep=False); raise typer.Exit(1)
1274
+ fail(console, f"planning failed — {e}")
1275
+ ws.cleanup(keep=False)
1276
+ raise typer.Exit(1)
1212
1277
  if not plan:
1213
- fail(console, "the planner produced no milestones."); ws.cleanup(keep=False); raise typer.Exit(1)
1278
+ fail(console, "the planner produced no milestones.")
1279
+ ws.cleanup(keep=False)
1280
+ raise typer.Exit(1)
1214
1281
 
1215
1282
  # Confirm loop: show plan → derive permits → gate. 'e' opens a real
1216
1283
  # back-and-forth — rocky ANSWERS your question, then shows the (revised or
@@ -1225,7 +1292,8 @@ def goal(
1225
1292
  blocked = [v for v in flags if v.action == "block"]
1226
1293
  if blocked:
1227
1294
  fail(console, f"plan names a blocked action: {blocked[0].reason}")
1228
- ws.cleanup(keep=False); raise typer.Exit(1)
1295
+ ws.cleanup(keep=False)
1296
+ raise typer.Exit(1)
1229
1297
  asks = [v for v in flags if v.action == "ask"]
1230
1298
  net_reason = network_intent(requires) or network_intent(scan_text)
1231
1299
  use_network = network if network is not None else bool(net_reason)