rockycode 0.1.0__tar.gz → 0.1.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (219) hide show
  1. {rockycode-0.1.0 → rockycode-0.1.2}/.github/workflows/ci.yml +3 -0
  2. rockycode-0.1.2/.github/workflows/release.yml +27 -0
  3. rockycode-0.1.2/CHANGELOG.md +61 -0
  4. {rockycode-0.1.0 → rockycode-0.1.2}/CONTRIBUTING.md +1 -1
  5. {rockycode-0.1.0 → rockycode-0.1.2}/PKG-INFO +77 -22
  6. {rockycode-0.1.0 → rockycode-0.1.2}/README.md +75 -20
  7. {rockycode-0.1.0 → rockycode-0.1.2}/README.zh-CN.md +50 -12
  8. {rockycode-0.1.0 → rockycode-0.1.2}/pyproject.toml +19 -1
  9. rockycode-0.1.2/rockycode/__init__.py +11 -0
  10. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/cli.py +88 -9
  11. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/config.py +25 -2
  12. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/artifact.py +211 -27
  13. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/compaction.py +27 -1
  14. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/container.py +4 -1
  15. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/goal.py +6 -3
  16. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/goal_session.py +1 -1
  17. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/headless.py +10 -2
  18. rockycode-0.1.2/rockycode/engine/images.py +201 -0
  19. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/loop.py +42 -30
  20. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/providers.py +16 -3
  21. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/sandbox.py +98 -3
  22. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/server.py +59 -2
  23. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/tools.py +46 -0
  24. rockycode-0.1.2/rockycode/engine/vision.py +179 -0
  25. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/onboarding.py +2 -2
  26. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/pricing.py +16 -14
  27. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/routines.py +2 -2
  28. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/session.py +81 -1
  29. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/app.py +479 -67
  30. rockycode-0.1.2/rockycode/tui/clipboard.py +129 -0
  31. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/goal_screen.py +4 -4
  32. rockycode-0.1.2/rockycode/tui/modelpicker.py +116 -0
  33. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/resume.py +1 -2
  34. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/media/chat.html +13 -0
  35. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/package-lock.json +2 -2
  36. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/package.json +34 -1
  37. rockycode-0.1.2/rockycode-vscode/src/artifactTree.ts +168 -0
  38. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/extension.ts +40 -1
  39. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/protocol.ts +28 -0
  40. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/rockyConnection.ts +33 -3
  41. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/rockyProvider.ts +8 -2
  42. {rockycode-0.1.0 → rockycode-0.1.2}/tests/run_all.py +7 -6
  43. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_artifact.py +163 -1
  44. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_checks.py +0 -1
  45. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_container.py +5 -1
  46. rockycode-0.1.2/tests/smoke_docker_sandbox_cancel.py +63 -0
  47. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_engine.py +1 -1
  48. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_exec.py +97 -0
  49. rockycode-0.1.2/tests/smoke_hardkill_resume.py +134 -0
  50. rockycode-0.1.2/tests/smoke_images.py +249 -0
  51. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_mdview.py +17 -1
  52. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_pricing.py +9 -9
  53. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_routines.py +1 -2
  54. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_serve.py +31 -4
  55. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_artifact_modal.py +25 -3
  56. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_model.py +22 -11
  57. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_proposals.py +1 -2
  58. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_read_grant.py +0 -1
  59. {rockycode-0.1.0 → rockycode-0.1.2}/uv.lock +34 -1
  60. rockycode-0.1.0/.github/workflows/release.yml +0 -73
  61. rockycode-0.1.0/CHANGELOG.md +0 -20
  62. rockycode-0.1.0/rockycode/__init__.py +0 -1
  63. {rockycode-0.1.0 → rockycode-0.1.2}/.dockerignore +0 -0
  64. {rockycode-0.1.0 → rockycode-0.1.2}/.gitignore +0 -0
  65. {rockycode-0.1.0 → rockycode-0.1.2}/Dockerfile +0 -0
  66. {rockycode-0.1.0 → rockycode-0.1.2}/Dockerfile.sandbox +0 -0
  67. {rockycode-0.1.0 → rockycode-0.1.2}/LICENSE +0 -0
  68. {rockycode-0.1.0 → rockycode-0.1.2}/SECURITY.md +0 -0
  69. {rockycode-0.1.0 → rockycode-0.1.2}/bench/tasks/dev10.json +0 -0
  70. {rockycode-0.1.0 → rockycode-0.1.2}/bench/tasks/test20.json +0 -0
  71. {rockycode-0.1.0 → rockycode-0.1.2}/brand/rockycode-note.svg +0 -0
  72. {rockycode-0.1.0 → rockycode-0.1.2}/brand/rockycode-wordmark.svg +0 -0
  73. {rockycode-0.1.0 → rockycode-0.1.2}/docker-compose.yml +0 -0
  74. {rockycode-0.1.0 → rockycode-0.1.2}/prompts/README.md +0 -0
  75. {rockycode-0.1.0 → rockycode-0.1.2}/prompts/rocky-v1.txt +0 -0
  76. {rockycode-0.1.0 → rockycode-0.1.2}/prompts/rocky-v2-search-first.txt +0 -0
  77. {rockycode-0.1.0 → rockycode-0.1.2}/prompts/rocky-v3-decisive.txt +0 -0
  78. {rockycode-0.1.0 → rockycode-0.1.2}/prompts/rocky-zh-closer.txt +0 -0
  79. {rockycode-0.1.0 → rockycode-0.1.2}/prompts/rocky-zh-full-closer.txt +0 -0
  80. {rockycode-0.1.0 → rockycode-0.1.2}/prompts/rocky-zh-full.txt +0 -0
  81. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/banner.py +0 -0
  82. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/dream/__init__.py +0 -0
  83. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/dream/core.py +0 -0
  84. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/dream/judge.py +0 -0
  85. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/dream/mining.py +0 -0
  86. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/dream/proposals.py +0 -0
  87. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/__init__.py +0 -0
  88. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/budget.py +0 -0
  89. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/checks.py +0 -0
  90. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/effort.py +0 -0
  91. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/events.py +0 -0
  92. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/explore.py +0 -0
  93. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/goal_review.py +0 -0
  94. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/lsp.py +0 -0
  95. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/mcp.py +0 -0
  96. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/modes.py +0 -0
  97. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/outcome.py +0 -0
  98. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/permission.py +0 -0
  99. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/planmode.py +0 -0
  100. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/redact.py +0 -0
  101. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/safety.py +0 -0
  102. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/skills.py +0 -0
  103. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/titler.py +0 -0
  104. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/trajectory.py +0 -0
  105. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/web.py +0 -0
  106. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/engine/worktree.py +0 -0
  107. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/memory/__init__.py +0 -0
  108. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/memory/index.py +0 -0
  109. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/memory/store.py +0 -0
  110. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/modes/learn/learn.md +0 -0
  111. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/modes/research/deep-research.md +0 -0
  112. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/modes/research/paper-reading.md +0 -0
  113. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/modes/research/prove.md +0 -0
  114. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/modes/research/whiteboard.md +0 -0
  115. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/palette.py +0 -0
  116. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/prompts/__init__.py +0 -0
  117. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/prompts/rocky.py +0 -0
  118. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/runners/__init__.py +0 -0
  119. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/runners/agent.py +0 -0
  120. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/runners/data.py +0 -0
  121. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/runners/raw.py +0 -0
  122. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/score.py +0 -0
  123. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/skills/architecture-viz/SKILL.md +0 -0
  124. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/skills/architecture-viz/template.html +0 -0
  125. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/skills/lean-prover/SKILL.md +0 -0
  126. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/skills/lean-prover/torchlean-api.md +0 -0
  127. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/__init__.py +0 -0
  128. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/exitsheet.py +0 -0
  129. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/mdterm.py +0 -0
  130. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/mdview.py +0 -0
  131. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/modepicker.py +0 -0
  132. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/permission.py +0 -0
  133. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/plangate.py +0 -0
  134. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/prompt_history.py +0 -0
  135. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/proposalcard.py +0 -0
  136. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/rocky_pet.py +0 -0
  137. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode/tui/routinecard.py +0 -0
  138. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/.gitignore +0 -0
  139. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/.vscode/launch.json +0 -0
  140. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/.vscode/tasks.json +0 -0
  141. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/.vscodeignore +0 -0
  142. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/CHANGELOG.md +0 -0
  143. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/LICENSE +0 -0
  144. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/README.md +0 -0
  145. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/esbuild.config.mjs +0 -0
  146. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/media/icon-marketplace.svg +0 -0
  147. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/media/icon.png +0 -0
  148. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/media/marked.js +0 -0
  149. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/media/rocky-icon.svg +0 -0
  150. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/diffManager.ts +0 -0
  151. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/editorContext.ts +0 -0
  152. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/permissionManager.ts +0 -0
  153. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/statusBar.ts +0 -0
  154. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/src/webview-highlight.ts +0 -0
  155. {rockycode-0.1.0 → rockycode-0.1.2}/rockycode-vscode/tsconfig.json +0 -0
  156. {rockycode-0.1.0 → rockycode-0.1.2}/tests/fake_lsp_server.py +0 -0
  157. {rockycode-0.1.0 → rockycode-0.1.2}/tests/fake_mcp_server.py +0 -0
  158. {rockycode-0.1.0 → rockycode-0.1.2}/tests/invariants.py +0 -0
  159. {rockycode-0.1.0 → rockycode-0.1.2}/tests/real_planmode.py +0 -0
  160. {rockycode-0.1.0 → rockycode-0.1.2}/tests/real_reasoning_roundtrip.py +0 -0
  161. {rockycode-0.1.0 → rockycode-0.1.2}/tests/real_tool_contract.py +0 -0
  162. {rockycode-0.1.0 → rockycode-0.1.2}/tests/run_real.py +0 -0
  163. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_approver.py +0 -0
  164. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_bash.py +0 -0
  165. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_bash_grant.py +0 -0
  166. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_bilang_prompt.py +0 -0
  167. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_budget.py +0 -0
  168. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_compaction.py +0 -0
  169. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_credentials.py +0 -0
  170. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_dream.py +0 -0
  171. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_dream_trigger.py +0 -0
  172. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_effort.py +0 -0
  173. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_explore.py +0 -0
  174. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_goal.py +0 -0
  175. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_goal_driver.py +0 -0
  176. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_goal_review.py +0 -0
  177. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_goal_runner.py +0 -0
  178. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_interrupt.py +0 -0
  179. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_judge.py +0 -0
  180. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_lsp.py +0 -0
  181. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_mcp.py +0 -0
  182. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_mdterm.py +0 -0
  183. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_memory.py +0 -0
  184. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_memory_index.py +0 -0
  185. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_mining.py +0 -0
  186. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_modes.py +0 -0
  187. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_onboarding.py +0 -0
  188. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_outcome.py +0 -0
  189. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_parallel.py +0 -0
  190. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_permission.py +0 -0
  191. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_planmode.py +0 -0
  192. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_prompt_history.py +0 -0
  193. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_proposals.py +0 -0
  194. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_providers.py +0 -0
  195. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_redact.py +0 -0
  196. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_resume_handoff.py +0 -0
  197. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_routine_run.py +0 -0
  198. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_safety.py +0 -0
  199. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_score.py +0 -0
  200. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_session.py +0 -0
  201. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_skills.py +0 -0
  202. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_submit_race.py +0 -0
  203. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_today.py +0 -0
  204. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tools.py +0 -0
  205. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tools_jail.py +0 -0
  206. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_trajectory.py +0 -0
  207. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_bash_gate.py +0 -0
  208. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_copy.py +0 -0
  209. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_envwarn.py +0 -0
  210. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_exitsheet.py +0 -0
  211. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_goal_screen.py +0 -0
  212. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_input_nav.py +0 -0
  213. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_permission.py +0 -0
  214. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_plan.py +0 -0
  215. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_routines.py +0 -0
  216. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_scroll.py +0 -0
  217. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_shell.py +0 -0
  218. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_tui_toggle.py +0 -0
  219. {rockycode-0.1.0 → rockycode-0.1.2}/tests/smoke_web.py +0 -0
@@ -23,5 +23,8 @@ jobs:
23
23
  - name: Sync deps from the lockfile
24
24
  run: uv sync --frozen
25
25
 
26
+ - name: Ruff (conservative correctness gate)
27
+ run: uv run ruff check .
28
+
26
29
  - name: Smoke tests (the CORE gate)
27
30
  run: uv run python tests/run_all.py
@@ -0,0 +1,27 @@
1
+ name: Release
2
+
3
+ # One job: build sdist+wheel on ubuntu, publish to PyPI via Trusted Publishing.
4
+ # The old macos-13 matrix (built a cbor2 x86_64-mac wheel for the release page)
5
+ # is gone: the runner label was retired, so it never actually ran — 0.1.0 and
6
+ # 0.1.1 both sat in its queue until cancelled and shipped fine without it — and
7
+ # cbor2 still publishes no x86_64-mac wheel to attach anyway (arm64 only; an
8
+ # Intel-Mac install builds it from source via Rust, PyPI-side, as it always
9
+ # has). GitHub releases are created by hand: gh release create --notes-file.
10
+ on:
11
+ push:
12
+ tags: ["v*"]
13
+ workflow_dispatch:
14
+
15
+ jobs:
16
+ pypi:
17
+ runs-on: ubuntu-latest
18
+ permissions:
19
+ id-token: write # OIDC — this is what Trusted Publishing uses
20
+ steps:
21
+ - uses: actions/checkout@v4
22
+ - name: Setup uv
23
+ uses: astral-sh/setup-uv@v5
24
+ - name: Build rockycode (sdist + wheel — only our project lands in dist/)
25
+ run: uv build
26
+ - name: Publish to PyPI (Trusted Publishing, no token)
27
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,61 @@
1
+ # Changelog
2
+
3
+ All notable changes to rockycode are recorded here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/), and the project follows
5
+ [Semantic Versioning](https://semver.org/) — pre-1.0, so the surface may still
6
+ change between minor versions.
7
+
8
+ ## [Unreleased]
9
+
10
+ _Nothing yet._
11
+
12
+ ## [0.1.2] — `--version` flag, GA price tables
13
+
14
+ ### Added
15
+ - `rockycode --version` / `-V` prints the installed version. One version
16
+ source: package metadata (pyproject) — the serve handshake reports the same
17
+ value instead of a hardcoded string.
18
+
19
+ ### Changed
20
+ - DeepSeek price tables refreshed to the GA snapshots (V4-Flash-0731 /
21
+ V4-Pro-0813), both USD and CNY, verified 2026-08-20 at the source;
22
+ peak-valley billing confirmed live (2× in the published UTC windows).
23
+ - README results updated: `deepseek-v4-flash` GA carries three clean full-500
24
+ SWE-bench Verified rounds (88.8% average, 95.4% pass@3); the
25
+ `deepseek-v4-pro` column is explicitly marked preview (pre-0813).
26
+
27
+ ### CI
28
+ - Release workflow reduced to the single ubuntu PyPI Trusted-Publishing job;
29
+ the dead macos-13 matrix (retired runner, never ran) is gone.
30
+
31
+ ## [0.1.1] — session artifacts, images in chat, real sandbox cancel
32
+
33
+ ### Added
34
+ - Images in chat: paste (`ctrl+v` / `/paste`) or drag an image in. Vision models
35
+ see it raw; no-vision models route through a provider sidecar, your own image
36
+ CLI, or a `view_image` tool — picked once, remembered. Known CLIs set up with
37
+ one word (`/config image_cli mmx` — rocky knows the invocation).
38
+ - Session artifact inventory: `/artifact list · open <n> · stop · live on|off`,
39
+ a footer badge with open-tab counts, and an Artifacts tree in the VS Code
40
+ extension fed live by `rockycode serve`.
41
+ - Bare `/model` opens a live provider + model picker.
42
+
43
+ ### Fixed
44
+ - Live artifacts no longer drop and reconnect every 30 s; the artifact server
45
+ stops/restarts cleanly and rebinds saved live pages to the new port.
46
+ - Sandbox cancel/timeout kills the in-container process group, not just the
47
+ host-side docker client; images without `python3` fall back to plain `bash -c`.
48
+ - Resume self-heals after a hard kill; closing the doc dock no longer wedges the TUI.
49
+
50
+ ### Internal
51
+ - CI: conservative ruff correctness gate (pinned `0.16.1`) ahead of the smoke suite.
52
+
53
+ ## [0.1.0] — first public release
54
+
55
+ Initial public release. One repo, one engine, three ways to use it — interactive
56
+ `chat`, autonomous `goal`, and the `bench` measurement rig — running on DeepSeek
57
+ or any OpenAI-compatible model.
58
+
59
+ Some capabilities ship as **experimental and default-off** (self-improvement,
60
+ `prove` / `lean-prover`, `explore`, and providers other than DeepSeek); see the
61
+ README's Experimental section for what they are and how to enable them.
@@ -40,7 +40,7 @@ rockycode goal "…" # autonomous run (sandboxed worktree copy)
40
40
  ## Before you open a PR
41
41
 
42
42
  1. **Run the test suite** — `python tests/run_all.py`. It should be green.
43
- 2. **Lint** — `ruff check .` (config is in `pyproject.toml`).
43
+ 2. **Lint** — `uv run ruff check .` (pinned version and conservative rules are in `pyproject.toml`).
44
44
  3. **Match the surrounding code.** Read the file you're editing and mirror its
45
45
  naming, comment density, and idioms. New code should be indistinguishable in
46
46
  style from what's around it.
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: rockycode
3
- Version: 0.1.0
3
+ Version: 0.1.2
4
4
  Summary: A coding agent harness, benchmarked on SWE-bench Verified. amaze!
5
5
  Author: rockycode contributors
6
6
  License: MIT
@@ -42,7 +42,8 @@ Built for the DeepSeek V4 series, with a unique research mode, bench-tested, and
42
42
 
43
43
  [English](README.md) · [简体中文](README.zh-CN.md)
44
44
 
45
- ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-~80%25_100--task_slice-7d5cc6)
45
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-88.8%25_V4--flash_GA-7d5cc6)
46
+ ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-74.9%25-8d6cd0)
46
47
  ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
47
48
  ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
48
49
 
@@ -76,17 +77,36 @@ against it.
76
77
 
77
78
  ## Results — SWE-bench Verified
78
79
 
79
- On SWE-bench Verified, a **randomly-chosen 100-task slice** (resources are
80
- limited — no repeated runs averaged over the full 500):
81
-
82
- **Result: ≈80% on a 100-task SWE-bench Verified slice** with `deepseek-v4-pro`,
83
- no tuning against those tasks — the mean of a four-arm configuration sweep.
84
-
85
- Read it with its limits. It's a representative **random 100-task slice**, not
86
- the official 500-task Verified set: its harder 20-task core scored 60–70% while
87
- the other 80 scored >80%. Treat it as an honest internal measurement, not a
88
- leaderboard entry; for the full breakdown, follow our X
89
- ([@rockycode_ai](https://x.com/rockycode_ai)).
80
+ Full-set numbers: **independent full-500 runs**, three rounds per model, same
81
+ harness and config for all (100-step cap, 32,768 max output tokens, reasoning
82
+ effort `max`, thinking on), scored with the official SWE-bench harness. No
83
+ tuning against the tasks. Mind the versions: the `deepseek-v4-flash` column is
84
+ the **GA release** (V4-Flash-0731), while the `deepseek-v4-pro` rounds ran
85
+ before the GA V4-Pro-0813 shipped — that column is the **preview** pro, so the
86
+ flash/pro gap reflects a version difference, not a same-vintage comparison.
87
+
88
+ | run | `deepseek-v4-flash` (GA) | `deepseek-v4-pro` (preview) | `minimax-m3` |
89
+ |---|---|---|---|
90
+ | round 1 | 90.0% (450/500) | 75.6% (378/500) | 72.8% (364/500) |
91
+ | round 2 | 88.6% (443/500) | 74.8% (374/500) | 71.2% (356/500) |
92
+ | round 3 | 87.8% (439/500) | 74.4% (372/500) | 70.0% (350/500) |
93
+ | **average** | **88.8%** | **74.9%** | **71.3%** |
94
+ | union of runs (pass@3) | 95.4% (477/500) | 81.8% (409/500) | 83.6% (418/500) |
95
+
96
+ Read the two summary rows differently. The **average** is the
97
+ leaderboard-comparable number — each round is an independent single-pass run
98
+ over the full 500. The **union** is pass@3: tasks solved by at least one
99
+ round. The gap between them (~7 points for both DeepSeek models, ~12 for
100
+ MiniMax) is run-to-run variance, not capability — the models already reach
101
+ these tasks under this harness, they just don't hold them every run. Closing
102
+ that gap (verify-before-finish gating and run selection, not more prompting)
103
+ is the current line of work. For reference, DeepSeek reported 80.6% for V4
104
+ Pro (Preview) with its own scaffold. Per-round breakdowns:
105
+ [@rockycode_ai](https://x.com/rockycode_ai).
106
+
107
+ Two earlier flash rounds (79.8% — previously listed here — and 81.2%) hit
108
+ local network outages mid-run, visible as contiguous blocks of empty-patch
109
+ tasks; they were replaced by clean re-runs rather than averaged in.
90
110
 
91
111
  We plan to add **DeepSWE-bench** support as well — currently in progress.
92
112
 
@@ -102,10 +122,42 @@ in a container: `goal` (autonomous runs), `exec` (headless delegation),
102
122
  run offline in the sandbox by design, so a delegated or unattended task cannot
103
123
  touch your host or reach the network.
104
124
 
125
+ ```bash
126
+ uv tool install rockycode # recommended — puts the `rockycode` command on your PATH
127
+ rockycode # the first run walks you through API-key setup
128
+ ```
129
+
130
+ Already installed and a new release is out? **`uv tool upgrade rockycode`** —
131
+ note that re-running `uv tool install` does NOT upgrade: it sees the existing
132
+ install and quietly keeps the old version.
133
+
134
+ Don't have uv yet? One command installs it:
135
+ `curl -LsSf https://astral.sh/uv/install.sh | sh` — Windows and other options
136
+ in the [uv install docs](https://docs.astral.sh/uv/getting-started/installation/).
137
+
138
+ Three ways to install — they look similar but land in different places:
139
+
140
+ - **`uv tool install rockycode`** (recommended) — gives the CLI its own
141
+ isolated environment and puts `rockycode` on your PATH; if your system
142
+ Python is older than 3.11, uv fetches a matching interpreter by itself.
143
+ The "install it like an app" path.
144
+ - **`uv pip install rockycode`** — installs into the **currently active
145
+ virtual environment** only: the `rockycode` command exists inside that
146
+ venv, so a new shell won't find it unless the venv is active (or run it
147
+ as `uv run rockycode`).
148
+ - **`pip install rockycode`** — same venv caveat as above, and it needs
149
+ Python 3.11+. On an older Python it fails with the misleading
150
+ `ERROR: No matching distribution found for rockycode`. Why: pip only
151
+ offers releases whose `requires-python` matches your interpreter, so on an
152
+ old Python it sees no installable version at all and reports that as a
153
+ missing package. If you hit this, don't fight it — use
154
+ `uv tool install rockycode` above; uv brings its own Python 3.11+.
155
+
156
+ Or install from source:
157
+
105
158
  ```bash
106
159
  git clone https://github.com/cicialgo/rockycode.git && cd rockycode
107
- uv tool install . # puts the `rockycode` command on your PATH
108
- rockycode # the first run walks you through API-key setup
160
+ uv tool install .
109
161
  ```
110
162
 
111
163
  On first launch you paste your API key once. It is stored in the OS keychain
@@ -157,7 +209,8 @@ clipboard" (or your terminal's equivalent) on the local end.
157
209
  | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
158
210
  | `/sandbox on\|off\|status` | Isolate tool execution in a container |
159
211
  | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
160
- | `/artifact live on\|off` | Auto-refresh HTML artifacts in the browser |
212
+ | `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
213
+ | `/paste` | Attach a clipboard image (or `ctrl+v`); no-vision models pick a route |
161
214
  | `/prompt` | Inspect the live system prompt |
162
215
  | `/mcp` | Connected MCP servers and their tools |
163
216
  | `/skills` | Installed skills |
@@ -191,7 +244,7 @@ OpenAI-compatible API.
191
244
 
192
245
  | Provider | Models |
193
246
  |---|---|
194
- | **deepseek** (default) | `deepseek-v4-pro`, `deepseek-v4-flash` |
247
+ | **deepseek** (default) | `deepseek-v4-flash` (default), `deepseek-v4-pro` (preview) |
195
248
  | **minimax** | `minimax-m3` |
196
249
  | **kimi** | `kimi-k3` |
197
250
  | **glm** | `glm-5.2` |
@@ -199,8 +252,9 @@ OpenAI-compatible API.
199
252
  Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
200
253
  and custom providers — including local vLLM/SGLang servers — go in
201
254
  `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
202
- keys are actually configured. Only DeepSeek is verified on the harness; the
203
- others are [experimental](#experimental).
255
+ keys are actually configured. DeepSeek and MiniMax both carry full-500 bench
256
+ numbers (see [Results](#results--swe-bench-verified)); Kimi and GLM are
257
+ [experimental](#experimental).
204
258
 
205
259
  The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
206
260
  provider maps it to its own reasoning tiers at the wire (DeepSeek, for
@@ -294,8 +348,9 @@ change. Anything that could act on its own is **off by default**.
294
348
  mechanically-verified report; the search noise never enters your session. It
295
349
  also grounds goal mode's branch review and milestone verification.
296
350
  - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
297
- OpenAI-compatible profiles (`/model`), but only DeepSeek is verified on the
298
- harness — treat the others as untested until they carry a bench number.
351
+ OpenAI-compatible profiles (`/model`). DeepSeek and MiniMax carry full
352
+ bench numbers (see Results); treat GLM and Kimi as untested until they do
353
+ too.
299
354
 
300
355
  ## Works with your existing setup
301
356
 
@@ -9,7 +9,8 @@ Built for the DeepSeek V4 series, with a unique research mode, bench-tested, and
9
9
 
10
10
  [English](README.md) · [简体中文](README.zh-CN.md)
11
11
 
12
- ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-~80%25_100--task_slice-7d5cc6)
12
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-88.8%25_V4--flash_GA-7d5cc6)
13
+ ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-74.9%25-8d6cd0)
13
14
  ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
14
15
  ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
15
16
 
@@ -43,17 +44,36 @@ against it.
43
44
 
44
45
  ## Results — SWE-bench Verified
45
46
 
46
- On SWE-bench Verified, a **randomly-chosen 100-task slice** (resources are
47
- limited — no repeated runs averaged over the full 500):
48
-
49
- **Result: ≈80% on a 100-task SWE-bench Verified slice** with `deepseek-v4-pro`,
50
- no tuning against those tasks — the mean of a four-arm configuration sweep.
51
-
52
- Read it with its limits. It's a representative **random 100-task slice**, not
53
- the official 500-task Verified set: its harder 20-task core scored 60–70% while
54
- the other 80 scored >80%. Treat it as an honest internal measurement, not a
55
- leaderboard entry; for the full breakdown, follow our X
56
- ([@rockycode_ai](https://x.com/rockycode_ai)).
47
+ Full-set numbers: **independent full-500 runs**, three rounds per model, same
48
+ harness and config for all (100-step cap, 32,768 max output tokens, reasoning
49
+ effort `max`, thinking on), scored with the official SWE-bench harness. No
50
+ tuning against the tasks. Mind the versions: the `deepseek-v4-flash` column is
51
+ the **GA release** (V4-Flash-0731), while the `deepseek-v4-pro` rounds ran
52
+ before the GA V4-Pro-0813 shipped — that column is the **preview** pro, so the
53
+ flash/pro gap reflects a version difference, not a same-vintage comparison.
54
+
55
+ | run | `deepseek-v4-flash` (GA) | `deepseek-v4-pro` (preview) | `minimax-m3` |
56
+ |---|---|---|---|
57
+ | round 1 | 90.0% (450/500) | 75.6% (378/500) | 72.8% (364/500) |
58
+ | round 2 | 88.6% (443/500) | 74.8% (374/500) | 71.2% (356/500) |
59
+ | round 3 | 87.8% (439/500) | 74.4% (372/500) | 70.0% (350/500) |
60
+ | **average** | **88.8%** | **74.9%** | **71.3%** |
61
+ | union of runs (pass@3) | 95.4% (477/500) | 81.8% (409/500) | 83.6% (418/500) |
62
+
63
+ Read the two summary rows differently. The **average** is the
64
+ leaderboard-comparable number — each round is an independent single-pass run
65
+ over the full 500. The **union** is pass@3: tasks solved by at least one
66
+ round. The gap between them (~7 points for both DeepSeek models, ~12 for
67
+ MiniMax) is run-to-run variance, not capability — the models already reach
68
+ these tasks under this harness, they just don't hold them every run. Closing
69
+ that gap (verify-before-finish gating and run selection, not more prompting)
70
+ is the current line of work. For reference, DeepSeek reported 80.6% for V4
71
+ Pro (Preview) with its own scaffold. Per-round breakdowns:
72
+ [@rockycode_ai](https://x.com/rockycode_ai).
73
+
74
+ Two earlier flash rounds (79.8% — previously listed here — and 81.2%) hit
75
+ local network outages mid-run, visible as contiguous blocks of empty-patch
76
+ tasks; they were replaced by clean re-runs rather than averaged in.
57
77
 
58
78
  We plan to add **DeepSWE-bench** support as well — currently in progress.
59
79
 
@@ -69,10 +89,42 @@ in a container: `goal` (autonomous runs), `exec` (headless delegation),
69
89
  run offline in the sandbox by design, so a delegated or unattended task cannot
70
90
  touch your host or reach the network.
71
91
 
92
+ ```bash
93
+ uv tool install rockycode # recommended — puts the `rockycode` command on your PATH
94
+ rockycode # the first run walks you through API-key setup
95
+ ```
96
+
97
+ Already installed and a new release is out? **`uv tool upgrade rockycode`** —
98
+ note that re-running `uv tool install` does NOT upgrade: it sees the existing
99
+ install and quietly keeps the old version.
100
+
101
+ Don't have uv yet? One command installs it:
102
+ `curl -LsSf https://astral.sh/uv/install.sh | sh` — Windows and other options
103
+ in the [uv install docs](https://docs.astral.sh/uv/getting-started/installation/).
104
+
105
+ Three ways to install — they look similar but land in different places:
106
+
107
+ - **`uv tool install rockycode`** (recommended) — gives the CLI its own
108
+ isolated environment and puts `rockycode` on your PATH; if your system
109
+ Python is older than 3.11, uv fetches a matching interpreter by itself.
110
+ The "install it like an app" path.
111
+ - **`uv pip install rockycode`** — installs into the **currently active
112
+ virtual environment** only: the `rockycode` command exists inside that
113
+ venv, so a new shell won't find it unless the venv is active (or run it
114
+ as `uv run rockycode`).
115
+ - **`pip install rockycode`** — same venv caveat as above, and it needs
116
+ Python 3.11+. On an older Python it fails with the misleading
117
+ `ERROR: No matching distribution found for rockycode`. Why: pip only
118
+ offers releases whose `requires-python` matches your interpreter, so on an
119
+ old Python it sees no installable version at all and reports that as a
120
+ missing package. If you hit this, don't fight it — use
121
+ `uv tool install rockycode` above; uv brings its own Python 3.11+.
122
+
123
+ Or install from source:
124
+
72
125
  ```bash
73
126
  git clone https://github.com/cicialgo/rockycode.git && cd rockycode
74
- uv tool install . # puts the `rockycode` command on your PATH
75
- rockycode # the first run walks you through API-key setup
127
+ uv tool install .
76
128
  ```
77
129
 
78
130
  On first launch you paste your API key once. It is stored in the OS keychain
@@ -124,7 +176,8 @@ clipboard" (or your terminal's equivalent) on the local end.
124
176
  | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
125
177
  | `/sandbox on\|off\|status` | Isolate tool execution in a container |
126
178
  | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
127
- | `/artifact live on\|off` | Auto-refresh HTML artifacts in the browser |
179
+ | `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
180
+ | `/paste` | Attach a clipboard image (or `ctrl+v`); no-vision models pick a route |
128
181
  | `/prompt` | Inspect the live system prompt |
129
182
  | `/mcp` | Connected MCP servers and their tools |
130
183
  | `/skills` | Installed skills |
@@ -158,7 +211,7 @@ OpenAI-compatible API.
158
211
 
159
212
  | Provider | Models |
160
213
  |---|---|
161
- | **deepseek** (default) | `deepseek-v4-pro`, `deepseek-v4-flash` |
214
+ | **deepseek** (default) | `deepseek-v4-flash` (default), `deepseek-v4-pro` (preview) |
162
215
  | **minimax** | `minimax-m3` |
163
216
  | **kimi** | `kimi-k3` |
164
217
  | **glm** | `glm-5.2` |
@@ -166,8 +219,9 @@ OpenAI-compatible API.
166
219
  Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
167
220
  and custom providers — including local vLLM/SGLang servers — go in
168
221
  `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
169
- keys are actually configured. Only DeepSeek is verified on the harness; the
170
- others are [experimental](#experimental).
222
+ keys are actually configured. DeepSeek and MiniMax both carry full-500 bench
223
+ numbers (see [Results](#results--swe-bench-verified)); Kimi and GLM are
224
+ [experimental](#experimental).
171
225
 
172
226
  The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
173
227
  provider maps it to its own reasoning tiers at the wire (DeepSeek, for
@@ -261,8 +315,9 @@ change. Anything that could act on its own is **off by default**.
261
315
  mechanically-verified report; the search noise never enters your session. It
262
316
  also grounds goal mode's branch review and milestone verification.
263
317
  - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
264
- OpenAI-compatible profiles (`/model`), but only DeepSeek is verified on the
265
- harness — treat the others as untested until they carry a bench number.
318
+ OpenAI-compatible profiles (`/model`). DeepSeek and MiniMax carry full
319
+ bench numbers (see Results); treat GLM and Kimi as untested until they do
320
+ too.
266
321
 
267
322
  ## Works with your existing setup
268
323
 
@@ -9,7 +9,8 @@
9
9
 
10
10
  [English](README.md) · [简体中文](README.zh-CN.md)
11
11
 
12
- ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-~80%25_100--task_slice-7d5cc6)
12
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-88.8%25_V4--flash_GA-7d5cc6)
13
+ ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-74.9%25-8d6cd0)
13
14
  ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
14
15
  ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
15
16
 
@@ -36,11 +37,19 @@ rockycode 是一个编程智能体 harness,为 DeepSeek V4 系列适配,也
36
37
 
37
38
  ## 能力量化:SWE bench
38
39
 
39
- 在 SWE-bench Verified 上,**随机抽选 100 题**测试(资源所限,没有在完整 500 题上做重复取平均):
40
+ 完整 500 题的成绩:**独立的完整 500 题运行**,每个模型各 3 轮,所有模型使用完全相同的 harness 与配置(步数上限 100、单次最大输出 32,768 token、推理力度 `max`、thinking 开启),由官方 SWE-bench harness 打分,未针对任务做任何调优。注意版本差异:`deepseek-v4-flash` 一列是**正式版(GA,V4-Flash-0731)**;而 `deepseek-v4-pro` 的三轮在 GA 版 V4-Pro-0813 发布之前运行,该列是 **preview 版**成绩 —— flash 与 pro 的差距里含有版本差,不是同代对比。
40
41
 
41
- **结果:在这个 100 题切片上约 80%** —— 用 `deepseek-v4-pro`,未针对这些任务做调优,是一次四臂配置扫描的均值。
42
+ | 轮次 | `deepseek-v4-flash`(GA) | `deepseek-v4-pro`(preview) | `minimax-m3` |
43
+ |---|---|---|---|
44
+ | 第 1 轮 | 90.0%(450/500) | 75.6%(378/500) | 72.8%(364/500) |
45
+ | 第 2 轮 | 88.6%(443/500) | 74.8%(374/500) | 71.2%(356/500) |
46
+ | 第 3 轮 | 87.8%(439/500) | 74.4%(372/500) | 70.0%(350/500) |
47
+ | **平均** | **88.8%** | **74.9%** | **71.3%** |
48
+ | 多轮并集(pass@3) | 95.4%(477/500) | 81.8%(409/500) | 83.6%(418/500) |
42
49
 
43
- 请连同它的边界一起看:这是一个**随机抽取的代表性切片**,不是官方 500 题的完整Verified 集;其中较难的 20 题核心得分 60–70%,另外 80 题得分 >80%。请把它当作诚实的内部测量,而非排行榜成绩;完整拆解请关注我们的X账号(@rockycode_ai)。
50
+ 两行汇总要分开读。**平均值**是可以与排行榜对比的数字 —— 每一轮都是独立的单次完整 500 题。**并集**是 pass@3:至少被某一轮解出的任务。两者之间的差距(DeepSeek 两个模型各约 7 个点、MiniMax 约 12 个点)是轮次间方差,不是能力上限 —— 模型在这套 harness 下已经"够得着"这些任务,只是无法每一轮都稳住。当前的工作重心就是收掉这个差距:finish 前的验证门控与多轮选择,而不是继续改提示词。作为参照:DeepSeek 用自家 scaffold 报告 V4 Pro (Preview) 为 80.6%。逐轮拆解请关注我们的X账号(@rockycode_ai)。
51
+
52
+ 另有两轮较早的 flash-preview 运行(79.8% —— 即此前列在这里的那一轮 —— 与 81.2%)在运行途中遭遇本地网络中断,表现为连续任务块返回空补丁;这两轮已用干净的重跑替换,未计入平均。
44
53
 
45
54
  我们计划支持DeepSWE bench,目前还在调试中。
46
55
 
@@ -50,10 +59,38 @@ rockycode 是一个编程智能体 harness,为 DeepSeek V4 系列适配,也
50
59
 
51
60
  **Docker Desktop** 仅在需要容器隔离工具执行的模式下才必需:`goal`(自主运行)、`exec`(自动化委托)、`bench`(SWE-bench 打分),以及 chat 里可选的`/sandbox`。这些模式在沙箱内默认离线运行,被委托或无人值守的任务因此无法触碰你的主机、无法访问网络。
52
61
 
62
+ ```bash
63
+ uv tool install rockycode # 推荐 —— `rockycode` 命令直接进 PATH
64
+ rockycode # 首次运行会引导你完成 API key 设置
65
+ ```
66
+
67
+ 已经装过、想升到新版本?用 **`uv tool upgrade rockycode`** ——
68
+ 注意重复执行 `uv tool install` 不会升级:它看到已有安装就静默保留旧版本。
69
+
70
+ 还没有 uv?一条命令安装:
71
+ `curl -LsSf https://astral.sh/uv/install.sh | sh`(Windows 及其他方式见
72
+ [uv 安装文档](https://docs.astral.sh/uv/getting-started/installation/))。
73
+
74
+ 三种装法看着像,落点完全不同:
75
+
76
+ - **`uv tool install rockycode`**(推荐)—— 给 CLI 一个独立的隔离环境,并把
77
+ `rockycode` 命令放上 PATH;系统 Python 低于 3.11 时,uv 会自动拉取一个匹配
78
+ 的解释器。想「当应用装」就用它。
79
+ - **`uv pip install rockycode`** —— 只装进**当前激活的虚拟环境**:`rockycode`
80
+ 命令只存在于那个 venv 里,新开一个 shell 会找不到它(要么先激活 venv,
81
+ 要么用 `uv run rockycode` 运行)。
82
+ - **`pip install rockycode`** —— 同样只进当前环境,且要求 Python 3.11+。在更老
83
+ 的 Python 上会报一个很有误导性的
84
+ `ERROR: No matching distribution found for rockycode`。原因:pip 只会提供
85
+ `requires-python` 与你解释器匹配的版本,老 Python 下它一个可装的版本都看
86
+ 不到,于是把「版本不满足」报成了「包不存在」。遇到这个错不用纠结 —— 直接用
87
+ 上面的 `uv tool install rockycode`,uv 自带 3.11+ 的 Python。
88
+
89
+ 或从源码安装:
90
+
53
91
  ```bash
54
92
  git clone https://github.com/cicialgo/rockycode.git && cd rockycode
55
- uv tool install . # 把 `rockycode` 命令装到 PATH
56
- rockycode # 首次运行会引导你完成 API key 设置
93
+ uv tool install .
57
94
  ```
58
95
 
59
96
  首次启动只需粘贴一次 API key。它被存入操作系统钥匙串(安装 `[keyring]`扩展时)或 `~/.rockycode/.env` 私有文件(权限 `0600`)—— 不进你的项目或者shell 配置。rockycode 从不读取项目内的 `.env`:克隆来的仓库不应有能力注入 key 或改写 endpoint,因此其中形似凭据的变量只会按名字给出警告,值不会被读取。
@@ -98,7 +135,8 @@ SSH 远程会话下剪贴板走 OSC 52 —— 在本地端开启「允许应用
98
135
  | `/permission yolo\|ask\|careful` | 本次会话的工具审批严格度 |
99
136
  | `/sandbox on\|off\|status` | 把工具执行隔离进容器 |
100
137
  | `/lsp` | 语言服务器状态;诊断信息随 `read_file` 一并返回 |
101
- | `/artifact live on\|off` | 浏览器中自动刷新 HTML artifact |
138
+ | `/artifact` | 本会话的 artifact:`list` · `open <n>` · `stop` · `live on\|off` |
139
+ | `/paste` | 粘贴剪贴板图片(或 `ctrl+v`);无视觉模型自选识图路由 |
102
140
  | `/prompt` | 查看当前生效的系统提示词 |
103
141
  | `/mcp` | 已连接的 MCP 服务及其工具 |
104
142
  | `/skills` | 已安装的技能 |
@@ -128,15 +166,15 @@ DeepSeek 是主场模型,但提供商是数据而非代码:每个提供商
128
166
 
129
167
  | 提供商 | 模型 |
130
168
  |---|---|
131
- | **deepseek**(默认) | `deepseek-v4-pro`、`deepseek-v4-flash` |
169
+ | **deepseek**(默认) | `deepseek-v4-flash`(默认)、`deepseek-v4-pro`(preview) |
132
170
  | **minimax** | `minimax-m3` |
133
171
  | **kimi** | `kimi-k3` |
134
172
  | **glm** | `glm-5.2` |
135
173
 
136
174
  区域端点写作 `<提供商>-<区域>`(如 `kimi-cn`);自定义提供商 —— 包括本地
137
175
  vLLM/SGLang 服务 —— 写进 `~/.rockycode/providers.toml`。`/model` 选择器
138
- 只展示已配置好 key 的提供商。只有 DeepSeek 在 harness 上验证过,其余为
139
- [实验性功能](#实验性功能)。
176
+ 只展示已配置好 key 的提供商。DeepSeek 与 MiniMax 都有完整 500 题的 bench
177
+ 成绩(见上方「能力量化」);Kimi 与 GLM 属于[实验性功能](#实验性功能)。
140
178
 
141
179
  推理深度旋钮(`/effort off|high|xhigh|max`)与提供商无关;各提供商在请求层
142
180
  把它映射到自己的档位(例如 DeepSeek 只区分 `high|max`,`xhigh` 会收敛为
@@ -220,8 +258,8 @@ sqlite-vec + FTS5 索引;没有 Ollama 则平滑退化为关键词检索。删
220
258
  有界的只读调查,只拿回带引用、经机械校验的报告;搜索噪声绝不进入你的会话。
221
259
  它同样为 goal 模式的分支评审与里程碑验证提供依据。
222
260
  - **DeepSeek 以外的提供商。** MiniMax、GLM / z.ai、Kimi 都以 OpenAI 兼容的
223
- profile 接入(`/model`),但只有 DeepSeek 在 harness 上验证过 —— 其余在拿到
224
- bench 分数前,请当作未验证。
261
+ profile 接入(`/model`)。DeepSeek 与 MiniMax 已有完整 bench 成绩(见
262
+ 「能力量化」);GLM 与 Kimi 在拿到 bench 分数前,请当作未验证。
225
263
 
226
264
  ## 复用你已有的配置
227
265
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "rockycode"
3
- version = "0.1.0"
3
+ version = "0.1.2"
4
4
  description = "A coding agent harness, benchmarked on SWE-bench Verified. amaze!"
5
5
  requires-python = ">=3.11"
6
6
  readme = "README.md"
@@ -68,6 +68,13 @@ pdf = [
68
68
  # uv pip install 'rockycode[keyring]'
69
69
  keyring = ["keyring>=24"]
70
70
 
71
+ [dependency-groups]
72
+ # Development-only quality gate. Pin it so contributors and CI enforce the
73
+ # exact same rules; Ruff is not shipped in the rockycode runtime wheel.
74
+ dev = [
75
+ "ruff==0.16.1",
76
+ ]
77
+
71
78
  [project.scripts]
72
79
  rockycode = "rockycode.cli:app"
73
80
 
@@ -81,3 +88,14 @@ packages = ["rockycode"]
81
88
  [tool.ruff]
82
89
  line-length = 110
83
90
  target-version = "py311"
91
+
92
+ [tool.ruff.lint]
93
+ # Conservative correctness baseline: import/syntax errors, undefined names,
94
+ # unused imports/variables, and a small set of unambiguous PEP 8 errors.
95
+ select = ["E4", "E7", "E9", "F"]
96
+
97
+ [tool.ruff.lint.per-file-ignores]
98
+ # Smoke tests are executable scripts: many intentionally set cwd/env before
99
+ # importing rockycode, and some use compact procedural assertions. Keep F/E9
100
+ # correctness checks active without forcing a wholesale test rewrite.
101
+ "tests/*.py" = ["E402", "E702", "E741"]
@@ -0,0 +1,11 @@
1
+ # Version comes from installed package metadata (single source: pyproject) —
2
+ # a hardcoded string here shipped stale once ("0.1.0" inside the 0.1.1 wheel).
3
+ # Lazy module __getattr__ keeps `import rockycode` free of the metadata scan.
4
+ def __getattr__(name: str):
5
+ if name == "__version__":
6
+ from importlib.metadata import PackageNotFoundError, version
7
+ try:
8
+ return version("rockycode")
9
+ except PackageNotFoundError: # running from a bare checkout, uninstalled
10
+ return "unknown"
11
+ raise AttributeError(name)