rockycode 0.1.1__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (238) hide show
  1. rockycode-0.2.0/.github/workflows/release.yml +27 -0
  2. rockycode-0.2.0/CHANGELOG.md +209 -0
  3. {rockycode-0.1.1 → rockycode-0.2.0}/PKG-INFO +130 -50
  4. {rockycode-0.1.1 → rockycode-0.2.0}/README.md +128 -48
  5. {rockycode-0.1.1 → rockycode-0.2.0}/README.zh-CN.md +105 -36
  6. {rockycode-0.1.1 → rockycode-0.2.0}/pyproject.toml +1 -1
  7. rockycode-0.2.0/rockycode/__init__.py +11 -0
  8. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/cli.py +126 -49
  9. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/config.py +36 -14
  10. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/compaction.py +17 -5
  11. rockycode-0.2.0/rockycode/engine/cron.py +527 -0
  12. rockycode-0.2.0/rockycode/engine/effort.py +122 -0
  13. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/events.py +15 -2
  14. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/headless.py +68 -11
  15. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/images.py +57 -4
  16. rockycode-0.2.0/rockycode/engine/localcheck.py +166 -0
  17. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/loop.py +289 -43
  18. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/permission.py +16 -0
  19. rockycode-0.2.0/rockycode/engine/providers.py +751 -0
  20. rockycode-0.2.0/rockycode/engine/selfconfig.py +164 -0
  21. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/server.py +4 -3
  22. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/titler.py +26 -4
  23. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/tools.py +35 -13
  24. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/vision.py +27 -14
  25. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/web.py +7 -1
  26. rockycode-0.2.0/rockycode/models.toml +239 -0
  27. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/onboarding.py +12 -7
  28. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/pricing.py +84 -54
  29. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/prompts/rocky.py +3 -0
  30. rockycode-0.2.0/rockycode/skills/rocky-setup/SKILL.md +57 -0
  31. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/app.py +737 -57
  32. rockycode-0.2.0/rockycode/tui/loopcard.py +135 -0
  33. rockycode-0.2.0/rockycode/tui/modelpicker.py +246 -0
  34. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/permission.py +108 -0
  35. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/package.json +1 -1
  36. rockycode-0.2.0/tests/smoke_cron.py +388 -0
  37. rockycode-0.2.0/tests/smoke_effort.py +61 -0
  38. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_engine.py +50 -0
  39. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_exec.py +47 -0
  40. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_images.py +54 -10
  41. rockycode-0.2.0/tests/smoke_local_models.py +260 -0
  42. rockycode-0.2.0/tests/smoke_models_registry.py +152 -0
  43. rockycode-0.2.0/tests/smoke_pricing.py +117 -0
  44. rockycode-0.2.0/tests/smoke_providers.py +237 -0
  45. rockycode-0.2.0/tests/smoke_selfconfig.py +85 -0
  46. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_serve.py +16 -2
  47. rockycode-0.2.0/tests/smoke_tui_loop.py +588 -0
  48. rockycode-0.2.0/tests/smoke_tui_model.py +130 -0
  49. rockycode-0.2.0/tests/smoke_tui_modeswitch.py +173 -0
  50. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_permission.py +1 -1
  51. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_toggle.py +1 -1
  52. {rockycode-0.1.1 → rockycode-0.2.0}/uv.lock +1 -1
  53. rockycode-0.1.1/.github/workflows/release.yml +0 -73
  54. rockycode-0.1.1/CHANGELOG.md +0 -42
  55. rockycode-0.1.1/rockycode/__init__.py +0 -1
  56. rockycode-0.1.1/rockycode/engine/effort.py +0 -46
  57. rockycode-0.1.1/rockycode/engine/providers.py +0 -209
  58. rockycode-0.1.1/rockycode/tui/modelpicker.py +0 -116
  59. rockycode-0.1.1/tests/smoke_effort.py +0 -45
  60. rockycode-0.1.1/tests/smoke_pricing.py +0 -95
  61. rockycode-0.1.1/tests/smoke_providers.py +0 -120
  62. rockycode-0.1.1/tests/smoke_tui_model.py +0 -81
  63. {rockycode-0.1.1 → rockycode-0.2.0}/.dockerignore +0 -0
  64. {rockycode-0.1.1 → rockycode-0.2.0}/.github/workflows/ci.yml +0 -0
  65. {rockycode-0.1.1 → rockycode-0.2.0}/.gitignore +0 -0
  66. {rockycode-0.1.1 → rockycode-0.2.0}/CONTRIBUTING.md +0 -0
  67. {rockycode-0.1.1 → rockycode-0.2.0}/Dockerfile +0 -0
  68. {rockycode-0.1.1 → rockycode-0.2.0}/Dockerfile.sandbox +0 -0
  69. {rockycode-0.1.1 → rockycode-0.2.0}/LICENSE +0 -0
  70. {rockycode-0.1.1 → rockycode-0.2.0}/SECURITY.md +0 -0
  71. {rockycode-0.1.1 → rockycode-0.2.0}/bench/tasks/dev10.json +0 -0
  72. {rockycode-0.1.1 → rockycode-0.2.0}/bench/tasks/test20.json +0 -0
  73. {rockycode-0.1.1 → rockycode-0.2.0}/brand/rockycode-note.svg +0 -0
  74. {rockycode-0.1.1 → rockycode-0.2.0}/brand/rockycode-wordmark.svg +0 -0
  75. {rockycode-0.1.1 → rockycode-0.2.0}/docker-compose.yml +0 -0
  76. {rockycode-0.1.1 → rockycode-0.2.0}/prompts/README.md +0 -0
  77. {rockycode-0.1.1 → rockycode-0.2.0}/prompts/rocky-v1.txt +0 -0
  78. {rockycode-0.1.1 → rockycode-0.2.0}/prompts/rocky-v2-search-first.txt +0 -0
  79. {rockycode-0.1.1 → rockycode-0.2.0}/prompts/rocky-v3-decisive.txt +0 -0
  80. {rockycode-0.1.1 → rockycode-0.2.0}/prompts/rocky-zh-closer.txt +0 -0
  81. {rockycode-0.1.1 → rockycode-0.2.0}/prompts/rocky-zh-full-closer.txt +0 -0
  82. {rockycode-0.1.1 → rockycode-0.2.0}/prompts/rocky-zh-full.txt +0 -0
  83. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/banner.py +0 -0
  84. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/dream/__init__.py +0 -0
  85. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/dream/core.py +0 -0
  86. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/dream/judge.py +0 -0
  87. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/dream/mining.py +0 -0
  88. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/dream/proposals.py +0 -0
  89. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/__init__.py +0 -0
  90. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/artifact.py +0 -0
  91. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/budget.py +0 -0
  92. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/checks.py +0 -0
  93. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/container.py +0 -0
  94. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/explore.py +0 -0
  95. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/goal.py +0 -0
  96. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/goal_review.py +0 -0
  97. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/goal_session.py +0 -0
  98. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/lsp.py +0 -0
  99. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/mcp.py +0 -0
  100. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/modes.py +0 -0
  101. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/outcome.py +0 -0
  102. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/planmode.py +0 -0
  103. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/redact.py +0 -0
  104. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/safety.py +0 -0
  105. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/sandbox.py +0 -0
  106. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/skills.py +0 -0
  107. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/trajectory.py +0 -0
  108. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/engine/worktree.py +0 -0
  109. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/memory/__init__.py +0 -0
  110. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/memory/index.py +0 -0
  111. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/memory/store.py +0 -0
  112. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/modes/learn/learn.md +0 -0
  113. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/modes/research/deep-research.md +0 -0
  114. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/modes/research/paper-reading.md +0 -0
  115. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/modes/research/prove.md +0 -0
  116. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/modes/research/whiteboard.md +0 -0
  117. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/palette.py +0 -0
  118. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/prompts/__init__.py +0 -0
  119. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/routines.py +0 -0
  120. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/runners/__init__.py +0 -0
  121. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/runners/agent.py +0 -0
  122. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/runners/data.py +0 -0
  123. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/runners/raw.py +0 -0
  124. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/score.py +0 -0
  125. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/session.py +0 -0
  126. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/skills/architecture-viz/SKILL.md +0 -0
  127. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/skills/architecture-viz/template.html +0 -0
  128. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/skills/lean-prover/SKILL.md +0 -0
  129. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/skills/lean-prover/torchlean-api.md +0 -0
  130. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/__init__.py +0 -0
  131. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/clipboard.py +0 -0
  132. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/exitsheet.py +0 -0
  133. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/goal_screen.py +0 -0
  134. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/mdterm.py +0 -0
  135. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/mdview.py +0 -0
  136. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/modepicker.py +0 -0
  137. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/plangate.py +0 -0
  138. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/prompt_history.py +0 -0
  139. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/proposalcard.py +0 -0
  140. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/resume.py +0 -0
  141. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/rocky_pet.py +0 -0
  142. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode/tui/routinecard.py +0 -0
  143. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/.gitignore +0 -0
  144. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/.vscode/launch.json +0 -0
  145. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/.vscode/tasks.json +0 -0
  146. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/.vscodeignore +0 -0
  147. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/CHANGELOG.md +0 -0
  148. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/LICENSE +0 -0
  149. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/README.md +0 -0
  150. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/esbuild.config.mjs +0 -0
  151. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/media/chat.html +0 -0
  152. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/media/icon-marketplace.svg +0 -0
  153. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/media/icon.png +0 -0
  154. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/media/marked.js +0 -0
  155. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/media/rocky-icon.svg +0 -0
  156. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/package-lock.json +0 -0
  157. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/artifactTree.ts +0 -0
  158. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/diffManager.ts +0 -0
  159. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/editorContext.ts +0 -0
  160. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/extension.ts +0 -0
  161. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/permissionManager.ts +0 -0
  162. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/protocol.ts +0 -0
  163. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/rockyConnection.ts +0 -0
  164. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/rockyProvider.ts +0 -0
  165. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/statusBar.ts +0 -0
  166. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/src/webview-highlight.ts +0 -0
  167. {rockycode-0.1.1 → rockycode-0.2.0}/rockycode-vscode/tsconfig.json +0 -0
  168. {rockycode-0.1.1 → rockycode-0.2.0}/tests/fake_lsp_server.py +0 -0
  169. {rockycode-0.1.1 → rockycode-0.2.0}/tests/fake_mcp_server.py +0 -0
  170. {rockycode-0.1.1 → rockycode-0.2.0}/tests/invariants.py +0 -0
  171. {rockycode-0.1.1 → rockycode-0.2.0}/tests/real_planmode.py +0 -0
  172. {rockycode-0.1.1 → rockycode-0.2.0}/tests/real_reasoning_roundtrip.py +0 -0
  173. {rockycode-0.1.1 → rockycode-0.2.0}/tests/real_tool_contract.py +0 -0
  174. {rockycode-0.1.1 → rockycode-0.2.0}/tests/run_all.py +0 -0
  175. {rockycode-0.1.1 → rockycode-0.2.0}/tests/run_real.py +0 -0
  176. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_approver.py +0 -0
  177. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_artifact.py +0 -0
  178. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_bash.py +0 -0
  179. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_bash_grant.py +0 -0
  180. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_bilang_prompt.py +0 -0
  181. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_budget.py +0 -0
  182. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_checks.py +0 -0
  183. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_compaction.py +0 -0
  184. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_container.py +0 -0
  185. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_credentials.py +0 -0
  186. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_docker_sandbox_cancel.py +0 -0
  187. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_dream.py +0 -0
  188. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_dream_trigger.py +0 -0
  189. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_explore.py +0 -0
  190. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_goal.py +0 -0
  191. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_goal_driver.py +0 -0
  192. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_goal_review.py +0 -0
  193. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_goal_runner.py +0 -0
  194. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_hardkill_resume.py +0 -0
  195. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_interrupt.py +0 -0
  196. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_judge.py +0 -0
  197. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_lsp.py +0 -0
  198. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_mcp.py +0 -0
  199. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_mdterm.py +0 -0
  200. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_mdview.py +0 -0
  201. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_memory.py +0 -0
  202. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_memory_index.py +0 -0
  203. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_mining.py +0 -0
  204. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_modes.py +0 -0
  205. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_onboarding.py +0 -0
  206. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_outcome.py +0 -0
  207. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_parallel.py +0 -0
  208. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_permission.py +0 -0
  209. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_planmode.py +0 -0
  210. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_prompt_history.py +0 -0
  211. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_proposals.py +0 -0
  212. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_redact.py +0 -0
  213. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_resume_handoff.py +0 -0
  214. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_routine_run.py +0 -0
  215. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_routines.py +0 -0
  216. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_safety.py +0 -0
  217. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_score.py +0 -0
  218. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_session.py +0 -0
  219. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_skills.py +0 -0
  220. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_submit_race.py +0 -0
  221. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_today.py +0 -0
  222. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tools.py +0 -0
  223. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tools_jail.py +0 -0
  224. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_trajectory.py +0 -0
  225. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_artifact_modal.py +0 -0
  226. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_bash_gate.py +0 -0
  227. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_copy.py +0 -0
  228. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_envwarn.py +0 -0
  229. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_exitsheet.py +0 -0
  230. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_goal_screen.py +0 -0
  231. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_input_nav.py +0 -0
  232. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_plan.py +0 -0
  233. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_proposals.py +0 -0
  234. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_read_grant.py +0 -0
  235. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_routines.py +0 -0
  236. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_scroll.py +0 -0
  237. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_tui_shell.py +0 -0
  238. {rockycode-0.1.1 → rockycode-0.2.0}/tests/smoke_web.py +0 -0
@@ -0,0 +1,27 @@
1
+ name: Release
2
+
3
+ # One job: build sdist+wheel on ubuntu, publish to PyPI via Trusted Publishing.
4
+ # The old macos-13 matrix (built a cbor2 x86_64-mac wheel for the release page)
5
+ # is gone: the runner label was retired, so it never actually ran — 0.1.0 and
6
+ # 0.1.1 both sat in its queue until cancelled and shipped fine without it — and
7
+ # cbor2 still publishes no x86_64-mac wheel to attach anyway (arm64 only; an
8
+ # Intel-Mac install builds it from source via Rust, PyPI-side, as it always
9
+ # has). GitHub releases are created by hand: gh release create --notes-file.
10
+ on:
11
+ push:
12
+ tags: ["v*"]
13
+ workflow_dispatch:
14
+
15
+ jobs:
16
+ pypi:
17
+ runs-on: ubuntu-latest
18
+ permissions:
19
+ id-token: write # OIDC — this is what Trusted Publishing uses
20
+ steps:
21
+ - uses: actions/checkout@v4
22
+ - name: Setup uv
23
+ uses: astral-sh/setup-uv@v5
24
+ - name: Build rockycode (sdist + wheel — only our project lands in dist/)
25
+ run: uv build
26
+ - name: Publish to PyPI (Trusted Publishing, no token)
27
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,209 @@
1
+ # Changelog
2
+
3
+ All notable changes to rockycode are recorded here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/), and the project follows
5
+ [Semantic Versioning](https://semver.org/) — pre-1.0, so the surface may still
6
+ change between minor versions.
7
+
8
+ ## [Unreleased]
9
+
10
+ ## [0.2.0] — models are data, loops, approval switch
11
+
12
+ ### Added
13
+ - **Loops** (`/loop`, alias `/cron`): re-fire one prompt into THIS chat on an
14
+ interval — `/loop 5m check whether results/run3 has a summary.json; if so
15
+ give me the headline numbers`. Each tick is an ordinary turn on the session
16
+ engine (it sees the whole conversation), runs only when the chat is idle,
17
+ ends with `LOOP QUIET / NOTE / DONE`, and never prompts: a call that needs
18
+ approval pauses the loop (`/loop allow` decides it, `/loop resume` retries,
19
+ `/permission yolo` or shift+tab resumes every approval-paused loop; a manual
20
+ `/loop pause` stays paused). Quiet ticks fold into one streak line and roll
21
+ out of live context (the trajectory keeps them). No cap unless you set one
22
+ (`for 3h`, `x12`, `max $2`); soft running-cost reminders otherwise. Rocky
23
+ can start one itself (`loop_start` / `loop_stop`, through the normal
24
+ approval). Session-only — a loop dies with the session; `/routines` stays
25
+ the cross-launch scheduler.
26
+ - **Models are data.** The whole model catalog now lives in
27
+ `rockycode/models.toml` — per provider a China base URL, a rocky-owned key
28
+ name, a reasoning wire shape (`thinking` · `effort` · `enable_thinking` ·
29
+ `minimax` · `openai` · `none`), its own effort tiers, whether thinking can
30
+ be switched off, and where usage reports cache hits; per model its context
31
+ window, output cap, vision flag, price tables (USD + CNY) and roles. The
32
+ engine consumes a `ModelSpec` and carries no model-specific numbers of its
33
+ own. `~/.rockycode/models.toml` (same shape) is deep-merged on top: add a
34
+ model, correct a limit, add a price, `hidden = true` to drop a row.
35
+ - The 2026-09 roster: `deepseek-flash` (V4.1 Flash — native vision, the
36
+ default and the sidecar/search model) + `deepseek-v4-pro`; `glm-5.3` +
37
+ `glm-5.3-flash`; `kimi-k3`; `minimax-m3`; `step-5-preview`; `qwen3.8-max`
38
+ + `qwen3.8-flash`; `mimo-v2.6-pro`; `ollama` (live-discovered). Limits and
39
+ DeepSeek's V4.1 prices verified at each provider's docs on 2026-09-29.
40
+ - Subscription-plan endpoints as their own picker rows: `qwen-plan` (Bailian
41
+ Token Plan), `mimo-plan`, `stepfun-plan` — own URL, own key
42
+ (`ROCKYCODE_<PROVIDER>_PLAN_API_KEY`).
43
+ - Context window and output cap follow the active model: config
44
+ `context_window` / `max_tokens` default to `0` (= the registry value) and
45
+ re-pace on every `/model` switch; a positive number pins your own ceiling.
46
+ `/config context_window 0` un-pins.
47
+ - `/effort off|low|high|max` — the dial now reaches `low` (DeepSeek, GLM and
48
+ Kimi all take low|high|max), is clamped onto each provider's own tiers by
49
+ position at the wire (StepFun's `low|medium|high` gets `medium` for
50
+ rocky's `high`), and sends the lowest tier for `off` on a model that
51
+ can't disable thinking (GLM, Kimi). The status line says what is sent.
52
+ `xhigh` stays accepted as `max`.
53
+ - `image_route auto` (new default): a pasted image on a text-only model is
54
+ described silently by the registry's sidecar (`deepseek-flash` on the home
55
+ key) or your image CLI — no picker in the way; `ask` keeps the picker.
56
+ - **Rocky configures rocky.** A built-in `rocky-setup` skill plus the
57
+ ask-tier `rocky_config` tool (show · set · set_url · add_provider) let a
58
+ user ask rocky to use a proxy, add a vLLM box, change the default model or
59
+ a limit — writes go under `~/.rockycode` only, through the same validated
60
+ setters as the CLI. Key-shaped values are refused with the env var to use.
61
+ - `rockycode exec --profile read|write|full`: `read` (read_file/grep/glob/
62
+ view_image) and `write` (+ jailed write_file/edit_file) have no shell, so
63
+ they run on the host with no Docker and start instantly — the profiles a
64
+ calling agent (Claude Code, Codex) wants for "look and tell" and small
65
+ edits. `full` keeps bash in the Docker sandbox by default. The stream is
66
+ now meta → text → result by default; `--events` restores the tool.*/turn.*
67
+ receipt lines (`--include-thinking` implies it). `meta.profile` reports
68
+ name, mode and tool set.
69
+ - Peak-hour pricing honors DeepSeek's calendar: Monday–Friday only, with a
70
+ `holidays` list (ISO dates, Beijing) on the schedule for Chinese public
71
+ holidays — weekends and holidays bill off-peak.
72
+ - Approval mode is switchable without typing: `shift+tab` steps it one notch
73
+ looser (careful → ask → yolo, wrapping back to careful, so a run of presses
74
+ never parks you in yolo). The status-bar 🔒 chip now spells that key out and
75
+ opens a picker when clicked — three rows saying what each mode actually
76
+ allows — and bare `/permission` opens the same picker. The cycle stays inert
77
+ while an approval prompt is waiting, and landing in yolo still prints the
78
+ "runs on your machine" warning.
79
+ - Local models: a builtin `ollama` provider (`http://localhost:11434/v1`,
80
+ override with `ROCKYCODE_OLLAMA_URL`) — keyless, priced `$0 · local`, model
81
+ list discovered live from the running server's `GET /v1/models` so the
82
+ `/model` picker shows what's actually pulled. Server down → a dimmed
83
+ "not running · start it: ollama serve" row instead of silence.
84
+ - `/model` readiness preflight for local providers: before the engine is
85
+ switched, rocky checks server reachability, that the model is pulled, that
86
+ it supports tool calling, and the serving context length (Ollama truncates
87
+ silently — the #1 agent-loop killer). Not ready → the switch is refused and
88
+ every failed check carries its exact fix (`ollama pull …`,
89
+ `export OLLAMA_CONTEXT_LENGTH=65536`); ready → rocky's `context_window` is
90
+ paced to the server's verified value (restored on switching back to a cloud
91
+ provider), and a server-reported vision capability turns image input on.
92
+ - Keyless endpoints: `local = true` on a `~/.rockycode/providers.toml`
93
+ provider (LM Studio, llama.cpp, vLLM, a remote box's ollama) marks its
94
+ endpoints keyless — always configured, no placeholder-key hack needed.
95
+ - Compaction on a local provider sends no tool schemas in the summarize call
96
+ (local compat layers ignore `tool_choice="none"` and may answer with a tool
97
+ call instead of a summary), and the thinking-off field is now shaped per
98
+ the provider's reasoning policy instead of always DeepSeek's.
99
+ - Vision is now per-MODEL, not per-provider (`Provider.vision_models`,
100
+ choice-level `❖` badge). Config key `vision_models` marks additional ids as
101
+ image-capable from any shell (`rockycode config vision_models <id>`) — for
102
+ when a provider ships vision on an existing model before the registry
103
+ catches up.
104
+ - Config key `model`: a sticky launch default (`rockycode config model
105
+ <spec>`), resolved against the registry; precedence `--model` flag →
106
+ `ROCKYCODE_MODEL` env → config. Global config only — a cloned repo can
107
+ never redirect requests.
108
+ - The `/model` picker is now two-step, model first: one row per model, and a
109
+ model with several endpoints then asks which URL serves it — including a
110
+ "custom base URL" row that remembers your own gateway/proxy per provider
111
+ (`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`, riding
112
+ the provider's key).
113
+ - Images are best-effort downscaled to 2048px before hitting the wire (PIL if
114
+ installed, macOS `sips` otherwise, original on any failure) — vision
115
+ providers bill image tokens by dimensions, and a Retina screenshot was
116
+ paying severalfold for nothing.
117
+
118
+ ### Changed
119
+ - One China endpoint per provider (no more `kimi-cn`/`kimi-en`, `zai`/`glm-cn`,
120
+ `minimax-cn`/`-en`): `/model kimi`, `/model glm`, `/model minimax`. Keys are
121
+ `ROCKYCODE_<PROVIDER>_API_KEY`; the older `_CN_`/`_EN_`/`ZAI_EN` names are
122
+ still read as aliases (the picker names the alias in use), so no setup
123
+ breaks. An international or proxy URL is the custom-URL row.
124
+ - `deepseek-v4-flash` and `deepseek-v4-flash-vision-exp` are retired from
125
+ the list (DeepSeek retired the models on 2026-09-10 and serves those names
126
+ with V4.1-Flash); rocky aliases both to `deepseek-flash`, so old configs,
127
+ trajectories and `/model` habits keep working. `deepseek-v4-pro` stays
128
+ listed, labelled: since 2026-09-14 DeepSeek routes it to V4.1-Flash at
129
+ Flash rates until V4.1-Pro ships.
130
+ - `glm-5.2` and `step-3.7-flash` drop off the roster in favor of GLM-5.3 /
131
+ Step 5 Preview.
132
+ - Session titles, image describe, and the native web search all use the
133
+ registry's `sidecar` / `search` role (deepseek-flash) — a session on Kimi
134
+ or GLM no longer sends a DeepSeek model id down its own client for the
135
+ title call.
136
+ - `--max-tokens` / `--context-window` default to `0` (= the model's registry
137
+ value) on `chat` and `serve`; `bench` keeps its pinned reproducible numbers.
138
+ - The CLI help, status line and context reminder no longer speak of
139
+ "DeepSeek V4" as the only model.
140
+ - `view_image` on a vision-capable active model now attaches the real image
141
+ to the conversation (as the next user message) instead of a sidecar text
142
+ description — the model reads the pixels and decides what matters itself.
143
+ Text-only models keep the describe routes unchanged.
144
+ - Launch honors the registry: starting with `--model minimax-m3` (or any
145
+ registry model) now gets the right reasoning params, tools flag, and vision
146
+ capability from step one instead of DeepSeek-shaped defaults until the
147
+ first `/model` switch. Unresolved specs and injected clients behave exactly
148
+ as before.
149
+ - `/model` spec resolution prefers exact ids, and a base model wins its own
150
+ substring — `glm:5.3` means `glm-5.3`, not ambiguity with `glm-5.3-flash`;
151
+ the variant stays reachable via `glm:flash`.
152
+
153
+ ### Fixed
154
+ - The prompt-cache observer and the cost ledger read cache hits from
155
+ OpenAI-style `prompt_tokens_details.cached_tokens` too (Kimi, GLM, MiniMax,
156
+ Qwen, MiMo), not only DeepSeek's `prompt_cache_hit_tokens`.
157
+ - Launching on a `<provider>-custom` endpoint of the home provider now uses
158
+ that URL (it used to fall back to the env base URL).
159
+
160
+ ## [0.1.2] — `--version` flag, GA price tables
161
+
162
+ ### Added
163
+ - `rockycode --version` / `-V` prints the installed version. One version
164
+ source: package metadata (pyproject) — the serve handshake reports the same
165
+ value instead of a hardcoded string.
166
+
167
+ ### Changed
168
+ - DeepSeek price tables refreshed to the GA snapshots (V4-Flash-0731 /
169
+ V4-Pro-0813), both USD and CNY, verified 2026-08-20 at the source;
170
+ peak-valley billing confirmed live (2× in the published UTC windows).
171
+ - README results updated: `deepseek-v4-flash` GA carries three clean full-500
172
+ SWE-bench Verified rounds (88.8% average, 95.4% pass@3); the
173
+ `deepseek-v4-pro` column is explicitly marked preview (pre-0813).
174
+
175
+ ### CI
176
+ - Release workflow reduced to the single ubuntu PyPI Trusted-Publishing job;
177
+ the dead macos-13 matrix (retired runner, never ran) is gone.
178
+
179
+ ## [0.1.1] — session artifacts, images in chat, real sandbox cancel
180
+
181
+ ### Added
182
+ - Images in chat: paste (`ctrl+v` / `/paste`) or drag an image in. Vision models
183
+ see it raw; no-vision models route through a provider sidecar, your own image
184
+ CLI, or a `view_image` tool — picked once, remembered. Known CLIs set up with
185
+ one word (`/config image_cli mmx` — rocky knows the invocation).
186
+ - Session artifact inventory: `/artifact list · open <n> · stop · live on|off`,
187
+ a footer badge with open-tab counts, and an Artifacts tree in the VS Code
188
+ extension fed live by `rockycode serve`.
189
+ - Bare `/model` opens a live provider + model picker.
190
+
191
+ ### Fixed
192
+ - Live artifacts no longer drop and reconnect every 30 s; the artifact server
193
+ stops/restarts cleanly and rebinds saved live pages to the new port.
194
+ - Sandbox cancel/timeout kills the in-container process group, not just the
195
+ host-side docker client; images without `python3` fall back to plain `bash -c`.
196
+ - Resume self-heals after a hard kill; closing the doc dock no longer wedges the TUI.
197
+
198
+ ### Internal
199
+ - CI: conservative ruff correctness gate (pinned `0.16.1`) ahead of the smoke suite.
200
+
201
+ ## [0.1.0] — first public release
202
+
203
+ Initial public release. One repo, one engine, three ways to use it — interactive
204
+ `chat`, autonomous `goal`, and the `bench` measurement rig — running on DeepSeek
205
+ or any OpenAI-compatible model.
206
+
207
+ Some capabilities ship as **experimental and default-off** (self-improvement,
208
+ `prove` / `lean-prover`, `explore`, and providers other than DeepSeek); see the
209
+ README's Experimental section for what they are and how to enable them.
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: rockycode
3
- Version: 0.1.1
3
+ Version: 0.2.0
4
4
  Summary: A coding agent harness, benchmarked on SWE-bench Verified. amaze!
5
5
  Author: rockycode contributors
6
6
  License: MIT
@@ -42,8 +42,8 @@ Built for the DeepSeek V4 series, with a unique research mode, bench-tested, and
42
42
 
43
43
  [English](README.md) · [简体中文](README.zh-CN.md)
44
44
 
45
- ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-79.8%25_V4--flash-7d5cc6)
46
- ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-81.8%25_pass@3-8d6cd0)
45
+ ![SWE-bench Verified](https://img.shields.io/badge/SWE--bench_Verified-88.8%25_V4--flash_GA-7d5cc6)
46
+ ![V4-pro preview](https://img.shields.io/badge/V4--pro_preview-74.9%25-8d6cd0)
47
47
  ![Python](https://img.shields.io/badge/Python-3.11%2B-9d7cd8)
48
48
  ![License](https://img.shields.io/badge/License-MIT-a9b1d6)
49
49
 
@@ -77,31 +77,37 @@ against it.
77
77
 
78
78
  ## Results — SWE-bench Verified
79
79
 
80
- Full-set numbers: **independent full-500 runs** (three each for
81
- `deepseek-v4-pro` and `minimax-m3`, one so far for `deepseek-v4-flash`), same
80
+ Full-set numbers: **independent full-500 runs**, three rounds per model, same
82
81
  harness and config for all (100-step cap, 32,768 max output tokens, reasoning
83
82
  effort `max`, thinking on), scored with the official SWE-bench harness. No
84
- tuning against the tasks.
83
+ tuning against the tasks. Mind the versions: the `deepseek-v4-flash` column is
84
+ the **GA release** (V4-Flash-0731), while the `deepseek-v4-pro` rounds ran
85
+ before the GA V4-Pro-0813 shipped — that column is the **preview** pro, so the
86
+ flash/pro gap reflects a version difference, not a same-vintage comparison.
85
87
 
86
- | run | `deepseek-v4-pro` | `deepseek-v4-flash` | `minimax-m3` |
88
+ | run | `deepseek-v4-flash` (GA) | `deepseek-v4-pro` (preview) | `minimax-m3` |
87
89
  |---|---|---|---|
88
- | round 1 | 75.6% (378/500) | 79.8% (399/500) | 72.8% (364/500) |
89
- | round 2 | 74.8% (374/500) | — | 71.2% (356/500) |
90
- | round 3 | 74.4% (372/500) | — | 70.0% (350/500) |
91
- | **average** | **74.9%** | **79.8%** *(1 run)* | **71.3%** |
92
- | union of runs (pass@3) | 81.8% (409/500) | — | 83.6% (418/500) |
90
+ | round 1 | 90.0% (450/500) | 75.6% (378/500) | 72.8% (364/500) |
91
+ | round 2 | 88.6% (443/500) | 74.8% (374/500) | 71.2% (356/500) |
92
+ | round 3 | 87.8% (439/500) | 74.4% (372/500) | 70.0% (350/500) |
93
+ | **average** | **88.8%** | **74.9%** | **71.3%** |
94
+ | union of runs (pass@3) | 95.4% (477/500) | 81.8% (409/500) | 83.6% (418/500) |
93
95
 
94
96
  Read the two summary rows differently. The **average** is the
95
97
  leaderboard-comparable number — each round is an independent single-pass run
96
98
  over the full 500. The **union** is pass@3: tasks solved by at least one
97
- round. The gap between them (~7 points for DeepSeek, ~12 for MiniMax) is
98
- run-to-run variance, not capability — the models already reach these tasks
99
- under this harness, they just don't hold them every run. Closing that gap
100
- (verify-before-finish gating and run selection, not more prompting) is the
101
- current line of work. For reference, DeepSeek reports 80.6% with its own
102
- scaffold; the three-run union crosses that mark. Per-round breakdowns:
99
+ round. The gap between them (~7 points for both DeepSeek models, ~12 for
100
+ MiniMax) is run-to-run variance, not capability — the models already reach
101
+ these tasks under this harness, they just don't hold them every run. Closing
102
+ that gap (verify-before-finish gating and run selection, not more prompting)
103
+ is the current line of work. For reference, DeepSeek reported 80.6% for V4
104
+ Pro (Preview) with its own scaffold. Per-round breakdowns:
103
105
  [@rockycode_ai](https://x.com/rockycode_ai).
104
106
 
107
+ Two earlier flash rounds (79.8% — previously listed here — and 81.2%) hit
108
+ local network outages mid-run, visible as contiguous blocks of empty-patch
109
+ tasks; they were replaced by clean re-runs rather than averaged in.
110
+
105
111
  We plan to add **DeepSWE-bench** support as well — currently in progress.
106
112
 
107
113
  ## Getting started
@@ -121,6 +127,10 @@ uv tool install rockycode # recommended — puts the `rockycode` command on
121
127
  rockycode # the first run walks you through API-key setup
122
128
  ```
123
129
 
130
+ Already installed and a new release is out? **`uv tool upgrade rockycode`** —
131
+ note that re-running `uv tool install` does NOT upgrade: it sees the existing
132
+ install and quietly keeps the old version.
133
+
124
134
  Don't have uv yet? One command installs it:
125
135
  `curl -LsSf https://astral.sh/uv/install.sh | sh` — Windows and other options
126
136
  in the [uv install docs](https://docs.astral.sh/uv/getting-started/installation/).
@@ -195,8 +205,8 @@ clipboard" (or your terminal's equivalent) on the local end.
195
205
  | `/research` | Research modes: deep-research · paper-reading · whiteboard · prove |
196
206
  | `/learn` | Tutor mode — your understanding is the goal, not the diff |
197
207
  | `/model` | Switch provider and model (see below) |
198
- | `/effort off\|high\|xhigh\|max` | Reasoning depth, adjustable live per session |
199
- | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
208
+ | `/effort off\|low\|high\|max` | Reasoning depth, adjustable live per session (clamped onto each provider's own tiers) |
209
+ | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session — bare opens a picker; `shift+tab` cycles it, or click the 🔒 chip in the status bar |
200
210
  | `/sandbox on\|off\|status` | Isolate tool execution in a container |
201
211
  | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
202
212
  | `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
@@ -228,27 +238,79 @@ clipboard" (or your terminal's equivalent) on the local end.
228
238
 
229
239
  ### Models and providers
230
240
 
231
- DeepSeek is the home model, but providers are data, not code: each is a
232
- base URL, a model list, a key variable, and a reasoning shape over the
233
- OpenAI-compatible API.
241
+ DeepSeek is the home model, but **models are data, not code**: the whole
242
+ catalog lives in `rockycode/models.toml` — per provider a China base URL, a
243
+ key name and a reasoning wire shape; per model its context window, output
244
+ cap, vision flag, price and roles. The engine reads that spec and carries no
245
+ model-specific numbers of its own, so a new model is a data edit. Your own
246
+ `~/.rockycode/models.toml` (same shape) is deep-merged on top: add a model,
247
+ correct a limit, add a price, hide a row.
234
248
 
235
- | Provider | Models |
236
- |---|---|
237
- | **deepseek** (default) | `deepseek-v4-flash` (default), `deepseek-v4-pro` (preview) |
238
- | **minimax** | `minimax-m3` |
239
- | **kimi** | `kimi-k3` |
240
- | **glm** | `glm-5.2` |
241
-
242
- Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
243
- and custom providers — including local vLLM/SGLang servers — go in
244
- `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
245
- keys are actually configured. DeepSeek and MiniMax both carry full-500 bench
246
- numbers (see [Results](#results--swe-bench-verified)); Kimi and GLM are
247
- [experimental](#experimental).
248
-
249
- The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
250
- provider maps it to its own reasoning tiers at the wire (DeepSeek, for
251
- example, only distinguishes `high|max`, so `xhigh` clamps to `max`).
249
+ | Provider | Models (❖ = takes image input) | ctx / max out |
250
+ |---|---|---|
251
+ | **deepseek** (default) | `deepseek-flash` (V4.1 Flash, default) ❖, `deepseek-v4-pro` | 1M / 384K |
252
+ | **glm** | `glm-5.3`, `glm-5.3-flash` ❖ | 1M / 128K |
253
+ | **kimi** | `kimi-k3` ❖ | 1M / 128K |
254
+ | **minimax** | `minimax-m3` ❖ | 1M / 128K |
255
+ | **stepfun** | `step-5-preview` ❖ | 1M / 64K |
256
+ | **qwen** | `qwen3.8-max` ❖, `qwen3.8-flash` ❖ | 1M / 64K |
257
+ | **mimo** | `mimo-v2.6-pro` ❖ | 1M / 128K |
258
+ | **ollama** (local, $0) | whatever you've pulled — discovered live from the running server | server-verified |
259
+
260
+ One China endpoint per provider (`ROCKYCODE_<PROVIDER>_API_KEY`; the older
261
+ `_CN_`/`_EN_` names are still read). Subscription plans with their own URL
262
+ and key are their own rows — `qwen-plan` (Bailian Token Plan), `mimo-plan`,
263
+ `stepfun-plan` — keyed as `ROCKYCODE_<PROVIDER>_PLAN_API_KEY`. The `/model`
264
+ picker lists **models first**, one row each; a model with several endpoints
265
+ then asks which URL serves it (official · plan · your own), and the "custom
266
+ base URL" row remembers a gateway or proxy per provider
267
+ (`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`). Typed
268
+ specs skip all of that: `/model glm:flash`, `/model qwen-plan:qwen3.8-max`.
269
+ The retired `deepseek-v4-flash` / `-vision-exp` ids still resolve (to
270
+ `deepseek-flash`, exactly as DeepSeek serves them). The picker only offers
271
+ providers whose keys are actually configured.
272
+
273
+ Vision is per-model: `deepseek-flash` sees images on the home key, so the
274
+ default session just takes a paste. A text-only model (`deepseek-v4-pro`,
275
+ `glm-5.3`) gets pasted images described by `deepseek-flash` silently
276
+ (`image_route auto`), or by your own CLI. `rockycode config model <spec>`
277
+ makes any pick the sticky launch default. Context window and output cap
278
+ follow the active model (config `context_window` / `max_tokens` = `0`); a
279
+ number pins your own ceiling across switches. DeepSeek and MiniMax carry
280
+ full-500 bench numbers (see [Results](#results--swe-bench-verified)); the
281
+ other providers are [experimental](#experimental).
282
+
283
+ The effort dial (`/effort off|low|high|max`) is provider-neutral; each
284
+ provider's own tiers come from the registry and the dial is clamped onto
285
+ them by position at the wire (StepFun's `low|medium|high` gets `medium` for
286
+ rocky's `high`; GLM and Kimi can't switch thinking off, so `off` sends their
287
+ lowest tier). `xhigh` is still accepted and means `max`.
288
+
289
+ Rocky can also configure itself: ask it to "use my proxy", "add my vLLM
290
+ server", or "switch the default model" and the built-in `rocky-setup` skill
291
+ plus the ask-tier `rocky_config` tool make the change under `~/.rockycode`.
292
+ Keys are the one thing it never touches — it names the variable and you
293
+ paste the value.
294
+
295
+ #### Local models (Ollama)
296
+
297
+ Rocky integrates the OpenAI-compatible *protocol*, never a runtime — and
298
+ recommends [Ollama](https://ollama.com) (its MLX engine covers Apple Silicon
299
+ since 0.19). No key, no config: run `ollama serve`, pull a tool-capable model
300
+ (`ollama pull qwen3.8:27b-mlx` is the tested recommendation), and it appears
301
+ in `/model` with what's actually pulled, priced `$0 · local`.
302
+
303
+ Switching to a local model runs a **readiness preflight** first — server up,
304
+ model pulled, tool-calling support, serving context — and refuses the switch
305
+ with the exact fix (`ollama pull …`, `export OLLAMA_CONTEXT_LENGTH=65536`)
306
+ when something would break mid-session. The one to respect: Ollama's default
307
+ context is small and it **truncates silently**, which kills agent sessions in
308
+ confusing ways — serve with `OLLAMA_CONTEXT_LENGTH=65536`. Rocky paces its own
309
+ `context_window` to the server's verified value on every switch.
310
+
311
+ Other local servers (LM Studio, llama.cpp, vLLM) work as data too: add a
312
+ provider with `local = true` in `~/.rockycode/providers.toml` and its
313
+ endpoints need no key.
252
314
 
253
315
  ## Autonomous use
254
316
 
@@ -278,11 +340,26 @@ rockycode goal "add a docstring to <fn> and run the linter" --max-usd 0.50 --max
278
340
  ### Headless delegation: `exec`
279
341
 
280
342
  `rockycode exec "<task>"` is the single-shot, non-interactive entry point,
281
- designed to be called by *other* agents and scripts. Events stream as JSONL on
282
- stdout; the Docker sandbox is **on by default** (the command classifier is
283
- defense-in-depth, not the boundary); budgets are always enforced; and exit
284
- codes distinguish success, failure, needs-approval, and budget-stop — so a
285
- calling agent can grant an approval and resume instead of guessing.
343
+ designed to be called by *other* agents and scripts. stdout is JSONL: a
344
+ `meta` line, the model's `text`, and a `result` envelope with evidence
345
+ (files changed, commands run, refusals) — never verdicts, the caller
346
+ verifies; `--events` adds the per-tool receipt lines. Budgets are always
347
+ enforced, and exit codes distinguish success, failure, needs-approval, and
348
+ budget-stop — so a calling agent can grant an approval and resume instead
349
+ of guessing.
350
+
351
+ Pick how much rocky may do with `--profile`: `read` (read_file / grep /
352
+ glob / view_image — no shell, no writes) and `write` (+ write_file /
353
+ edit_file jailed to `--workdir`) run on the host with **no Docker** and
354
+ start instantly — what Claude Code or Codex wants for "look at this repo and
355
+ tell me" or a small edit on a cheap, fast model. `full` adds bash, in the
356
+ Docker sandbox **by default** (the command classifier is defense-in-depth,
357
+ not the boundary).
358
+
359
+ ```bash
360
+ rockycode exec --profile read "which module owns retry logic, and where is it called?"
361
+ rockycode exec --profile write "add a docstring to every public function in utils.py"
362
+ ```
286
363
 
287
364
  ### Editor integration: `serve` and the VS Code extension
288
365
 
@@ -337,10 +414,13 @@ change. Anything that could act on its own is **off by default**.
337
414
  investigation from a fresh-context child that returns only a cited,
338
415
  mechanically-verified report; the search noise never enters your session. It
339
416
  also grounds goal mode's branch review and milestone verification.
340
- - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
341
- OpenAI-compatible profiles (`/model`). DeepSeek and MiniMax carry full
342
- bench numbers (see Results); treat GLM and Kimi as untested until they do
343
- too.
417
+ - **Providers beyond DeepSeek.** GLM, Kimi, MiniMax, StepFun, Qwen, and MiMo
418
+ are wired as OpenAI-compatible registry entries (`/model`). DeepSeek and
419
+ MiniMax carry full bench numbers (see Results); treat the rest as untested
420
+ until they do too. Registry entries marked `note = "… verify …"` in
421
+ `models.toml` (MiniMax's endpoint host, MiMo's auth header, the plan URLs)
422
+ were taken from each provider's docs on 2026-09-29 and not yet exercised
423
+ live — a wrong one is a one-line data fix.
344
424
 
345
425
  ## Works with your existing setup
346
426