rockycode 0.1.2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (236) hide show
  1. rockycode-0.2.0/CHANGELOG.md +209 -0
  2. {rockycode-0.1.2 → rockycode-0.2.0}/PKG-INFO +102 -32
  3. {rockycode-0.1.2 → rockycode-0.2.0}/README.md +101 -31
  4. {rockycode-0.1.2 → rockycode-0.2.0}/README.zh-CN.md +90 -26
  5. {rockycode-0.1.2 → rockycode-0.2.0}/pyproject.toml +1 -1
  6. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/cli.py +113 -47
  7. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/config.py +36 -14
  8. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/compaction.py +17 -5
  9. rockycode-0.2.0/rockycode/engine/cron.py +527 -0
  10. rockycode-0.2.0/rockycode/engine/effort.py +122 -0
  11. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/events.py +15 -2
  12. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/headless.py +68 -11
  13. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/images.py +57 -4
  14. rockycode-0.2.0/rockycode/engine/localcheck.py +166 -0
  15. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/loop.py +289 -43
  16. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/permission.py +16 -0
  17. rockycode-0.2.0/rockycode/engine/providers.py +751 -0
  18. rockycode-0.2.0/rockycode/engine/selfconfig.py +164 -0
  19. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/server.py +2 -2
  20. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/titler.py +26 -4
  21. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/tools.py +35 -13
  22. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/vision.py +27 -14
  23. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/web.py +7 -1
  24. rockycode-0.2.0/rockycode/models.toml +239 -0
  25. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/onboarding.py +12 -7
  26. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/pricing.py +84 -56
  27. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/prompts/rocky.py +3 -0
  28. rockycode-0.2.0/rockycode/skills/rocky-setup/SKILL.md +57 -0
  29. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/app.py +737 -57
  30. rockycode-0.2.0/rockycode/tui/loopcard.py +135 -0
  31. rockycode-0.2.0/rockycode/tui/modelpicker.py +246 -0
  32. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/permission.py +108 -0
  33. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/package.json +1 -1
  34. rockycode-0.2.0/tests/smoke_cron.py +388 -0
  35. rockycode-0.2.0/tests/smoke_effort.py +61 -0
  36. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_engine.py +50 -0
  37. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_exec.py +47 -0
  38. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_images.py +54 -10
  39. rockycode-0.2.0/tests/smoke_local_models.py +260 -0
  40. rockycode-0.2.0/tests/smoke_models_registry.py +152 -0
  41. rockycode-0.2.0/tests/smoke_pricing.py +117 -0
  42. rockycode-0.2.0/tests/smoke_providers.py +237 -0
  43. rockycode-0.2.0/tests/smoke_selfconfig.py +85 -0
  44. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_serve.py +3 -3
  45. rockycode-0.2.0/tests/smoke_tui_loop.py +588 -0
  46. rockycode-0.2.0/tests/smoke_tui_model.py +130 -0
  47. rockycode-0.2.0/tests/smoke_tui_modeswitch.py +173 -0
  48. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_permission.py +1 -1
  49. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_toggle.py +1 -1
  50. {rockycode-0.1.2 → rockycode-0.2.0}/uv.lock +1 -1
  51. rockycode-0.1.2/CHANGELOG.md +0 -61
  52. rockycode-0.1.2/rockycode/engine/effort.py +0 -46
  53. rockycode-0.1.2/rockycode/engine/providers.py +0 -209
  54. rockycode-0.1.2/rockycode/tui/modelpicker.py +0 -116
  55. rockycode-0.1.2/tests/smoke_effort.py +0 -45
  56. rockycode-0.1.2/tests/smoke_pricing.py +0 -95
  57. rockycode-0.1.2/tests/smoke_providers.py +0 -120
  58. rockycode-0.1.2/tests/smoke_tui_model.py +0 -81
  59. {rockycode-0.1.2 → rockycode-0.2.0}/.dockerignore +0 -0
  60. {rockycode-0.1.2 → rockycode-0.2.0}/.github/workflows/ci.yml +0 -0
  61. {rockycode-0.1.2 → rockycode-0.2.0}/.github/workflows/release.yml +0 -0
  62. {rockycode-0.1.2 → rockycode-0.2.0}/.gitignore +0 -0
  63. {rockycode-0.1.2 → rockycode-0.2.0}/CONTRIBUTING.md +0 -0
  64. {rockycode-0.1.2 → rockycode-0.2.0}/Dockerfile +0 -0
  65. {rockycode-0.1.2 → rockycode-0.2.0}/Dockerfile.sandbox +0 -0
  66. {rockycode-0.1.2 → rockycode-0.2.0}/LICENSE +0 -0
  67. {rockycode-0.1.2 → rockycode-0.2.0}/SECURITY.md +0 -0
  68. {rockycode-0.1.2 → rockycode-0.2.0}/bench/tasks/dev10.json +0 -0
  69. {rockycode-0.1.2 → rockycode-0.2.0}/bench/tasks/test20.json +0 -0
  70. {rockycode-0.1.2 → rockycode-0.2.0}/brand/rockycode-note.svg +0 -0
  71. {rockycode-0.1.2 → rockycode-0.2.0}/brand/rockycode-wordmark.svg +0 -0
  72. {rockycode-0.1.2 → rockycode-0.2.0}/docker-compose.yml +0 -0
  73. {rockycode-0.1.2 → rockycode-0.2.0}/prompts/README.md +0 -0
  74. {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-v1.txt +0 -0
  75. {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-v2-search-first.txt +0 -0
  76. {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-v3-decisive.txt +0 -0
  77. {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-zh-closer.txt +0 -0
  78. {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-zh-full-closer.txt +0 -0
  79. {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-zh-full.txt +0 -0
  80. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/__init__.py +0 -0
  81. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/banner.py +0 -0
  82. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/__init__.py +0 -0
  83. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/core.py +0 -0
  84. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/judge.py +0 -0
  85. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/mining.py +0 -0
  86. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/proposals.py +0 -0
  87. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/__init__.py +0 -0
  88. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/artifact.py +0 -0
  89. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/budget.py +0 -0
  90. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/checks.py +0 -0
  91. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/container.py +0 -0
  92. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/explore.py +0 -0
  93. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/goal.py +0 -0
  94. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/goal_review.py +0 -0
  95. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/goal_session.py +0 -0
  96. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/lsp.py +0 -0
  97. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/mcp.py +0 -0
  98. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/modes.py +0 -0
  99. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/outcome.py +0 -0
  100. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/planmode.py +0 -0
  101. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/redact.py +0 -0
  102. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/safety.py +0 -0
  103. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/sandbox.py +0 -0
  104. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/skills.py +0 -0
  105. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/trajectory.py +0 -0
  106. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/worktree.py +0 -0
  107. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/memory/__init__.py +0 -0
  108. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/memory/index.py +0 -0
  109. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/memory/store.py +0 -0
  110. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/learn/learn.md +0 -0
  111. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/deep-research.md +0 -0
  112. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/paper-reading.md +0 -0
  113. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/prove.md +0 -0
  114. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/whiteboard.md +0 -0
  115. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/palette.py +0 -0
  116. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/prompts/__init__.py +0 -0
  117. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/routines.py +0 -0
  118. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/__init__.py +0 -0
  119. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/agent.py +0 -0
  120. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/data.py +0 -0
  121. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/raw.py +0 -0
  122. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/score.py +0 -0
  123. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/session.py +0 -0
  124. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/architecture-viz/SKILL.md +0 -0
  125. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/architecture-viz/template.html +0 -0
  126. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/lean-prover/SKILL.md +0 -0
  127. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/lean-prover/torchlean-api.md +0 -0
  128. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/__init__.py +0 -0
  129. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/clipboard.py +0 -0
  130. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/exitsheet.py +0 -0
  131. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/goal_screen.py +0 -0
  132. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/mdterm.py +0 -0
  133. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/mdview.py +0 -0
  134. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/modepicker.py +0 -0
  135. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/plangate.py +0 -0
  136. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/prompt_history.py +0 -0
  137. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/proposalcard.py +0 -0
  138. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/resume.py +0 -0
  139. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/rocky_pet.py +0 -0
  140. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/routinecard.py +0 -0
  141. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.gitignore +0 -0
  142. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.vscode/launch.json +0 -0
  143. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.vscode/tasks.json +0 -0
  144. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.vscodeignore +0 -0
  145. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/CHANGELOG.md +0 -0
  146. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/LICENSE +0 -0
  147. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/README.md +0 -0
  148. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/esbuild.config.mjs +0 -0
  149. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/chat.html +0 -0
  150. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/icon-marketplace.svg +0 -0
  151. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/icon.png +0 -0
  152. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/marked.js +0 -0
  153. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/rocky-icon.svg +0 -0
  154. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/package-lock.json +0 -0
  155. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/artifactTree.ts +0 -0
  156. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/diffManager.ts +0 -0
  157. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/editorContext.ts +0 -0
  158. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/extension.ts +0 -0
  159. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/permissionManager.ts +0 -0
  160. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/protocol.ts +0 -0
  161. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/rockyConnection.ts +0 -0
  162. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/rockyProvider.ts +0 -0
  163. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/statusBar.ts +0 -0
  164. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/webview-highlight.ts +0 -0
  165. {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/tsconfig.json +0 -0
  166. {rockycode-0.1.2 → rockycode-0.2.0}/tests/fake_lsp_server.py +0 -0
  167. {rockycode-0.1.2 → rockycode-0.2.0}/tests/fake_mcp_server.py +0 -0
  168. {rockycode-0.1.2 → rockycode-0.2.0}/tests/invariants.py +0 -0
  169. {rockycode-0.1.2 → rockycode-0.2.0}/tests/real_planmode.py +0 -0
  170. {rockycode-0.1.2 → rockycode-0.2.0}/tests/real_reasoning_roundtrip.py +0 -0
  171. {rockycode-0.1.2 → rockycode-0.2.0}/tests/real_tool_contract.py +0 -0
  172. {rockycode-0.1.2 → rockycode-0.2.0}/tests/run_all.py +0 -0
  173. {rockycode-0.1.2 → rockycode-0.2.0}/tests/run_real.py +0 -0
  174. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_approver.py +0 -0
  175. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_artifact.py +0 -0
  176. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_bash.py +0 -0
  177. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_bash_grant.py +0 -0
  178. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_bilang_prompt.py +0 -0
  179. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_budget.py +0 -0
  180. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_checks.py +0 -0
  181. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_compaction.py +0 -0
  182. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_container.py +0 -0
  183. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_credentials.py +0 -0
  184. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_docker_sandbox_cancel.py +0 -0
  185. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_dream.py +0 -0
  186. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_dream_trigger.py +0 -0
  187. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_explore.py +0 -0
  188. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal.py +0 -0
  189. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal_driver.py +0 -0
  190. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal_review.py +0 -0
  191. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal_runner.py +0 -0
  192. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_hardkill_resume.py +0 -0
  193. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_interrupt.py +0 -0
  194. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_judge.py +0 -0
  195. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_lsp.py +0 -0
  196. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mcp.py +0 -0
  197. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mdterm.py +0 -0
  198. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mdview.py +0 -0
  199. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_memory.py +0 -0
  200. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_memory_index.py +0 -0
  201. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mining.py +0 -0
  202. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_modes.py +0 -0
  203. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_onboarding.py +0 -0
  204. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_outcome.py +0 -0
  205. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_parallel.py +0 -0
  206. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_permission.py +0 -0
  207. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_planmode.py +0 -0
  208. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_prompt_history.py +0 -0
  209. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_proposals.py +0 -0
  210. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_redact.py +0 -0
  211. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_resume_handoff.py +0 -0
  212. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_routine_run.py +0 -0
  213. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_routines.py +0 -0
  214. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_safety.py +0 -0
  215. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_score.py +0 -0
  216. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_session.py +0 -0
  217. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_skills.py +0 -0
  218. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_submit_race.py +0 -0
  219. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_today.py +0 -0
  220. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tools.py +0 -0
  221. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tools_jail.py +0 -0
  222. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_trajectory.py +0 -0
  223. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_artifact_modal.py +0 -0
  224. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_bash_gate.py +0 -0
  225. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_copy.py +0 -0
  226. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_envwarn.py +0 -0
  227. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_exitsheet.py +0 -0
  228. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_goal_screen.py +0 -0
  229. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_input_nav.py +0 -0
  230. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_plan.py +0 -0
  231. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_proposals.py +0 -0
  232. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_read_grant.py +0 -0
  233. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_routines.py +0 -0
  234. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_scroll.py +0 -0
  235. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_shell.py +0 -0
  236. {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_web.py +0 -0
@@ -0,0 +1,209 @@
1
+ # Changelog
2
+
3
+ All notable changes to rockycode are recorded here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/), and the project follows
5
+ [Semantic Versioning](https://semver.org/) — pre-1.0, so the surface may still
6
+ change between minor versions.
7
+
8
+ ## [Unreleased]
9
+
10
+ ## [0.2.0] — models are data, loops, approval switch
11
+
12
+ ### Added
13
+ - **Loops** (`/loop`, alias `/cron`): re-fire one prompt into THIS chat on an
14
+ interval — `/loop 5m check whether results/run3 has a summary.json; if so
15
+ give me the headline numbers`. Each tick is an ordinary turn on the session
16
+ engine (it sees the whole conversation), runs only when the chat is idle,
17
+ ends with `LOOP QUIET / NOTE / DONE`, and never prompts: a call that needs
18
+ approval pauses the loop (`/loop allow` decides it, `/loop resume` retries,
19
+ `/permission yolo` or shift+tab resumes every approval-paused loop; a manual
20
+ `/loop pause` stays paused). Quiet ticks fold into one streak line and roll
21
+ out of live context (the trajectory keeps them). No cap unless you set one
22
+ (`for 3h`, `x12`, `max $2`); soft running-cost reminders otherwise. Rocky
23
+ can start one itself (`loop_start` / `loop_stop`, through the normal
24
+ approval). Session-only — a loop dies with the session; `/routines` stays
25
+ the cross-launch scheduler.
26
+ - **Models are data.** The whole model catalog now lives in
27
+ `rockycode/models.toml` — per provider a China base URL, a rocky-owned key
28
+ name, a reasoning wire shape (`thinking` · `effort` · `enable_thinking` ·
29
+ `minimax` · `openai` · `none`), its own effort tiers, whether thinking can
30
+ be switched off, and where usage reports cache hits; per model its context
31
+ window, output cap, vision flag, price tables (USD + CNY) and roles. The
32
+ engine consumes a `ModelSpec` and carries no model-specific numbers of its
33
+ own. `~/.rockycode/models.toml` (same shape) is deep-merged on top: add a
34
+ model, correct a limit, add a price, `hidden = true` to drop a row.
35
+ - The 2026-09 roster: `deepseek-flash` (V4.1 Flash — native vision, the
36
+ default and the sidecar/search model) + `deepseek-v4-pro`; `glm-5.3` +
37
+ `glm-5.3-flash`; `kimi-k3`; `minimax-m3`; `step-5-preview`; `qwen3.8-max`
38
+ + `qwen3.8-flash`; `mimo-v2.6-pro`; `ollama` (live-discovered). Limits and
39
+ DeepSeek's V4.1 prices verified at each provider's docs on 2026-09-29.
40
+ - Subscription-plan endpoints as their own picker rows: `qwen-plan` (Bailian
41
+ Token Plan), `mimo-plan`, `stepfun-plan` — own URL, own key
42
+ (`ROCKYCODE_<PROVIDER>_PLAN_API_KEY`).
43
+ - Context window and output cap follow the active model: config
44
+ `context_window` / `max_tokens` default to `0` (= the registry value) and
45
+ re-pace on every `/model` switch; a positive number pins your own ceiling.
46
+ `/config context_window 0` un-pins.
47
+ - `/effort off|low|high|max` — the dial now reaches `low` (DeepSeek, GLM and
48
+ Kimi all take low|high|max), is clamped onto each provider's own tiers by
49
+ position at the wire (StepFun's `low|medium|high` gets `medium` for
50
+ rocky's `high`), and sends the lowest tier for `off` on a model that
51
+ can't disable thinking (GLM, Kimi). The status line says what is sent.
52
+ `xhigh` stays accepted as `max`.
53
+ - `image_route auto` (new default): a pasted image on a text-only model is
54
+ described silently by the registry's sidecar (`deepseek-flash` on the home
55
+ key) or your image CLI — no picker in the way; `ask` keeps the picker.
56
+ - **Rocky configures rocky.** A built-in `rocky-setup` skill plus the
57
+ ask-tier `rocky_config` tool (show · set · set_url · add_provider) let a
58
+ user ask rocky to use a proxy, add a vLLM box, change the default model or
59
+ a limit — writes go under `~/.rockycode` only, through the same validated
60
+ setters as the CLI. Key-shaped values are refused with the env var to use.
61
+ - `rockycode exec --profile read|write|full`: `read` (read_file/grep/glob/
62
+ view_image) and `write` (+ jailed write_file/edit_file) have no shell, so
63
+ they run on the host with no Docker and start instantly — the profiles a
64
+ calling agent (Claude Code, Codex) wants for "look and tell" and small
65
+ edits. `full` keeps bash in the Docker sandbox by default. The stream is
66
+ now meta → text → result by default; `--events` restores the tool.*/turn.*
67
+ receipt lines (`--include-thinking` implies it). `meta.profile` reports
68
+ name, mode and tool set.
69
+ - Peak-hour pricing honors DeepSeek's calendar: Monday–Friday only, with a
70
+ `holidays` list (ISO dates, Beijing) on the schedule for Chinese public
71
+ holidays — weekends and holidays bill off-peak.
72
+ - Approval mode is switchable without typing: `shift+tab` steps it one notch
73
+ looser (careful → ask → yolo, wrapping back to careful, so a run of presses
74
+ never parks you in yolo). The status-bar 🔒 chip now spells that key out and
75
+ opens a picker when clicked — three rows saying what each mode actually
76
+ allows — and bare `/permission` opens the same picker. The cycle stays inert
77
+ while an approval prompt is waiting, and landing in yolo still prints the
78
+ "runs on your machine" warning.
79
+ - Local models: a builtin `ollama` provider (`http://localhost:11434/v1`,
80
+ override with `ROCKYCODE_OLLAMA_URL`) — keyless, priced `$0 · local`, model
81
+ list discovered live from the running server's `GET /v1/models` so the
82
+ `/model` picker shows what's actually pulled. Server down → a dimmed
83
+ "not running · start it: ollama serve" row instead of silence.
84
+ - `/model` readiness preflight for local providers: before the engine is
85
+ switched, rocky checks server reachability, that the model is pulled, that
86
+ it supports tool calling, and the serving context length (Ollama truncates
87
+ silently — the #1 agent-loop killer). Not ready → the switch is refused and
88
+ every failed check carries its exact fix (`ollama pull …`,
89
+ `export OLLAMA_CONTEXT_LENGTH=65536`); ready → rocky's `context_window` is
90
+ paced to the server's verified value (restored on switching back to a cloud
91
+ provider), and a server-reported vision capability turns image input on.
92
+ - Keyless endpoints: `local = true` on a `~/.rockycode/providers.toml`
93
+ provider (LM Studio, llama.cpp, vLLM, a remote box's ollama) marks its
94
+ endpoints keyless — always configured, no placeholder-key hack needed.
95
+ - Compaction on a local provider sends no tool schemas in the summarize call
96
+ (local compat layers ignore `tool_choice="none"` and may answer with a tool
97
+ call instead of a summary), and the thinking-off field is now shaped per
98
+ the provider's reasoning policy instead of always DeepSeek's.
99
+ - Vision is now per-MODEL, not per-provider (`Provider.vision_models`,
100
+ choice-level `❖` badge). Config key `vision_models` marks additional ids as
101
+ image-capable from any shell (`rockycode config vision_models <id>`) — for
102
+ when a provider ships vision on an existing model before the registry
103
+ catches up.
104
+ - Config key `model`: a sticky launch default (`rockycode config model
105
+ <spec>`), resolved against the registry; precedence `--model` flag →
106
+ `ROCKYCODE_MODEL` env → config. Global config only — a cloned repo can
107
+ never redirect requests.
108
+ - The `/model` picker is now two-step, model first: one row per model, and a
109
+ model with several endpoints then asks which URL serves it — including a
110
+ "custom base URL" row that remembers your own gateway/proxy per provider
111
+ (`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`, riding
112
+ the provider's key).
113
+ - Images are best-effort downscaled to 2048px before hitting the wire (PIL if
114
+ installed, macOS `sips` otherwise, original on any failure) — vision
115
+ providers bill image tokens by dimensions, and a Retina screenshot was
116
+ paying severalfold for nothing.
117
+
118
+ ### Changed
119
+ - One China endpoint per provider (no more `kimi-cn`/`kimi-en`, `zai`/`glm-cn`,
120
+ `minimax-cn`/`-en`): `/model kimi`, `/model glm`, `/model minimax`. Keys are
121
+ `ROCKYCODE_<PROVIDER>_API_KEY`; the older `_CN_`/`_EN_`/`ZAI_EN` names are
122
+ still read as aliases (the picker names the alias in use), so no setup
123
+ breaks. An international or proxy URL is the custom-URL row.
124
+ - `deepseek-v4-flash` and `deepseek-v4-flash-vision-exp` are retired from
125
+ the list (DeepSeek retired the models on 2026-09-10 and serves those names
126
+ with V4.1-Flash); rocky aliases both to `deepseek-flash`, so old configs,
127
+ trajectories and `/model` habits keep working. `deepseek-v4-pro` stays
128
+ listed, labelled: since 2026-09-14 DeepSeek routes it to V4.1-Flash at
129
+ Flash rates until V4.1-Pro ships.
130
+ - `glm-5.2` and `step-3.7-flash` drop off the roster in favor of GLM-5.3 /
131
+ Step 5 Preview.
132
+ - Session titles, image describe, and the native web search all use the
133
+ registry's `sidecar` / `search` role (deepseek-flash) — a session on Kimi
134
+ or GLM no longer sends a DeepSeek model id down its own client for the
135
+ title call.
136
+ - `--max-tokens` / `--context-window` default to `0` (= the model's registry
137
+ value) on `chat` and `serve`; `bench` keeps its pinned reproducible numbers.
138
+ - The CLI help, status line and context reminder no longer speak of
139
+ "DeepSeek V4" as the only model.
140
+ - `view_image` on a vision-capable active model now attaches the real image
141
+ to the conversation (as the next user message) instead of a sidecar text
142
+ description — the model reads the pixels and decides what matters itself.
143
+ Text-only models keep the describe routes unchanged.
144
+ - Launch honors the registry: starting with `--model minimax-m3` (or any
145
+ registry model) now gets the right reasoning params, tools flag, and vision
146
+ capability from step one instead of DeepSeek-shaped defaults until the
147
+ first `/model` switch. Unresolved specs and injected clients behave exactly
148
+ as before.
149
+ - `/model` spec resolution prefers exact ids, and a base model wins its own
150
+ substring — `glm:5.3` means `glm-5.3`, not ambiguity with `glm-5.3-flash`;
151
+ the variant stays reachable via `glm:flash`.
152
+
153
+ ### Fixed
154
+ - The prompt-cache observer and the cost ledger read cache hits from
155
+ OpenAI-style `prompt_tokens_details.cached_tokens` too (Kimi, GLM, MiniMax,
156
+ Qwen, MiMo), not only DeepSeek's `prompt_cache_hit_tokens`.
157
+ - Launching on a `<provider>-custom` endpoint of the home provider now uses
158
+ that URL (it used to fall back to the env base URL).
159
+
160
+ ## [0.1.2] — `--version` flag, GA price tables
161
+
162
+ ### Added
163
+ - `rockycode --version` / `-V` prints the installed version. One version
164
+ source: package metadata (pyproject) — the serve handshake reports the same
165
+ value instead of a hardcoded string.
166
+
167
+ ### Changed
168
+ - DeepSeek price tables refreshed to the GA snapshots (V4-Flash-0731 /
169
+ V4-Pro-0813), both USD and CNY, verified 2026-08-20 at the source;
170
+ peak-valley billing confirmed live (2× in the published UTC windows).
171
+ - README results updated: `deepseek-v4-flash` GA carries three clean full-500
172
+ SWE-bench Verified rounds (88.8% average, 95.4% pass@3); the
173
+ `deepseek-v4-pro` column is explicitly marked preview (pre-0813).
174
+
175
+ ### CI
176
+ - Release workflow reduced to the single ubuntu PyPI Trusted-Publishing job;
177
+ the dead macos-13 matrix (retired runner, never ran) is gone.
178
+
179
+ ## [0.1.1] — session artifacts, images in chat, real sandbox cancel
180
+
181
+ ### Added
182
+ - Images in chat: paste (`ctrl+v` / `/paste`) or drag an image in. Vision models
183
+ see it raw; no-vision models route through a provider sidecar, your own image
184
+ CLI, or a `view_image` tool — picked once, remembered. Known CLIs set up with
185
+ one word (`/config image_cli mmx` — rocky knows the invocation).
186
+ - Session artifact inventory: `/artifact list · open <n> · stop · live on|off`,
187
+ a footer badge with open-tab counts, and an Artifacts tree in the VS Code
188
+ extension fed live by `rockycode serve`.
189
+ - Bare `/model` opens a live provider + model picker.
190
+
191
+ ### Fixed
192
+ - Live artifacts no longer drop and reconnect every 30 s; the artifact server
193
+ stops/restarts cleanly and rebinds saved live pages to the new port.
194
+ - Sandbox cancel/timeout kills the in-container process group, not just the
195
+ host-side docker client; images without `python3` fall back to plain `bash -c`.
196
+ - Resume self-heals after a hard kill; closing the doc dock no longer wedges the TUI.
197
+
198
+ ### Internal
199
+ - CI: conservative ruff correctness gate (pinned `0.16.1`) ahead of the smoke suite.
200
+
201
+ ## [0.1.0] — first public release
202
+
203
+ Initial public release. One repo, one engine, three ways to use it — interactive
204
+ `chat`, autonomous `goal`, and the `bench` measurement rig — running on DeepSeek
205
+ or any OpenAI-compatible model.
206
+
207
+ Some capabilities ship as **experimental and default-off** (self-improvement,
208
+ `prove` / `lean-prover`, `explore`, and providers other than DeepSeek); see the
209
+ README's Experimental section for what they are and how to enable them.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: rockycode
3
- Version: 0.1.2
3
+ Version: 0.2.0
4
4
  Summary: A coding agent harness, benchmarked on SWE-bench Verified. amaze!
5
5
  Author: rockycode contributors
6
6
  License: MIT
@@ -205,8 +205,8 @@ clipboard" (or your terminal's equivalent) on the local end.
205
205
  | `/research` | Research modes: deep-research · paper-reading · whiteboard · prove |
206
206
  | `/learn` | Tutor mode — your understanding is the goal, not the diff |
207
207
  | `/model` | Switch provider and model (see below) |
208
- | `/effort off\|high\|xhigh\|max` | Reasoning depth, adjustable live per session |
209
- | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
208
+ | `/effort off\|low\|high\|max` | Reasoning depth, adjustable live per session (clamped onto each provider's own tiers) |
209
+ | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session — bare opens a picker; `shift+tab` cycles it, or click the 🔒 chip in the status bar |
210
210
  | `/sandbox on\|off\|status` | Isolate tool execution in a container |
211
211
  | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
212
212
  | `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
@@ -238,27 +238,79 @@ clipboard" (or your terminal's equivalent) on the local end.
238
238
 
239
239
  ### Models and providers
240
240
 
241
- DeepSeek is the home model, but providers are data, not code: each is a
242
- base URL, a model list, a key variable, and a reasoning shape over the
243
- OpenAI-compatible API.
241
+ DeepSeek is the home model, but **models are data, not code**: the whole
242
+ catalog lives in `rockycode/models.toml` — per provider a China base URL, a
243
+ key name and a reasoning wire shape; per model its context window, output
244
+ cap, vision flag, price and roles. The engine reads that spec and carries no
245
+ model-specific numbers of its own, so a new model is a data edit. Your own
246
+ `~/.rockycode/models.toml` (same shape) is deep-merged on top: add a model,
247
+ correct a limit, add a price, hide a row.
244
248
 
245
- | Provider | Models |
246
- |---|---|
247
- | **deepseek** (default) | `deepseek-v4-flash` (default), `deepseek-v4-pro` (preview) |
248
- | **minimax** | `minimax-m3` |
249
- | **kimi** | `kimi-k3` |
250
- | **glm** | `glm-5.2` |
251
-
252
- Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
253
- and custom providers — including local vLLM/SGLang servers — go in
254
- `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
255
- keys are actually configured. DeepSeek and MiniMax both carry full-500 bench
256
- numbers (see [Results](#results--swe-bench-verified)); Kimi and GLM are
257
- [experimental](#experimental).
258
-
259
- The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
260
- provider maps it to its own reasoning tiers at the wire (DeepSeek, for
261
- example, only distinguishes `high|max`, so `xhigh` clamps to `max`).
249
+ | Provider | Models (❖ = takes image input) | ctx / max out |
250
+ |---|---|---|
251
+ | **deepseek** (default) | `deepseek-flash` (V4.1 Flash, default) ❖, `deepseek-v4-pro` | 1M / 384K |
252
+ | **glm** | `glm-5.3`, `glm-5.3-flash` ❖ | 1M / 128K |
253
+ | **kimi** | `kimi-k3` ❖ | 1M / 128K |
254
+ | **minimax** | `minimax-m3` ❖ | 1M / 128K |
255
+ | **stepfun** | `step-5-preview` ❖ | 1M / 64K |
256
+ | **qwen** | `qwen3.8-max` ❖, `qwen3.8-flash` ❖ | 1M / 64K |
257
+ | **mimo** | `mimo-v2.6-pro` ❖ | 1M / 128K |
258
+ | **ollama** (local, $0) | whatever you've pulled — discovered live from the running server | server-verified |
259
+
260
+ One China endpoint per provider (`ROCKYCODE_<PROVIDER>_API_KEY`; the older
261
+ `_CN_`/`_EN_` names are still read). Subscription plans with their own URL
262
+ and key are their own rows — `qwen-plan` (Bailian Token Plan), `mimo-plan`,
263
+ `stepfun-plan` — keyed as `ROCKYCODE_<PROVIDER>_PLAN_API_KEY`. The `/model`
264
+ picker lists **models first**, one row each; a model with several endpoints
265
+ then asks which URL serves it (official · plan · your own), and the "custom
266
+ base URL" row remembers a gateway or proxy per provider
267
+ (`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`). Typed
268
+ specs skip all of that: `/model glm:flash`, `/model qwen-plan:qwen3.8-max`.
269
+ The retired `deepseek-v4-flash` / `-vision-exp` ids still resolve (to
270
+ `deepseek-flash`, exactly as DeepSeek serves them). The picker only offers
271
+ providers whose keys are actually configured.
272
+
273
+ Vision is per-model: `deepseek-flash` sees images on the home key, so the
274
+ default session just takes a paste. A text-only model (`deepseek-v4-pro`,
275
+ `glm-5.3`) gets pasted images described by `deepseek-flash` silently
276
+ (`image_route auto`), or by your own CLI. `rockycode config model <spec>`
277
+ makes any pick the sticky launch default. Context window and output cap
278
+ follow the active model (config `context_window` / `max_tokens` = `0`); a
279
+ number pins your own ceiling across switches. DeepSeek and MiniMax carry
280
+ full-500 bench numbers (see [Results](#results--swe-bench-verified)); the
281
+ other providers are [experimental](#experimental).
282
+
283
+ The effort dial (`/effort off|low|high|max`) is provider-neutral; each
284
+ provider's own tiers come from the registry and the dial is clamped onto
285
+ them by position at the wire (StepFun's `low|medium|high` gets `medium` for
286
+ rocky's `high`; GLM and Kimi can't switch thinking off, so `off` sends their
287
+ lowest tier). `xhigh` is still accepted and means `max`.
288
+
289
+ Rocky can also configure itself: ask it to "use my proxy", "add my vLLM
290
+ server", or "switch the default model" and the built-in `rocky-setup` skill
291
+ plus the ask-tier `rocky_config` tool make the change under `~/.rockycode`.
292
+ Keys are the one thing it never touches — it names the variable and you
293
+ paste the value.
294
+
295
+ #### Local models (Ollama)
296
+
297
+ Rocky integrates the OpenAI-compatible *protocol*, never a runtime — and
298
+ recommends [Ollama](https://ollama.com) (its MLX engine covers Apple Silicon
299
+ since 0.19). No key, no config: run `ollama serve`, pull a tool-capable model
300
+ (`ollama pull qwen3.8:27b-mlx` is the tested recommendation), and it appears
301
+ in `/model` with what's actually pulled, priced `$0 · local`.
302
+
303
+ Switching to a local model runs a **readiness preflight** first — server up,
304
+ model pulled, tool-calling support, serving context — and refuses the switch
305
+ with the exact fix (`ollama pull …`, `export OLLAMA_CONTEXT_LENGTH=65536`)
306
+ when something would break mid-session. The one to respect: Ollama's default
307
+ context is small and it **truncates silently**, which kills agent sessions in
308
+ confusing ways — serve with `OLLAMA_CONTEXT_LENGTH=65536`. Rocky paces its own
309
+ `context_window` to the server's verified value on every switch.
310
+
311
+ Other local servers (LM Studio, llama.cpp, vLLM) work as data too: add a
312
+ provider with `local = true` in `~/.rockycode/providers.toml` and its
313
+ endpoints need no key.
262
314
 
263
315
  ## Autonomous use
264
316
 
@@ -288,11 +340,26 @@ rockycode goal "add a docstring to <fn> and run the linter" --max-usd 0.50 --max
288
340
  ### Headless delegation: `exec`
289
341
 
290
342
  `rockycode exec "<task>"` is the single-shot, non-interactive entry point,
291
- designed to be called by *other* agents and scripts. Events stream as JSONL on
292
- stdout; the Docker sandbox is **on by default** (the command classifier is
293
- defense-in-depth, not the boundary); budgets are always enforced; and exit
294
- codes distinguish success, failure, needs-approval, and budget-stop — so a
295
- calling agent can grant an approval and resume instead of guessing.
343
+ designed to be called by *other* agents and scripts. stdout is JSONL: a
344
+ `meta` line, the model's `text`, and a `result` envelope with evidence
345
+ (files changed, commands run, refusals) — never verdicts, the caller
346
+ verifies; `--events` adds the per-tool receipt lines. Budgets are always
347
+ enforced, and exit codes distinguish success, failure, needs-approval, and
348
+ budget-stop — so a calling agent can grant an approval and resume instead
349
+ of guessing.
350
+
351
+ Pick how much rocky may do with `--profile`: `read` (read_file / grep /
352
+ glob / view_image — no shell, no writes) and `write` (+ write_file /
353
+ edit_file jailed to `--workdir`) run on the host with **no Docker** and
354
+ start instantly — what Claude Code or Codex wants for "look at this repo and
355
+ tell me" or a small edit on a cheap, fast model. `full` adds bash, in the
356
+ Docker sandbox **by default** (the command classifier is defense-in-depth,
357
+ not the boundary).
358
+
359
+ ```bash
360
+ rockycode exec --profile read "which module owns retry logic, and where is it called?"
361
+ rockycode exec --profile write "add a docstring to every public function in utils.py"
362
+ ```
296
363
 
297
364
  ### Editor integration: `serve` and the VS Code extension
298
365
 
@@ -347,10 +414,13 @@ change. Anything that could act on its own is **off by default**.
347
414
  investigation from a fresh-context child that returns only a cited,
348
415
  mechanically-verified report; the search noise never enters your session. It
349
416
  also grounds goal mode's branch review and milestone verification.
350
- - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
351
- OpenAI-compatible profiles (`/model`). DeepSeek and MiniMax carry full
352
- bench numbers (see Results); treat GLM and Kimi as untested until they do
353
- too.
417
+ - **Providers beyond DeepSeek.** GLM, Kimi, MiniMax, StepFun, Qwen, and MiMo
418
+ are wired as OpenAI-compatible registry entries (`/model`). DeepSeek and
419
+ MiniMax carry full bench numbers (see Results); treat the rest as untested
420
+ until they do too. Registry entries marked `note = "… verify …"` in
421
+ `models.toml` (MiniMax's endpoint host, MiMo's auth header, the plan URLs)
422
+ were taken from each provider's docs on 2026-09-29 and not yet exercised
423
+ live — a wrong one is a one-line data fix.
354
424
 
355
425
  ## Works with your existing setup
356
426
 
@@ -172,8 +172,8 @@ clipboard" (or your terminal's equivalent) on the local end.
172
172
  | `/research` | Research modes: deep-research · paper-reading · whiteboard · prove |
173
173
  | `/learn` | Tutor mode — your understanding is the goal, not the diff |
174
174
  | `/model` | Switch provider and model (see below) |
175
- | `/effort off\|high\|xhigh\|max` | Reasoning depth, adjustable live per session |
176
- | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
175
+ | `/effort off\|low\|high\|max` | Reasoning depth, adjustable live per session (clamped onto each provider's own tiers) |
176
+ | `/permission yolo\|ask\|careful` | Tool-approval strictness for the session — bare opens a picker; `shift+tab` cycles it, or click the 🔒 chip in the status bar |
177
177
  | `/sandbox on\|off\|status` | Isolate tool execution in a container |
178
178
  | `/lsp` | Language-server status; diagnostics ride along with `read_file` |
179
179
  | `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
@@ -205,27 +205,79 @@ clipboard" (or your terminal's equivalent) on the local end.
205
205
 
206
206
  ### Models and providers
207
207
 
208
- DeepSeek is the home model, but providers are data, not code: each is a
209
- base URL, a model list, a key variable, and a reasoning shape over the
210
- OpenAI-compatible API.
208
+ DeepSeek is the home model, but **models are data, not code**: the whole
209
+ catalog lives in `rockycode/models.toml` — per provider a China base URL, a
210
+ key name and a reasoning wire shape; per model its context window, output
211
+ cap, vision flag, price and roles. The engine reads that spec and carries no
212
+ model-specific numbers of its own, so a new model is a data edit. Your own
213
+ `~/.rockycode/models.toml` (same shape) is deep-merged on top: add a model,
214
+ correct a limit, add a price, hide a row.
211
215
 
212
- | Provider | Models |
213
- |---|---|
214
- | **deepseek** (default) | `deepseek-v4-flash` (default), `deepseek-v4-pro` (preview) |
215
- | **minimax** | `minimax-m3` |
216
- | **kimi** | `kimi-k3` |
217
- | **glm** | `glm-5.2` |
218
-
219
- Regional endpoints are addressable as `<provider>-<region>` (e.g. `kimi-cn`),
220
- and custom providers — including local vLLM/SGLang servers — go in
221
- `~/.rockycode/providers.toml`. The `/model` picker only offers providers whose
222
- keys are actually configured. DeepSeek and MiniMax both carry full-500 bench
223
- numbers (see [Results](#results--swe-bench-verified)); Kimi and GLM are
224
- [experimental](#experimental).
225
-
226
- The effort dial (`/effort off|high|xhigh|max`) is provider-neutral; each
227
- provider maps it to its own reasoning tiers at the wire (DeepSeek, for
228
- example, only distinguishes `high|max`, so `xhigh` clamps to `max`).
216
+ | Provider | Models (❖ = takes image input) | ctx / max out |
217
+ |---|---|---|
218
+ | **deepseek** (default) | `deepseek-flash` (V4.1 Flash, default) ❖, `deepseek-v4-pro` | 1M / 384K |
219
+ | **glm** | `glm-5.3`, `glm-5.3-flash` ❖ | 1M / 128K |
220
+ | **kimi** | `kimi-k3` ❖ | 1M / 128K |
221
+ | **minimax** | `minimax-m3` ❖ | 1M / 128K |
222
+ | **stepfun** | `step-5-preview` ❖ | 1M / 64K |
223
+ | **qwen** | `qwen3.8-max` ❖, `qwen3.8-flash` ❖ | 1M / 64K |
224
+ | **mimo** | `mimo-v2.6-pro` ❖ | 1M / 128K |
225
+ | **ollama** (local, $0) | whatever you've pulled — discovered live from the running server | server-verified |
226
+
227
+ One China endpoint per provider (`ROCKYCODE_<PROVIDER>_API_KEY`; the older
228
+ `_CN_`/`_EN_` names are still read). Subscription plans with their own URL
229
+ and key are their own rows — `qwen-plan` (Bailian Token Plan), `mimo-plan`,
230
+ `stepfun-plan` — keyed as `ROCKYCODE_<PROVIDER>_PLAN_API_KEY`. The `/model`
231
+ picker lists **models first**, one row each; a model with several endpoints
232
+ then asks which URL serves it (official · plan · your own), and the "custom
233
+ base URL" row remembers a gateway or proxy per provider
234
+ (`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`). Typed
235
+ specs skip all of that: `/model glm:flash`, `/model qwen-plan:qwen3.8-max`.
236
+ The retired `deepseek-v4-flash` / `-vision-exp` ids still resolve (to
237
+ `deepseek-flash`, exactly as DeepSeek serves them). The picker only offers
238
+ providers whose keys are actually configured.
239
+
240
+ Vision is per-model: `deepseek-flash` sees images on the home key, so the
241
+ default session just takes a paste. A text-only model (`deepseek-v4-pro`,
242
+ `glm-5.3`) gets pasted images described by `deepseek-flash` silently
243
+ (`image_route auto`), or by your own CLI. `rockycode config model <spec>`
244
+ makes any pick the sticky launch default. Context window and output cap
245
+ follow the active model (config `context_window` / `max_tokens` = `0`); a
246
+ number pins your own ceiling across switches. DeepSeek and MiniMax carry
247
+ full-500 bench numbers (see [Results](#results--swe-bench-verified)); the
248
+ other providers are [experimental](#experimental).
249
+
250
+ The effort dial (`/effort off|low|high|max`) is provider-neutral; each
251
+ provider's own tiers come from the registry and the dial is clamped onto
252
+ them by position at the wire (StepFun's `low|medium|high` gets `medium` for
253
+ rocky's `high`; GLM and Kimi can't switch thinking off, so `off` sends their
254
+ lowest tier). `xhigh` is still accepted and means `max`.
255
+
256
+ Rocky can also configure itself: ask it to "use my proxy", "add my vLLM
257
+ server", or "switch the default model" and the built-in `rocky-setup` skill
258
+ plus the ask-tier `rocky_config` tool make the change under `~/.rockycode`.
259
+ Keys are the one thing it never touches — it names the variable and you
260
+ paste the value.
261
+
262
+ #### Local models (Ollama)
263
+
264
+ Rocky integrates the OpenAI-compatible *protocol*, never a runtime — and
265
+ recommends [Ollama](https://ollama.com) (its MLX engine covers Apple Silicon
266
+ since 0.19). No key, no config: run `ollama serve`, pull a tool-capable model
267
+ (`ollama pull qwen3.8:27b-mlx` is the tested recommendation), and it appears
268
+ in `/model` with what's actually pulled, priced `$0 · local`.
269
+
270
+ Switching to a local model runs a **readiness preflight** first — server up,
271
+ model pulled, tool-calling support, serving context — and refuses the switch
272
+ with the exact fix (`ollama pull …`, `export OLLAMA_CONTEXT_LENGTH=65536`)
273
+ when something would break mid-session. The one to respect: Ollama's default
274
+ context is small and it **truncates silently**, which kills agent sessions in
275
+ confusing ways — serve with `OLLAMA_CONTEXT_LENGTH=65536`. Rocky paces its own
276
+ `context_window` to the server's verified value on every switch.
277
+
278
+ Other local servers (LM Studio, llama.cpp, vLLM) work as data too: add a
279
+ provider with `local = true` in `~/.rockycode/providers.toml` and its
280
+ endpoints need no key.
229
281
 
230
282
  ## Autonomous use
231
283
 
@@ -255,11 +307,26 @@ rockycode goal "add a docstring to <fn> and run the linter" --max-usd 0.50 --max
255
307
  ### Headless delegation: `exec`
256
308
 
257
309
  `rockycode exec "<task>"` is the single-shot, non-interactive entry point,
258
- designed to be called by *other* agents and scripts. Events stream as JSONL on
259
- stdout; the Docker sandbox is **on by default** (the command classifier is
260
- defense-in-depth, not the boundary); budgets are always enforced; and exit
261
- codes distinguish success, failure, needs-approval, and budget-stop — so a
262
- calling agent can grant an approval and resume instead of guessing.
310
+ designed to be called by *other* agents and scripts. stdout is JSONL: a
311
+ `meta` line, the model's `text`, and a `result` envelope with evidence
312
+ (files changed, commands run, refusals) — never verdicts, the caller
313
+ verifies; `--events` adds the per-tool receipt lines. Budgets are always
314
+ enforced, and exit codes distinguish success, failure, needs-approval, and
315
+ budget-stop — so a calling agent can grant an approval and resume instead
316
+ of guessing.
317
+
318
+ Pick how much rocky may do with `--profile`: `read` (read_file / grep /
319
+ glob / view_image — no shell, no writes) and `write` (+ write_file /
320
+ edit_file jailed to `--workdir`) run on the host with **no Docker** and
321
+ start instantly — what Claude Code or Codex wants for "look at this repo and
322
+ tell me" or a small edit on a cheap, fast model. `full` adds bash, in the
323
+ Docker sandbox **by default** (the command classifier is defense-in-depth,
324
+ not the boundary).
325
+
326
+ ```bash
327
+ rockycode exec --profile read "which module owns retry logic, and where is it called?"
328
+ rockycode exec --profile write "add a docstring to every public function in utils.py"
329
+ ```
263
330
 
264
331
  ### Editor integration: `serve` and the VS Code extension
265
332
 
@@ -314,10 +381,13 @@ change. Anything that could act on its own is **off by default**.
314
381
  investigation from a fresh-context child that returns only a cited,
315
382
  mechanically-verified report; the search noise never enters your session. It
316
383
  also grounds goal mode's branch review and milestone verification.
317
- - **Providers beyond DeepSeek.** MiniMax, GLM / z.ai, and Kimi are wired as
318
- OpenAI-compatible profiles (`/model`). DeepSeek and MiniMax carry full
319
- bench numbers (see Results); treat GLM and Kimi as untested until they do
320
- too.
384
+ - **Providers beyond DeepSeek.** GLM, Kimi, MiniMax, StepFun, Qwen, and MiMo
385
+ are wired as OpenAI-compatible registry entries (`/model`). DeepSeek and
386
+ MiniMax carry full bench numbers (see Results); treat the rest as untested
387
+ until they do too. Registry entries marked `note = "… verify …"` in
388
+ `models.toml` (MiniMax's endpoint host, MiMo's auth header, the plan URLs)
389
+ were taken from each provider's docs on 2026-09-29 and not yet exercised
390
+ live — a wrong one is a one-line data fix.
321
391
 
322
392
  ## Works with your existing setup
323
393