rockycode 0.1.2__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode-0.2.0/CHANGELOG.md +209 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/PKG-INFO +102 -32
- {rockycode-0.1.2 → rockycode-0.2.0}/README.md +101 -31
- {rockycode-0.1.2 → rockycode-0.2.0}/README.zh-CN.md +90 -26
- {rockycode-0.1.2 → rockycode-0.2.0}/pyproject.toml +1 -1
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/cli.py +113 -47
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/config.py +36 -14
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/compaction.py +17 -5
- rockycode-0.2.0/rockycode/engine/cron.py +527 -0
- rockycode-0.2.0/rockycode/engine/effort.py +122 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/events.py +15 -2
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/headless.py +68 -11
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/images.py +57 -4
- rockycode-0.2.0/rockycode/engine/localcheck.py +166 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/loop.py +289 -43
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/permission.py +16 -0
- rockycode-0.2.0/rockycode/engine/providers.py +751 -0
- rockycode-0.2.0/rockycode/engine/selfconfig.py +164 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/server.py +2 -2
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/titler.py +26 -4
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/tools.py +35 -13
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/vision.py +27 -14
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/web.py +7 -1
- rockycode-0.2.0/rockycode/models.toml +239 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/onboarding.py +12 -7
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/pricing.py +84 -56
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/prompts/rocky.py +3 -0
- rockycode-0.2.0/rockycode/skills/rocky-setup/SKILL.md +57 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/app.py +737 -57
- rockycode-0.2.0/rockycode/tui/loopcard.py +135 -0
- rockycode-0.2.0/rockycode/tui/modelpicker.py +246 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/permission.py +108 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/package.json +1 -1
- rockycode-0.2.0/tests/smoke_cron.py +388 -0
- rockycode-0.2.0/tests/smoke_effort.py +61 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_engine.py +50 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_exec.py +47 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_images.py +54 -10
- rockycode-0.2.0/tests/smoke_local_models.py +260 -0
- rockycode-0.2.0/tests/smoke_models_registry.py +152 -0
- rockycode-0.2.0/tests/smoke_pricing.py +117 -0
- rockycode-0.2.0/tests/smoke_providers.py +237 -0
- rockycode-0.2.0/tests/smoke_selfconfig.py +85 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_serve.py +3 -3
- rockycode-0.2.0/tests/smoke_tui_loop.py +588 -0
- rockycode-0.2.0/tests/smoke_tui_model.py +130 -0
- rockycode-0.2.0/tests/smoke_tui_modeswitch.py +173 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_permission.py +1 -1
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_toggle.py +1 -1
- {rockycode-0.1.2 → rockycode-0.2.0}/uv.lock +1 -1
- rockycode-0.1.2/CHANGELOG.md +0 -61
- rockycode-0.1.2/rockycode/engine/effort.py +0 -46
- rockycode-0.1.2/rockycode/engine/providers.py +0 -209
- rockycode-0.1.2/rockycode/tui/modelpicker.py +0 -116
- rockycode-0.1.2/tests/smoke_effort.py +0 -45
- rockycode-0.1.2/tests/smoke_pricing.py +0 -95
- rockycode-0.1.2/tests/smoke_providers.py +0 -120
- rockycode-0.1.2/tests/smoke_tui_model.py +0 -81
- {rockycode-0.1.2 → rockycode-0.2.0}/.dockerignore +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/.github/workflows/ci.yml +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/.github/workflows/release.yml +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/.gitignore +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/CONTRIBUTING.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/Dockerfile +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/Dockerfile.sandbox +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/LICENSE +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/SECURITY.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/bench/tasks/dev10.json +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/bench/tasks/test20.json +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/brand/rockycode-note.svg +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/brand/rockycode-wordmark.svg +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/docker-compose.yml +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/prompts/README.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-v1.txt +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-v2-search-first.txt +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-v3-decisive.txt +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-zh-closer.txt +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-zh-full-closer.txt +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/prompts/rocky-zh-full.txt +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/__init__.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/banner.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/__init__.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/core.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/judge.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/mining.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/dream/proposals.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/__init__.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/artifact.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/budget.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/checks.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/container.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/explore.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/goal.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/goal_review.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/goal_session.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/lsp.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/mcp.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/modes.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/outcome.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/planmode.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/redact.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/safety.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/sandbox.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/skills.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/trajectory.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/engine/worktree.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/memory/__init__.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/memory/index.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/memory/store.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/learn/learn.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/deep-research.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/paper-reading.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/prove.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/modes/research/whiteboard.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/palette.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/prompts/__init__.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/routines.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/__init__.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/agent.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/data.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/runners/raw.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/score.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/session.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/architecture-viz/SKILL.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/architecture-viz/template.html +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/lean-prover/SKILL.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/skills/lean-prover/torchlean-api.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/__init__.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/clipboard.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/exitsheet.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/goal_screen.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/mdterm.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/mdview.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/modepicker.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/plangate.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/prompt_history.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/proposalcard.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/resume.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/rocky_pet.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode/tui/routinecard.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.gitignore +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.vscode/launch.json +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.vscode/tasks.json +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/.vscodeignore +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/CHANGELOG.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/LICENSE +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/README.md +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/esbuild.config.mjs +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/chat.html +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/icon-marketplace.svg +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/icon.png +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/marked.js +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/media/rocky-icon.svg +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/package-lock.json +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/artifactTree.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/diffManager.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/editorContext.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/extension.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/permissionManager.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/protocol.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/rockyConnection.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/rockyProvider.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/statusBar.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/src/webview-highlight.ts +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/rockycode-vscode/tsconfig.json +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/fake_lsp_server.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/fake_mcp_server.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/invariants.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/real_planmode.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/real_reasoning_roundtrip.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/real_tool_contract.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/run_all.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/run_real.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_approver.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_artifact.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_bash.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_bash_grant.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_bilang_prompt.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_budget.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_checks.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_compaction.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_container.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_credentials.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_docker_sandbox_cancel.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_dream.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_dream_trigger.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_explore.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal_driver.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal_review.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_goal_runner.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_hardkill_resume.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_interrupt.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_judge.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_lsp.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mcp.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mdterm.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mdview.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_memory.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_memory_index.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_mining.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_modes.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_onboarding.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_outcome.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_parallel.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_permission.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_planmode.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_prompt_history.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_proposals.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_redact.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_resume_handoff.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_routine_run.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_routines.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_safety.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_score.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_session.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_skills.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_submit_race.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_today.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tools.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tools_jail.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_trajectory.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_artifact_modal.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_bash_gate.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_copy.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_envwarn.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_exitsheet.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_goal_screen.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_input_nav.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_plan.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_proposals.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_read_grant.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_routines.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_scroll.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_tui_shell.py +0 -0
- {rockycode-0.1.2 → rockycode-0.2.0}/tests/smoke_web.py +0 -0
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to rockycode are recorded here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/), and the project follows
|
|
5
|
+
[Semantic Versioning](https://semver.org/) — pre-1.0, so the surface may still
|
|
6
|
+
change between minor versions.
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
## [0.2.0] — models are data, loops, approval switch
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
- **Loops** (`/loop`, alias `/cron`): re-fire one prompt into THIS chat on an
|
|
14
|
+
interval — `/loop 5m check whether results/run3 has a summary.json; if so
|
|
15
|
+
give me the headline numbers`. Each tick is an ordinary turn on the session
|
|
16
|
+
engine (it sees the whole conversation), runs only when the chat is idle,
|
|
17
|
+
ends with `LOOP QUIET / NOTE / DONE`, and never prompts: a call that needs
|
|
18
|
+
approval pauses the loop (`/loop allow` decides it, `/loop resume` retries,
|
|
19
|
+
`/permission yolo` or shift+tab resumes every approval-paused loop; a manual
|
|
20
|
+
`/loop pause` stays paused). Quiet ticks fold into one streak line and roll
|
|
21
|
+
out of live context (the trajectory keeps them). No cap unless you set one
|
|
22
|
+
(`for 3h`, `x12`, `max $2`); soft running-cost reminders otherwise. Rocky
|
|
23
|
+
can start one itself (`loop_start` / `loop_stop`, through the normal
|
|
24
|
+
approval). Session-only — a loop dies with the session; `/routines` stays
|
|
25
|
+
the cross-launch scheduler.
|
|
26
|
+
- **Models are data.** The whole model catalog now lives in
|
|
27
|
+
`rockycode/models.toml` — per provider a China base URL, a rocky-owned key
|
|
28
|
+
name, a reasoning wire shape (`thinking` · `effort` · `enable_thinking` ·
|
|
29
|
+
`minimax` · `openai` · `none`), its own effort tiers, whether thinking can
|
|
30
|
+
be switched off, and where usage reports cache hits; per model its context
|
|
31
|
+
window, output cap, vision flag, price tables (USD + CNY) and roles. The
|
|
32
|
+
engine consumes a `ModelSpec` and carries no model-specific numbers of its
|
|
33
|
+
own. `~/.rockycode/models.toml` (same shape) is deep-merged on top: add a
|
|
34
|
+
model, correct a limit, add a price, `hidden = true` to drop a row.
|
|
35
|
+
- The 2026-09 roster: `deepseek-flash` (V4.1 Flash — native vision, the
|
|
36
|
+
default and the sidecar/search model) + `deepseek-v4-pro`; `glm-5.3` +
|
|
37
|
+
`glm-5.3-flash`; `kimi-k3`; `minimax-m3`; `step-5-preview`; `qwen3.8-max`
|
|
38
|
+
+ `qwen3.8-flash`; `mimo-v2.6-pro`; `ollama` (live-discovered). Limits and
|
|
39
|
+
DeepSeek's V4.1 prices verified at each provider's docs on 2026-09-29.
|
|
40
|
+
- Subscription-plan endpoints as their own picker rows: `qwen-plan` (Bailian
|
|
41
|
+
Token Plan), `mimo-plan`, `stepfun-plan` — own URL, own key
|
|
42
|
+
(`ROCKYCODE_<PROVIDER>_PLAN_API_KEY`).
|
|
43
|
+
- Context window and output cap follow the active model: config
|
|
44
|
+
`context_window` / `max_tokens` default to `0` (= the registry value) and
|
|
45
|
+
re-pace on every `/model` switch; a positive number pins your own ceiling.
|
|
46
|
+
`/config context_window 0` un-pins.
|
|
47
|
+
- `/effort off|low|high|max` — the dial now reaches `low` (DeepSeek, GLM and
|
|
48
|
+
Kimi all take low|high|max), is clamped onto each provider's own tiers by
|
|
49
|
+
position at the wire (StepFun's `low|medium|high` gets `medium` for
|
|
50
|
+
rocky's `high`), and sends the lowest tier for `off` on a model that
|
|
51
|
+
can't disable thinking (GLM, Kimi). The status line says what is sent.
|
|
52
|
+
`xhigh` stays accepted as `max`.
|
|
53
|
+
- `image_route auto` (new default): a pasted image on a text-only model is
|
|
54
|
+
described silently by the registry's sidecar (`deepseek-flash` on the home
|
|
55
|
+
key) or your image CLI — no picker in the way; `ask` keeps the picker.
|
|
56
|
+
- **Rocky configures rocky.** A built-in `rocky-setup` skill plus the
|
|
57
|
+
ask-tier `rocky_config` tool (show · set · set_url · add_provider) let a
|
|
58
|
+
user ask rocky to use a proxy, add a vLLM box, change the default model or
|
|
59
|
+
a limit — writes go under `~/.rockycode` only, through the same validated
|
|
60
|
+
setters as the CLI. Key-shaped values are refused with the env var to use.
|
|
61
|
+
- `rockycode exec --profile read|write|full`: `read` (read_file/grep/glob/
|
|
62
|
+
view_image) and `write` (+ jailed write_file/edit_file) have no shell, so
|
|
63
|
+
they run on the host with no Docker and start instantly — the profiles a
|
|
64
|
+
calling agent (Claude Code, Codex) wants for "look and tell" and small
|
|
65
|
+
edits. `full` keeps bash in the Docker sandbox by default. The stream is
|
|
66
|
+
now meta → text → result by default; `--events` restores the tool.*/turn.*
|
|
67
|
+
receipt lines (`--include-thinking` implies it). `meta.profile` reports
|
|
68
|
+
name, mode and tool set.
|
|
69
|
+
- Peak-hour pricing honors DeepSeek's calendar: Monday–Friday only, with a
|
|
70
|
+
`holidays` list (ISO dates, Beijing) on the schedule for Chinese public
|
|
71
|
+
holidays — weekends and holidays bill off-peak.
|
|
72
|
+
- Approval mode is switchable without typing: `shift+tab` steps it one notch
|
|
73
|
+
looser (careful → ask → yolo, wrapping back to careful, so a run of presses
|
|
74
|
+
never parks you in yolo). The status-bar 🔒 chip now spells that key out and
|
|
75
|
+
opens a picker when clicked — three rows saying what each mode actually
|
|
76
|
+
allows — and bare `/permission` opens the same picker. The cycle stays inert
|
|
77
|
+
while an approval prompt is waiting, and landing in yolo still prints the
|
|
78
|
+
"runs on your machine" warning.
|
|
79
|
+
- Local models: a builtin `ollama` provider (`http://localhost:11434/v1`,
|
|
80
|
+
override with `ROCKYCODE_OLLAMA_URL`) — keyless, priced `$0 · local`, model
|
|
81
|
+
list discovered live from the running server's `GET /v1/models` so the
|
|
82
|
+
`/model` picker shows what's actually pulled. Server down → a dimmed
|
|
83
|
+
"not running · start it: ollama serve" row instead of silence.
|
|
84
|
+
- `/model` readiness preflight for local providers: before the engine is
|
|
85
|
+
switched, rocky checks server reachability, that the model is pulled, that
|
|
86
|
+
it supports tool calling, and the serving context length (Ollama truncates
|
|
87
|
+
silently — the #1 agent-loop killer). Not ready → the switch is refused and
|
|
88
|
+
every failed check carries its exact fix (`ollama pull …`,
|
|
89
|
+
`export OLLAMA_CONTEXT_LENGTH=65536`); ready → rocky's `context_window` is
|
|
90
|
+
paced to the server's verified value (restored on switching back to a cloud
|
|
91
|
+
provider), and a server-reported vision capability turns image input on.
|
|
92
|
+
- Keyless endpoints: `local = true` on a `~/.rockycode/providers.toml`
|
|
93
|
+
provider (LM Studio, llama.cpp, vLLM, a remote box's ollama) marks its
|
|
94
|
+
endpoints keyless — always configured, no placeholder-key hack needed.
|
|
95
|
+
- Compaction on a local provider sends no tool schemas in the summarize call
|
|
96
|
+
(local compat layers ignore `tool_choice="none"` and may answer with a tool
|
|
97
|
+
call instead of a summary), and the thinking-off field is now shaped per
|
|
98
|
+
the provider's reasoning policy instead of always DeepSeek's.
|
|
99
|
+
- Vision is now per-MODEL, not per-provider (`Provider.vision_models`,
|
|
100
|
+
choice-level `❖` badge). Config key `vision_models` marks additional ids as
|
|
101
|
+
image-capable from any shell (`rockycode config vision_models <id>`) — for
|
|
102
|
+
when a provider ships vision on an existing model before the registry
|
|
103
|
+
catches up.
|
|
104
|
+
- Config key `model`: a sticky launch default (`rockycode config model
|
|
105
|
+
<spec>`), resolved against the registry; precedence `--model` flag →
|
|
106
|
+
`ROCKYCODE_MODEL` env → config. Global config only — a cloned repo can
|
|
107
|
+
never redirect requests.
|
|
108
|
+
- The `/model` picker is now two-step, model first: one row per model, and a
|
|
109
|
+
model with several endpoints then asks which URL serves it — including a
|
|
110
|
+
"custom base URL" row that remembers your own gateway/proxy per provider
|
|
111
|
+
(`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`, riding
|
|
112
|
+
the provider's key).
|
|
113
|
+
- Images are best-effort downscaled to 2048px before hitting the wire (PIL if
|
|
114
|
+
installed, macOS `sips` otherwise, original on any failure) — vision
|
|
115
|
+
providers bill image tokens by dimensions, and a Retina screenshot was
|
|
116
|
+
paying severalfold for nothing.
|
|
117
|
+
|
|
118
|
+
### Changed
|
|
119
|
+
- One China endpoint per provider (no more `kimi-cn`/`kimi-en`, `zai`/`glm-cn`,
|
|
120
|
+
`minimax-cn`/`-en`): `/model kimi`, `/model glm`, `/model minimax`. Keys are
|
|
121
|
+
`ROCKYCODE_<PROVIDER>_API_KEY`; the older `_CN_`/`_EN_`/`ZAI_EN` names are
|
|
122
|
+
still read as aliases (the picker names the alias in use), so no setup
|
|
123
|
+
breaks. An international or proxy URL is the custom-URL row.
|
|
124
|
+
- `deepseek-v4-flash` and `deepseek-v4-flash-vision-exp` are retired from
|
|
125
|
+
the list (DeepSeek retired the models on 2026-09-10 and serves those names
|
|
126
|
+
with V4.1-Flash); rocky aliases both to `deepseek-flash`, so old configs,
|
|
127
|
+
trajectories and `/model` habits keep working. `deepseek-v4-pro` stays
|
|
128
|
+
listed, labelled: since 2026-09-14 DeepSeek routes it to V4.1-Flash at
|
|
129
|
+
Flash rates until V4.1-Pro ships.
|
|
130
|
+
- `glm-5.2` and `step-3.7-flash` drop off the roster in favor of GLM-5.3 /
|
|
131
|
+
Step 5 Preview.
|
|
132
|
+
- Session titles, image describe, and the native web search all use the
|
|
133
|
+
registry's `sidecar` / `search` role (deepseek-flash) — a session on Kimi
|
|
134
|
+
or GLM no longer sends a DeepSeek model id down its own client for the
|
|
135
|
+
title call.
|
|
136
|
+
- `--max-tokens` / `--context-window` default to `0` (= the model's registry
|
|
137
|
+
value) on `chat` and `serve`; `bench` keeps its pinned reproducible numbers.
|
|
138
|
+
- The CLI help, status line and context reminder no longer speak of
|
|
139
|
+
"DeepSeek V4" as the only model.
|
|
140
|
+
- `view_image` on a vision-capable active model now attaches the real image
|
|
141
|
+
to the conversation (as the next user message) instead of a sidecar text
|
|
142
|
+
description — the model reads the pixels and decides what matters itself.
|
|
143
|
+
Text-only models keep the describe routes unchanged.
|
|
144
|
+
- Launch honors the registry: starting with `--model minimax-m3` (or any
|
|
145
|
+
registry model) now gets the right reasoning params, tools flag, and vision
|
|
146
|
+
capability from step one instead of DeepSeek-shaped defaults until the
|
|
147
|
+
first `/model` switch. Unresolved specs and injected clients behave exactly
|
|
148
|
+
as before.
|
|
149
|
+
- `/model` spec resolution prefers exact ids, and a base model wins its own
|
|
150
|
+
substring — `glm:5.3` means `glm-5.3`, not ambiguity with `glm-5.3-flash`;
|
|
151
|
+
the variant stays reachable via `glm:flash`.
|
|
152
|
+
|
|
153
|
+
### Fixed
|
|
154
|
+
- The prompt-cache observer and the cost ledger read cache hits from
|
|
155
|
+
OpenAI-style `prompt_tokens_details.cached_tokens` too (Kimi, GLM, MiniMax,
|
|
156
|
+
Qwen, MiMo), not only DeepSeek's `prompt_cache_hit_tokens`.
|
|
157
|
+
- Launching on a `<provider>-custom` endpoint of the home provider now uses
|
|
158
|
+
that URL (it used to fall back to the env base URL).
|
|
159
|
+
|
|
160
|
+
## [0.1.2] — `--version` flag, GA price tables
|
|
161
|
+
|
|
162
|
+
### Added
|
|
163
|
+
- `rockycode --version` / `-V` prints the installed version. One version
|
|
164
|
+
source: package metadata (pyproject) — the serve handshake reports the same
|
|
165
|
+
value instead of a hardcoded string.
|
|
166
|
+
|
|
167
|
+
### Changed
|
|
168
|
+
- DeepSeek price tables refreshed to the GA snapshots (V4-Flash-0731 /
|
|
169
|
+
V4-Pro-0813), both USD and CNY, verified 2026-08-20 at the source;
|
|
170
|
+
peak-valley billing confirmed live (2× in the published UTC windows).
|
|
171
|
+
- README results updated: `deepseek-v4-flash` GA carries three clean full-500
|
|
172
|
+
SWE-bench Verified rounds (88.8% average, 95.4% pass@3); the
|
|
173
|
+
`deepseek-v4-pro` column is explicitly marked preview (pre-0813).
|
|
174
|
+
|
|
175
|
+
### CI
|
|
176
|
+
- Release workflow reduced to the single ubuntu PyPI Trusted-Publishing job;
|
|
177
|
+
the dead macos-13 matrix (retired runner, never ran) is gone.
|
|
178
|
+
|
|
179
|
+
## [0.1.1] — session artifacts, images in chat, real sandbox cancel
|
|
180
|
+
|
|
181
|
+
### Added
|
|
182
|
+
- Images in chat: paste (`ctrl+v` / `/paste`) or drag an image in. Vision models
|
|
183
|
+
see it raw; no-vision models route through a provider sidecar, your own image
|
|
184
|
+
CLI, or a `view_image` tool — picked once, remembered. Known CLIs set up with
|
|
185
|
+
one word (`/config image_cli mmx` — rocky knows the invocation).
|
|
186
|
+
- Session artifact inventory: `/artifact list · open <n> · stop · live on|off`,
|
|
187
|
+
a footer badge with open-tab counts, and an Artifacts tree in the VS Code
|
|
188
|
+
extension fed live by `rockycode serve`.
|
|
189
|
+
- Bare `/model` opens a live provider + model picker.
|
|
190
|
+
|
|
191
|
+
### Fixed
|
|
192
|
+
- Live artifacts no longer drop and reconnect every 30 s; the artifact server
|
|
193
|
+
stops/restarts cleanly and rebinds saved live pages to the new port.
|
|
194
|
+
- Sandbox cancel/timeout kills the in-container process group, not just the
|
|
195
|
+
host-side docker client; images without `python3` fall back to plain `bash -c`.
|
|
196
|
+
- Resume self-heals after a hard kill; closing the doc dock no longer wedges the TUI.
|
|
197
|
+
|
|
198
|
+
### Internal
|
|
199
|
+
- CI: conservative ruff correctness gate (pinned `0.16.1`) ahead of the smoke suite.
|
|
200
|
+
|
|
201
|
+
## [0.1.0] — first public release
|
|
202
|
+
|
|
203
|
+
Initial public release. One repo, one engine, three ways to use it — interactive
|
|
204
|
+
`chat`, autonomous `goal`, and the `bench` measurement rig — running on DeepSeek
|
|
205
|
+
or any OpenAI-compatible model.
|
|
206
|
+
|
|
207
|
+
Some capabilities ship as **experimental and default-off** (self-improvement,
|
|
208
|
+
`prove` / `lean-prover`, `explore`, and providers other than DeepSeek); see the
|
|
209
|
+
README's Experimental section for what they are and how to enable them.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: rockycode
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A coding agent harness, benchmarked on SWE-bench Verified. amaze!
|
|
5
5
|
Author: rockycode contributors
|
|
6
6
|
License: MIT
|
|
@@ -205,8 +205,8 @@ clipboard" (or your terminal's equivalent) on the local end.
|
|
|
205
205
|
| `/research` | Research modes: deep-research · paper-reading · whiteboard · prove |
|
|
206
206
|
| `/learn` | Tutor mode — your understanding is the goal, not the diff |
|
|
207
207
|
| `/model` | Switch provider and model (see below) |
|
|
208
|
-
| `/effort off\|high\|
|
|
209
|
-
| `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
|
|
208
|
+
| `/effort off\|low\|high\|max` | Reasoning depth, adjustable live per session (clamped onto each provider's own tiers) |
|
|
209
|
+
| `/permission yolo\|ask\|careful` | Tool-approval strictness for the session — bare opens a picker; `shift+tab` cycles it, or click the 🔒 chip in the status bar |
|
|
210
210
|
| `/sandbox on\|off\|status` | Isolate tool execution in a container |
|
|
211
211
|
| `/lsp` | Language-server status; diagnostics ride along with `read_file` |
|
|
212
212
|
| `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
|
|
@@ -238,27 +238,79 @@ clipboard" (or your terminal's equivalent) on the local end.
|
|
|
238
238
|
|
|
239
239
|
### Models and providers
|
|
240
240
|
|
|
241
|
-
DeepSeek is the home model, but
|
|
242
|
-
|
|
243
|
-
|
|
241
|
+
DeepSeek is the home model, but **models are data, not code**: the whole
|
|
242
|
+
catalog lives in `rockycode/models.toml` — per provider a China base URL, a
|
|
243
|
+
key name and a reasoning wire shape; per model its context window, output
|
|
244
|
+
cap, vision flag, price and roles. The engine reads that spec and carries no
|
|
245
|
+
model-specific numbers of its own, so a new model is a data edit. Your own
|
|
246
|
+
`~/.rockycode/models.toml` (same shape) is deep-merged on top: add a model,
|
|
247
|
+
correct a limit, add a price, hide a row.
|
|
244
248
|
|
|
245
|
-
| Provider | Models |
|
|
246
|
-
|
|
247
|
-
| **deepseek** (default) | `deepseek-
|
|
248
|
-
| **
|
|
249
|
-
| **kimi** | `kimi-k3` |
|
|
250
|
-
| **
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
249
|
+
| Provider | Models (❖ = takes image input) | ctx / max out |
|
|
250
|
+
|---|---|---|
|
|
251
|
+
| **deepseek** (default) | `deepseek-flash` (V4.1 Flash, default) ❖, `deepseek-v4-pro` | 1M / 384K |
|
|
252
|
+
| **glm** | `glm-5.3`, `glm-5.3-flash` ❖ | 1M / 128K |
|
|
253
|
+
| **kimi** | `kimi-k3` ❖ | 1M / 128K |
|
|
254
|
+
| **minimax** | `minimax-m3` ❖ | 1M / 128K |
|
|
255
|
+
| **stepfun** | `step-5-preview` ❖ | 1M / 64K |
|
|
256
|
+
| **qwen** | `qwen3.8-max` ❖, `qwen3.8-flash` ❖ | 1M / 64K |
|
|
257
|
+
| **mimo** | `mimo-v2.6-pro` ❖ | 1M / 128K |
|
|
258
|
+
| **ollama** (local, $0) | whatever you've pulled — discovered live from the running server | server-verified |
|
|
259
|
+
|
|
260
|
+
One China endpoint per provider (`ROCKYCODE_<PROVIDER>_API_KEY`; the older
|
|
261
|
+
`_CN_`/`_EN_` names are still read). Subscription plans with their own URL
|
|
262
|
+
and key are their own rows — `qwen-plan` (Bailian Token Plan), `mimo-plan`,
|
|
263
|
+
`stepfun-plan` — keyed as `ROCKYCODE_<PROVIDER>_PLAN_API_KEY`. The `/model`
|
|
264
|
+
picker lists **models first**, one row each; a model with several endpoints
|
|
265
|
+
then asks which URL serves it (official · plan · your own), and the "custom
|
|
266
|
+
base URL" row remembers a gateway or proxy per provider
|
|
267
|
+
(`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`). Typed
|
|
268
|
+
specs skip all of that: `/model glm:flash`, `/model qwen-plan:qwen3.8-max`.
|
|
269
|
+
The retired `deepseek-v4-flash` / `-vision-exp` ids still resolve (to
|
|
270
|
+
`deepseek-flash`, exactly as DeepSeek serves them). The picker only offers
|
|
271
|
+
providers whose keys are actually configured.
|
|
272
|
+
|
|
273
|
+
Vision is per-model: `deepseek-flash` sees images on the home key, so the
|
|
274
|
+
default session just takes a paste. A text-only model (`deepseek-v4-pro`,
|
|
275
|
+
`glm-5.3`) gets pasted images described by `deepseek-flash` silently
|
|
276
|
+
(`image_route auto`), or by your own CLI. `rockycode config model <spec>`
|
|
277
|
+
makes any pick the sticky launch default. Context window and output cap
|
|
278
|
+
follow the active model (config `context_window` / `max_tokens` = `0`); a
|
|
279
|
+
number pins your own ceiling across switches. DeepSeek and MiniMax carry
|
|
280
|
+
full-500 bench numbers (see [Results](#results--swe-bench-verified)); the
|
|
281
|
+
other providers are [experimental](#experimental).
|
|
282
|
+
|
|
283
|
+
The effort dial (`/effort off|low|high|max`) is provider-neutral; each
|
|
284
|
+
provider's own tiers come from the registry and the dial is clamped onto
|
|
285
|
+
them by position at the wire (StepFun's `low|medium|high` gets `medium` for
|
|
286
|
+
rocky's `high`; GLM and Kimi can't switch thinking off, so `off` sends their
|
|
287
|
+
lowest tier). `xhigh` is still accepted and means `max`.
|
|
288
|
+
|
|
289
|
+
Rocky can also configure itself: ask it to "use my proxy", "add my vLLM
|
|
290
|
+
server", or "switch the default model" and the built-in `rocky-setup` skill
|
|
291
|
+
plus the ask-tier `rocky_config` tool make the change under `~/.rockycode`.
|
|
292
|
+
Keys are the one thing it never touches — it names the variable and you
|
|
293
|
+
paste the value.
|
|
294
|
+
|
|
295
|
+
#### Local models (Ollama)
|
|
296
|
+
|
|
297
|
+
Rocky integrates the OpenAI-compatible *protocol*, never a runtime — and
|
|
298
|
+
recommends [Ollama](https://ollama.com) (its MLX engine covers Apple Silicon
|
|
299
|
+
since 0.19). No key, no config: run `ollama serve`, pull a tool-capable model
|
|
300
|
+
(`ollama pull qwen3.8:27b-mlx` is the tested recommendation), and it appears
|
|
301
|
+
in `/model` with what's actually pulled, priced `$0 · local`.
|
|
302
|
+
|
|
303
|
+
Switching to a local model runs a **readiness preflight** first — server up,
|
|
304
|
+
model pulled, tool-calling support, serving context — and refuses the switch
|
|
305
|
+
with the exact fix (`ollama pull …`, `export OLLAMA_CONTEXT_LENGTH=65536`)
|
|
306
|
+
when something would break mid-session. The one to respect: Ollama's default
|
|
307
|
+
context is small and it **truncates silently**, which kills agent sessions in
|
|
308
|
+
confusing ways — serve with `OLLAMA_CONTEXT_LENGTH=65536`. Rocky paces its own
|
|
309
|
+
`context_window` to the server's verified value on every switch.
|
|
310
|
+
|
|
311
|
+
Other local servers (LM Studio, llama.cpp, vLLM) work as data too: add a
|
|
312
|
+
provider with `local = true` in `~/.rockycode/providers.toml` and its
|
|
313
|
+
endpoints need no key.
|
|
262
314
|
|
|
263
315
|
## Autonomous use
|
|
264
316
|
|
|
@@ -288,11 +340,26 @@ rockycode goal "add a docstring to <fn> and run the linter" --max-usd 0.50 --max
|
|
|
288
340
|
### Headless delegation: `exec`
|
|
289
341
|
|
|
290
342
|
`rockycode exec "<task>"` is the single-shot, non-interactive entry point,
|
|
291
|
-
designed to be called by *other* agents and scripts.
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
343
|
+
designed to be called by *other* agents and scripts. stdout is JSONL: a
|
|
344
|
+
`meta` line, the model's `text`, and a `result` envelope with evidence
|
|
345
|
+
(files changed, commands run, refusals) — never verdicts, the caller
|
|
346
|
+
verifies; `--events` adds the per-tool receipt lines. Budgets are always
|
|
347
|
+
enforced, and exit codes distinguish success, failure, needs-approval, and
|
|
348
|
+
budget-stop — so a calling agent can grant an approval and resume instead
|
|
349
|
+
of guessing.
|
|
350
|
+
|
|
351
|
+
Pick how much rocky may do with `--profile`: `read` (read_file / grep /
|
|
352
|
+
glob / view_image — no shell, no writes) and `write` (+ write_file /
|
|
353
|
+
edit_file jailed to `--workdir`) run on the host with **no Docker** and
|
|
354
|
+
start instantly — what Claude Code or Codex wants for "look at this repo and
|
|
355
|
+
tell me" or a small edit on a cheap, fast model. `full` adds bash, in the
|
|
356
|
+
Docker sandbox **by default** (the command classifier is defense-in-depth,
|
|
357
|
+
not the boundary).
|
|
358
|
+
|
|
359
|
+
```bash
|
|
360
|
+
rockycode exec --profile read "which module owns retry logic, and where is it called?"
|
|
361
|
+
rockycode exec --profile write "add a docstring to every public function in utils.py"
|
|
362
|
+
```
|
|
296
363
|
|
|
297
364
|
### Editor integration: `serve` and the VS Code extension
|
|
298
365
|
|
|
@@ -347,10 +414,13 @@ change. Anything that could act on its own is **off by default**.
|
|
|
347
414
|
investigation from a fresh-context child that returns only a cited,
|
|
348
415
|
mechanically-verified report; the search noise never enters your session. It
|
|
349
416
|
also grounds goal mode's branch review and milestone verification.
|
|
350
|
-
- **Providers beyond DeepSeek.**
|
|
351
|
-
OpenAI-compatible
|
|
352
|
-
bench numbers (see Results); treat
|
|
353
|
-
too.
|
|
417
|
+
- **Providers beyond DeepSeek.** GLM, Kimi, MiniMax, StepFun, Qwen, and MiMo
|
|
418
|
+
are wired as OpenAI-compatible registry entries (`/model`). DeepSeek and
|
|
419
|
+
MiniMax carry full bench numbers (see Results); treat the rest as untested
|
|
420
|
+
until they do too. Registry entries marked `note = "… verify …"` in
|
|
421
|
+
`models.toml` (MiniMax's endpoint host, MiMo's auth header, the plan URLs)
|
|
422
|
+
were taken from each provider's docs on 2026-09-29 and not yet exercised
|
|
423
|
+
live — a wrong one is a one-line data fix.
|
|
354
424
|
|
|
355
425
|
## Works with your existing setup
|
|
356
426
|
|
|
@@ -172,8 +172,8 @@ clipboard" (or your terminal's equivalent) on the local end.
|
|
|
172
172
|
| `/research` | Research modes: deep-research · paper-reading · whiteboard · prove |
|
|
173
173
|
| `/learn` | Tutor mode — your understanding is the goal, not the diff |
|
|
174
174
|
| `/model` | Switch provider and model (see below) |
|
|
175
|
-
| `/effort off\|high\|
|
|
176
|
-
| `/permission yolo\|ask\|careful` | Tool-approval strictness for the session |
|
|
175
|
+
| `/effort off\|low\|high\|max` | Reasoning depth, adjustable live per session (clamped onto each provider's own tiers) |
|
|
176
|
+
| `/permission yolo\|ask\|careful` | Tool-approval strictness for the session — bare opens a picker; `shift+tab` cycles it, or click the 🔒 chip in the status bar |
|
|
177
177
|
| `/sandbox on\|off\|status` | Isolate tool execution in a container |
|
|
178
178
|
| `/lsp` | Language-server status; diagnostics ride along with `read_file` |
|
|
179
179
|
| `/artifact` | Session artifacts: `list` · `open <n>` · `stop` · `live on\|off` |
|
|
@@ -205,27 +205,79 @@ clipboard" (or your terminal's equivalent) on the local end.
|
|
|
205
205
|
|
|
206
206
|
### Models and providers
|
|
207
207
|
|
|
208
|
-
DeepSeek is the home model, but
|
|
209
|
-
|
|
210
|
-
|
|
208
|
+
DeepSeek is the home model, but **models are data, not code**: the whole
|
|
209
|
+
catalog lives in `rockycode/models.toml` — per provider a China base URL, a
|
|
210
|
+
key name and a reasoning wire shape; per model its context window, output
|
|
211
|
+
cap, vision flag, price and roles. The engine reads that spec and carries no
|
|
212
|
+
model-specific numbers of its own, so a new model is a data edit. Your own
|
|
213
|
+
`~/.rockycode/models.toml` (same shape) is deep-merged on top: add a model,
|
|
214
|
+
correct a limit, add a price, hide a row.
|
|
211
215
|
|
|
212
|
-
| Provider | Models |
|
|
213
|
-
|
|
214
|
-
| **deepseek** (default) | `deepseek-
|
|
215
|
-
| **
|
|
216
|
-
| **kimi** | `kimi-k3` |
|
|
217
|
-
| **
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
216
|
+
| Provider | Models (❖ = takes image input) | ctx / max out |
|
|
217
|
+
|---|---|---|
|
|
218
|
+
| **deepseek** (default) | `deepseek-flash` (V4.1 Flash, default) ❖, `deepseek-v4-pro` | 1M / 384K |
|
|
219
|
+
| **glm** | `glm-5.3`, `glm-5.3-flash` ❖ | 1M / 128K |
|
|
220
|
+
| **kimi** | `kimi-k3` ❖ | 1M / 128K |
|
|
221
|
+
| **minimax** | `minimax-m3` ❖ | 1M / 128K |
|
|
222
|
+
| **stepfun** | `step-5-preview` ❖ | 1M / 64K |
|
|
223
|
+
| **qwen** | `qwen3.8-max` ❖, `qwen3.8-flash` ❖ | 1M / 64K |
|
|
224
|
+
| **mimo** | `mimo-v2.6-pro` ❖ | 1M / 128K |
|
|
225
|
+
| **ollama** (local, $0) | whatever you've pulled — discovered live from the running server | server-verified |
|
|
226
|
+
|
|
227
|
+
One China endpoint per provider (`ROCKYCODE_<PROVIDER>_API_KEY`; the older
|
|
228
|
+
`_CN_`/`_EN_` names are still read). Subscription plans with their own URL
|
|
229
|
+
and key are their own rows — `qwen-plan` (Bailian Token Plan), `mimo-plan`,
|
|
230
|
+
`stepfun-plan` — keyed as `ROCKYCODE_<PROVIDER>_PLAN_API_KEY`. The `/model`
|
|
231
|
+
picker lists **models first**, one row each; a model with several endpoints
|
|
232
|
+
then asks which URL serves it (official · plan · your own), and the "custom
|
|
233
|
+
base URL" row remembers a gateway or proxy per provider
|
|
234
|
+
(`~/.rockycode/endpoints.toml`, addressable as `<provider>-custom`). Typed
|
|
235
|
+
specs skip all of that: `/model glm:flash`, `/model qwen-plan:qwen3.8-max`.
|
|
236
|
+
The retired `deepseek-v4-flash` / `-vision-exp` ids still resolve (to
|
|
237
|
+
`deepseek-flash`, exactly as DeepSeek serves them). The picker only offers
|
|
238
|
+
providers whose keys are actually configured.
|
|
239
|
+
|
|
240
|
+
Vision is per-model: `deepseek-flash` sees images on the home key, so the
|
|
241
|
+
default session just takes a paste. A text-only model (`deepseek-v4-pro`,
|
|
242
|
+
`glm-5.3`) gets pasted images described by `deepseek-flash` silently
|
|
243
|
+
(`image_route auto`), or by your own CLI. `rockycode config model <spec>`
|
|
244
|
+
makes any pick the sticky launch default. Context window and output cap
|
|
245
|
+
follow the active model (config `context_window` / `max_tokens` = `0`); a
|
|
246
|
+
number pins your own ceiling across switches. DeepSeek and MiniMax carry
|
|
247
|
+
full-500 bench numbers (see [Results](#results--swe-bench-verified)); the
|
|
248
|
+
other providers are [experimental](#experimental).
|
|
249
|
+
|
|
250
|
+
The effort dial (`/effort off|low|high|max`) is provider-neutral; each
|
|
251
|
+
provider's own tiers come from the registry and the dial is clamped onto
|
|
252
|
+
them by position at the wire (StepFun's `low|medium|high` gets `medium` for
|
|
253
|
+
rocky's `high`; GLM and Kimi can't switch thinking off, so `off` sends their
|
|
254
|
+
lowest tier). `xhigh` is still accepted and means `max`.
|
|
255
|
+
|
|
256
|
+
Rocky can also configure itself: ask it to "use my proxy", "add my vLLM
|
|
257
|
+
server", or "switch the default model" and the built-in `rocky-setup` skill
|
|
258
|
+
plus the ask-tier `rocky_config` tool make the change under `~/.rockycode`.
|
|
259
|
+
Keys are the one thing it never touches — it names the variable and you
|
|
260
|
+
paste the value.
|
|
261
|
+
|
|
262
|
+
#### Local models (Ollama)
|
|
263
|
+
|
|
264
|
+
Rocky integrates the OpenAI-compatible *protocol*, never a runtime — and
|
|
265
|
+
recommends [Ollama](https://ollama.com) (its MLX engine covers Apple Silicon
|
|
266
|
+
since 0.19). No key, no config: run `ollama serve`, pull a tool-capable model
|
|
267
|
+
(`ollama pull qwen3.8:27b-mlx` is the tested recommendation), and it appears
|
|
268
|
+
in `/model` with what's actually pulled, priced `$0 · local`.
|
|
269
|
+
|
|
270
|
+
Switching to a local model runs a **readiness preflight** first — server up,
|
|
271
|
+
model pulled, tool-calling support, serving context — and refuses the switch
|
|
272
|
+
with the exact fix (`ollama pull …`, `export OLLAMA_CONTEXT_LENGTH=65536`)
|
|
273
|
+
when something would break mid-session. The one to respect: Ollama's default
|
|
274
|
+
context is small and it **truncates silently**, which kills agent sessions in
|
|
275
|
+
confusing ways — serve with `OLLAMA_CONTEXT_LENGTH=65536`. Rocky paces its own
|
|
276
|
+
`context_window` to the server's verified value on every switch.
|
|
277
|
+
|
|
278
|
+
Other local servers (LM Studio, llama.cpp, vLLM) work as data too: add a
|
|
279
|
+
provider with `local = true` in `~/.rockycode/providers.toml` and its
|
|
280
|
+
endpoints need no key.
|
|
229
281
|
|
|
230
282
|
## Autonomous use
|
|
231
283
|
|
|
@@ -255,11 +307,26 @@ rockycode goal "add a docstring to <fn> and run the linter" --max-usd 0.50 --max
|
|
|
255
307
|
### Headless delegation: `exec`
|
|
256
308
|
|
|
257
309
|
`rockycode exec "<task>"` is the single-shot, non-interactive entry point,
|
|
258
|
-
designed to be called by *other* agents and scripts.
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
310
|
+
designed to be called by *other* agents and scripts. stdout is JSONL: a
|
|
311
|
+
`meta` line, the model's `text`, and a `result` envelope with evidence
|
|
312
|
+
(files changed, commands run, refusals) — never verdicts, the caller
|
|
313
|
+
verifies; `--events` adds the per-tool receipt lines. Budgets are always
|
|
314
|
+
enforced, and exit codes distinguish success, failure, needs-approval, and
|
|
315
|
+
budget-stop — so a calling agent can grant an approval and resume instead
|
|
316
|
+
of guessing.
|
|
317
|
+
|
|
318
|
+
Pick how much rocky may do with `--profile`: `read` (read_file / grep /
|
|
319
|
+
glob / view_image — no shell, no writes) and `write` (+ write_file /
|
|
320
|
+
edit_file jailed to `--workdir`) run on the host with **no Docker** and
|
|
321
|
+
start instantly — what Claude Code or Codex wants for "look at this repo and
|
|
322
|
+
tell me" or a small edit on a cheap, fast model. `full` adds bash, in the
|
|
323
|
+
Docker sandbox **by default** (the command classifier is defense-in-depth,
|
|
324
|
+
not the boundary).
|
|
325
|
+
|
|
326
|
+
```bash
|
|
327
|
+
rockycode exec --profile read "which module owns retry logic, and where is it called?"
|
|
328
|
+
rockycode exec --profile write "add a docstring to every public function in utils.py"
|
|
329
|
+
```
|
|
263
330
|
|
|
264
331
|
### Editor integration: `serve` and the VS Code extension
|
|
265
332
|
|
|
@@ -314,10 +381,13 @@ change. Anything that could act on its own is **off by default**.
|
|
|
314
381
|
investigation from a fresh-context child that returns only a cited,
|
|
315
382
|
mechanically-verified report; the search noise never enters your session. It
|
|
316
383
|
also grounds goal mode's branch review and milestone verification.
|
|
317
|
-
- **Providers beyond DeepSeek.**
|
|
318
|
-
OpenAI-compatible
|
|
319
|
-
bench numbers (see Results); treat
|
|
320
|
-
too.
|
|
384
|
+
- **Providers beyond DeepSeek.** GLM, Kimi, MiniMax, StepFun, Qwen, and MiMo
|
|
385
|
+
are wired as OpenAI-compatible registry entries (`/model`). DeepSeek and
|
|
386
|
+
MiniMax carry full bench numbers (see Results); treat the rest as untested
|
|
387
|
+
until they do too. Registry entries marked `note = "… verify …"` in
|
|
388
|
+
`models.toml` (MiniMax's endpoint host, MiMo's auth header, the plan URLs)
|
|
389
|
+
were taken from each provider's docs on 2026-09-29 and not yet exercised
|
|
390
|
+
live — a wrong one is a one-line data fix.
|
|
321
391
|
|
|
322
392
|
## Works with your existing setup
|
|
323
393
|
|