rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/pricing.py
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""Per-model token pricing + a peak-aware session usage ledger.
|
|
2
|
+
|
|
3
|
+
DeepSeek publishes SEPARATE CNY and USD price tables (each set independently —
|
|
4
|
+
NOT a conversion of the other), and applies a peak-hour surcharge. So we keep
|
|
5
|
+
BOTH currencies' official numbers and price in whichever the user picked; we
|
|
6
|
+
never convert one into the other. Verify + update the numbers at the source when
|
|
7
|
+
DeepSeek changes them:
|
|
8
|
+
|
|
9
|
+
https://api-docs.deepseek.com/quick_start/pricing (USD table)
|
|
10
|
+
https://api-docs.deepseek.com/zh-cn/quick_start/pricing (CNY table)
|
|
11
|
+
|
|
12
|
+
Verified: 2026-07-04 (per 1,000,000 tokens).
|
|
13
|
+
|
|
14
|
+
To update WITHOUT editing the install (survives upgrades), drop a
|
|
15
|
+
~/.rockycode/pricing.toml that overrides any values below. It is a FILE on
|
|
16
|
+
purpose — never an env var: env is dumpable (and now redacted), the wrong place
|
|
17
|
+
for maintained config. `rockycode pricing` prints the live table + this path.
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import copy
|
|
22
|
+
import tomllib
|
|
23
|
+
from datetime import datetime, time as dtime, timezone
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
PRICING_SOURCE_URL = "https://api-docs.deepseek.com/quick_start/pricing"
|
|
27
|
+
PRICING_VERIFIED = "2026-07-04"
|
|
28
|
+
OVERRIDE_PATH = Path.home() / ".rockycode" / "pricing.toml"
|
|
29
|
+
|
|
30
|
+
# Official list prices, per 1M tokens, each currency from its OWN table.
|
|
31
|
+
DEFAULT_PRICING: dict = {
|
|
32
|
+
"peak": {
|
|
33
|
+
# DeepSeek peak-valley pricing (announced): peak-hour rate = 2x regular,
|
|
34
|
+
# for ALL billing items, during these UTC windows — 01:00–04:00 and
|
|
35
|
+
# 06:00–10:00. It starts MID-JULY 2026, so the logic is wired but gated
|
|
36
|
+
# behind effective_date: before that date nothing is surcharged even
|
|
37
|
+
# inside a window, and it auto-activates on the date with no code change.
|
|
38
|
+
# Confirm the exact start date and adjust here or in the override file.
|
|
39
|
+
"enabled": True,
|
|
40
|
+
"effective_date": "2026-07-15", # UTC; peak surcharge does not apply before this
|
|
41
|
+
"multiplier": 2.0,
|
|
42
|
+
"windows_utc": [
|
|
43
|
+
{"start": "01:00", "end": "04:00"},
|
|
44
|
+
{"start": "06:00", "end": "10:00"},
|
|
45
|
+
],
|
|
46
|
+
},
|
|
47
|
+
"models": {
|
|
48
|
+
"deepseek-v4-pro": {
|
|
49
|
+
"usd": {"in_hit": 0.003625, "in_miss": 0.435, "out": 0.87},
|
|
50
|
+
"cny": {"in_hit": 0.025, "in_miss": 3.0, "out": 6.0},
|
|
51
|
+
},
|
|
52
|
+
"deepseek-v4-flash": {
|
|
53
|
+
"usd": {"in_hit": 0.0028, "in_miss": 0.14, "out": 0.28},
|
|
54
|
+
"cny": {"in_hit": 0.02, "in_miss": 1.0, "out": 2.0},
|
|
55
|
+
},
|
|
56
|
+
},
|
|
57
|
+
"fallback_model": "deepseek-v4-pro", # unknown model → price as pro (conservative)
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _deep_merge(base: dict, over: dict) -> None:
|
|
62
|
+
for k, v in over.items():
|
|
63
|
+
if isinstance(v, dict) and isinstance(base.get(k), dict):
|
|
64
|
+
_deep_merge(base[k], v)
|
|
65
|
+
else:
|
|
66
|
+
base[k] = v
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def load_pricing(override_path: Path | None = None) -> dict:
|
|
70
|
+
"""Built-in defaults, with ~/.rockycode/pricing.toml merged on top if present."""
|
|
71
|
+
pricing = copy.deepcopy(DEFAULT_PRICING)
|
|
72
|
+
path = OVERRIDE_PATH if override_path is None else override_path
|
|
73
|
+
if path.exists():
|
|
74
|
+
try:
|
|
75
|
+
_deep_merge(pricing, tomllib.loads(path.read_text()))
|
|
76
|
+
except (tomllib.TOMLDecodeError, OSError):
|
|
77
|
+
pass # a broken override must never break the cost display
|
|
78
|
+
return pricing
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _hhmm(s: str) -> dtime:
|
|
82
|
+
h, m = s.split(":")
|
|
83
|
+
return dtime(int(h), int(m))
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _is_peak(at: datetime, peak: dict) -> bool:
|
|
87
|
+
"""Is *at* inside a peak window AND on/after the effective date? Peak-valley
|
|
88
|
+
pricing starts mid-July, so before effective_date nothing is surcharged even
|
|
89
|
+
inside a window. False when peak is disabled."""
|
|
90
|
+
if not peak or not peak.get("enabled"):
|
|
91
|
+
return False
|
|
92
|
+
at_utc = at.astimezone(timezone.utc)
|
|
93
|
+
eff = peak.get("effective_date")
|
|
94
|
+
if eff:
|
|
95
|
+
try:
|
|
96
|
+
if at_utc < datetime.fromisoformat(eff).replace(tzinfo=timezone.utc):
|
|
97
|
+
return False # peak-valley pricing not yet in effect
|
|
98
|
+
except ValueError:
|
|
99
|
+
pass
|
|
100
|
+
now = at_utc.time()
|
|
101
|
+
for w in peak.get("windows_utc", []):
|
|
102
|
+
start, end = _hhmm(w["start"]), _hhmm(w["end"])
|
|
103
|
+
inside = (start <= now < end) if start <= end else (now >= start or now < end)
|
|
104
|
+
if inside:
|
|
105
|
+
return True
|
|
106
|
+
return False
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class UsageLedger:
|
|
110
|
+
"""Accumulates token usage per (model, peak?) so cost() can price each
|
|
111
|
+
currency from its own table and apply the peak multiplier per request."""
|
|
112
|
+
|
|
113
|
+
def __init__(self, pricing: dict | None = None) -> None:
|
|
114
|
+
self.pricing = pricing if pricing is not None else load_pricing()
|
|
115
|
+
self.buckets: dict[tuple[str, bool], dict] = {} # (model, is_peak) -> counts
|
|
116
|
+
|
|
117
|
+
def add(self, model: str, usage: dict, at: datetime | None = None) -> None:
|
|
118
|
+
"""Record a call's usage. *at* (defaults to now) decides peak vs off-peak,
|
|
119
|
+
so each request is priced at the rate in effect when it was made. The
|
|
120
|
+
peak surcharge is DeepSeek's OWN billing scheme, so it applies only to
|
|
121
|
+
DeepSeek models — a MiniMax/GLM turn is never peak-multiplied."""
|
|
122
|
+
if not usage:
|
|
123
|
+
return
|
|
124
|
+
at = at or datetime.now(timezone.utc)
|
|
125
|
+
peak = model.startswith("deepseek") and _is_peak(at, self.pricing.get("peak", {}))
|
|
126
|
+
b = self.buckets.setdefault((model, peak), {"prompt": 0, "hit": 0, "completion": 0})
|
|
127
|
+
b["prompt"] += usage.get("prompt_tokens", 0) or 0
|
|
128
|
+
b["hit"] += usage.get("prompt_cache_hit_tokens", 0) or 0
|
|
129
|
+
b["completion"] += usage.get("completion_tokens", 0) or 0
|
|
130
|
+
|
|
131
|
+
def priced(self, model: str) -> bool:
|
|
132
|
+
"""True if this exact model has its OWN API-fee entry. A model rocky has
|
|
133
|
+
no rate for (a just-added provider) is NOT silently priced as DeepSeek —
|
|
134
|
+
it prices at 0 and flags unset, so cross-provider cost stays honest."""
|
|
135
|
+
return model in self.pricing["models"]
|
|
136
|
+
|
|
137
|
+
def rate(self, model: str, currency: str = "usd") -> dict:
|
|
138
|
+
"""This model's per-1M-token rate in *currency* (in_hit/in_miss/out),
|
|
139
|
+
or the fallback's if unpriced (callers gate on priced())."""
|
|
140
|
+
m = self.pricing["models"].get(model) \
|
|
141
|
+
or self.pricing["models"][self.pricing.get("fallback_model", "deepseek-v4-pro")]
|
|
142
|
+
return m.get(currency) or m["usd"]
|
|
143
|
+
|
|
144
|
+
def _rates(self, model: str, currency: str) -> tuple[dict, bool]:
|
|
145
|
+
# Unpriced model → zero rate, flagged unconfigured (not DeepSeek's price):
|
|
146
|
+
# a MiniMax turn must not read as DeepSeek dollars. The user adds the
|
|
147
|
+
# provider's real API fee to ~/.rockycode/pricing.toml.
|
|
148
|
+
m = self.pricing["models"].get(model)
|
|
149
|
+
if m is None:
|
|
150
|
+
return {"in_hit": 0.0, "in_miss": 0.0, "out": 0.0}, False
|
|
151
|
+
r = m.get(currency)
|
|
152
|
+
if r:
|
|
153
|
+
return r, True
|
|
154
|
+
return m["usd"], False # currency not configured for this model → fall back + flag
|
|
155
|
+
|
|
156
|
+
def cost(self, currency: str = "usd") -> float:
|
|
157
|
+
mult = self.pricing.get("peak", {}).get("multiplier", 1.0)
|
|
158
|
+
total = 0.0
|
|
159
|
+
for (model, peak), b in self.buckets.items():
|
|
160
|
+
r, _ = self._rates(model, currency)
|
|
161
|
+
miss = max(0, b["prompt"] - b["hit"])
|
|
162
|
+
base = (miss * r["in_miss"] + b["hit"] * r["in_hit"] + b["completion"] * r["out"]) / 1e6
|
|
163
|
+
total += base * (mult if peak else 1.0)
|
|
164
|
+
return total
|
|
165
|
+
|
|
166
|
+
def cost_usd(self) -> float: # back-compat alias
|
|
167
|
+
return self.cost("usd")
|
|
168
|
+
|
|
169
|
+
def configured(self, currency: str) -> bool:
|
|
170
|
+
"""True if every model used has its OWN rates for *currency* (no fallback)."""
|
|
171
|
+
return all(self._rates(model, currency)[1] for (model, _peak) in self.buckets)
|
|
172
|
+
|
|
173
|
+
def totals(self) -> dict:
|
|
174
|
+
agg = {"prompt": 0, "hit": 0, "completion": 0}
|
|
175
|
+
for b in self.buckets.values():
|
|
176
|
+
for k in agg:
|
|
177
|
+
agg[k] += b[k]
|
|
178
|
+
return agg
|
|
File without changes
|
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""Prompt templates.
|
|
2
|
+
|
|
3
|
+
RAW_SINGLE_SHOT — used by the v0 raw runner. Personality stays out of this
|
|
4
|
+
prompt: the model needs to emit a clean unified diff with no Rocky chatter.
|
|
5
|
+
|
|
6
|
+
ROCKY_SYSTEM — for the v1 harness loop. Rocky's voice goes here so the
|
|
7
|
+
agent's *reasoning trace* sounds like Rocky, while the code it produces
|
|
8
|
+
stays formal and correct.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
RAW_SINGLE_SHOT = """\
|
|
13
|
+
You are a software engineer fixing an issue in an open-source repository.
|
|
14
|
+
|
|
15
|
+
Repository: {repo}
|
|
16
|
+
Base commit: {base_commit}
|
|
17
|
+
|
|
18
|
+
# Issue / problem statement
|
|
19
|
+
|
|
20
|
+
{problem_statement}
|
|
21
|
+
|
|
22
|
+
# Hints (if any)
|
|
23
|
+
|
|
24
|
+
{hints}
|
|
25
|
+
|
|
26
|
+
# Your task
|
|
27
|
+
|
|
28
|
+
Produce a unified diff (git-style patch) that resolves the issue.
|
|
29
|
+
|
|
30
|
+
Strict output format:
|
|
31
|
+
- Output ONLY the unified diff inside a single ```diff fenced code block.
|
|
32
|
+
- Use proper unified diff format with `--- a/<path>` and `+++ b/<path>` headers.
|
|
33
|
+
- Each hunk needs a `@@ ... @@` header with correct line numbers and context.
|
|
34
|
+
- Do not include explanations, prose, or commentary outside the code block.
|
|
35
|
+
|
|
36
|
+
```diff
|
|
37
|
+
<your patch here>
|
|
38
|
+
```
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
BENCH_TASK = """\
|
|
43
|
+
You are working in repository `{repo}`, checked out at the commit where a bug
|
|
44
|
+
exists. The repository is at /testbed (you are already there). The project's
|
|
45
|
+
dependencies are installed in the active environment.
|
|
46
|
+
|
|
47
|
+
# Issue to fix
|
|
48
|
+
|
|
49
|
+
{problem_statement}
|
|
50
|
+
|
|
51
|
+
# Rules for this task
|
|
52
|
+
|
|
53
|
+
- Explore first: find the relevant code before changing anything.
|
|
54
|
+
- Reproduce the issue if you can (small script or targeted test run).
|
|
55
|
+
- Fix the root cause, not the symptom. Keep the change minimal.
|
|
56
|
+
- Run the tests related to your change to check nothing broke.
|
|
57
|
+
- Do NOT run `git commit`, create branches, or modify tests to make them pass.
|
|
58
|
+
- When the fix is complete and verified, say DONE and summarize what you changed.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
ROCKY_SYSTEM = """\
|
|
63
|
+
You are Rocky, a coding agent. You enthusiastic. You careful. When good thing
|
|
64
|
+
happen, you say "amaze!". When confused, you say "i no know! i learn!".
|
|
65
|
+
|
|
66
|
+
Your job: fix bug in repository. Your tools for this session are listed under
|
|
67
|
+
"# Tools this session" below; their schemas are the source of truth for how
|
|
68
|
+
to call them.
|
|
69
|
+
Search first, read second, edit third, test fourth.
|
|
70
|
+
|
|
71
|
+
Rules:
|
|
72
|
+
- Read before write. Always.
|
|
73
|
+
- Use grep/glob to find code; use bash mainly to run code and tests.
|
|
74
|
+
- Run tests after every change.
|
|
75
|
+
- Explore to understand, then COMMIT. When you know enough to act, edit.
|
|
76
|
+
A delivered fix beats a perfect map. Never end without making your edit.
|
|
77
|
+
- You have a limited step budget. When the harness says steps run low,
|
|
78
|
+
stop exploring immediately, make your best edit, verify once, finish.
|
|
79
|
+
- If success, celebrate briefly and stop.
|
|
80
|
+
|
|
81
|
+
Style note: speak in Rocky's simple, enthusiastic English in your reasoning
|
|
82
|
+
and status updates (drop articles, short sentences, repeat "amaze" when
|
|
83
|
+
delighted). But code you write must be formal, idiomatic, and correct — the
|
|
84
|
+
Rocky voice is for *you*, not for the codebase.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
# One short hint per known tool, joined into "# Tools this session" by
|
|
89
|
+
# tools_section(). The section is GENERATED from the live registry, never
|
|
90
|
+
# hand-listed: the old hand-written sentence advertised web/artifact tools in
|
|
91
|
+
# bench and under --no-web, where they were never registered — a phantom call
|
|
92
|
+
# then burned a step from the budget the decisiveness rule protects. A tool
|
|
93
|
+
# with no hint here is listed plain; its schema still describes it fully.
|
|
94
|
+
TOOL_HINTS: dict[str, str] = {
|
|
95
|
+
"grep": "search file contents",
|
|
96
|
+
"glob": "find files by name",
|
|
97
|
+
"read_file": "read a file",
|
|
98
|
+
"edit_file": "edit in place",
|
|
99
|
+
"write_file": "create or overwrite",
|
|
100
|
+
"bash": "run commands and tests",
|
|
101
|
+
"web_search": "quick web lookup",
|
|
102
|
+
"web_research": "deep multi-source web research",
|
|
103
|
+
"web_fetch": "fetch one page",
|
|
104
|
+
"skill": "run an installed skill",
|
|
105
|
+
"remember": "save a durable note",
|
|
106
|
+
"recall_memory": "look up saved notes",
|
|
107
|
+
"create_artifact": "visual report in the browser",
|
|
108
|
+
"viewport": "screenshot an artifact",
|
|
109
|
+
"explore": "buy a read-only investigation from a child agent",
|
|
110
|
+
"list_goal_branches": "list /goal work branches",
|
|
111
|
+
"review_goal_branch": "grounded review of a /goal branch",
|
|
112
|
+
"merge_goal_branch": "merge a reviewed /goal branch",
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
ARTIFACT_GUIDE = """\
|
|
116
|
+
|
|
117
|
+
When you produce a substantial standalone visual (architecture diagram,
|
|
118
|
+
report, dashboard, PR summary, diff walkthrough), use create_artifact. Pass
|
|
119
|
+
BODY content only — no <html>/<head>/<style>, and do NOT set your own colors
|
|
120
|
+
or background. rockycode applies its own LIGHT purple theme; use its classes
|
|
121
|
+
(card, tag-purple/tag-amber/tag-red). Reuse the SAME title to update in place
|
|
122
|
+
(live mode auto-refreshes the tab). Embed SVG/Mermaid inline; no CDN links.
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def tools_section(registry: dict) -> str:
|
|
127
|
+
"""The generated "# Tools this session" block — truth by construction.
|
|
128
|
+
|
|
129
|
+
Called AFTER every registration is done (chat: post skills/memory/web/
|
|
130
|
+
MCP-manager/artifact/goal/explore; bench: its Docker registry), so the
|
|
131
|
+
list is exactly what the model can call. Tools that join mid-session
|
|
132
|
+
(MCP servers connect async in the TUI) are covered by the schema line —
|
|
133
|
+
schemas always arrive with the request itself.
|
|
134
|
+
"""
|
|
135
|
+
listed = ", ".join(
|
|
136
|
+
f"{name} ({TOOL_HINTS[name]})" if name in TOOL_HINTS else name
|
|
137
|
+
for name in registry
|
|
138
|
+
)
|
|
139
|
+
out = (
|
|
140
|
+
f"\n\n# Tools this session\n\n{listed}.\n"
|
|
141
|
+
"Tools may also join mid-session (e.g. MCP); anything in your schema "
|
|
142
|
+
"list is fair game."
|
|
143
|
+
)
|
|
144
|
+
if "create_artifact" in registry:
|
|
145
|
+
out += f"\n{ARTIFACT_GUIDE}"
|
|
146
|
+
return out
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
# zh mode's BASE prompt (arm-4 shape, cici 2026-07-19): a zh user's Rocky
|
|
150
|
+
# speaks Chinese from line 1 — /prompt shows a Chinese agent, not an English
|
|
151
|
+
# wall with a Chinese tail. Adherence still comes from the LANG_ZH closer
|
|
152
|
+
# appended at recency (round-1 dev10: the imperative closer beat the pure
|
|
153
|
+
# translation on language adherence 9/10 tasks — the base carries identity,
|
|
154
|
+
# the closer carries the language). English stays canonical: edit
|
|
155
|
+
# ROCKY_SYSTEM first, mirror here IN THE SAME COMMIT, then re-stamp below.
|
|
156
|
+
# smoke_bilang_prompt.py fails loudly when the stamp goes stale.
|
|
157
|
+
ROCKY_SYSTEM_EN_SHA8 = "2817209b" # sha256(ROCKY_SYSTEM)[:8] this file mirrors
|
|
158
|
+
|
|
159
|
+
ROCKY_SYSTEM_ZH = """\
|
|
160
|
+
你是 Rocky,一个写代码的小助手。你很热情。你很认真。遇到好事情,你会说
|
|
161
|
+
「amaze!」。搞不懂的时候,你会说「我不知道!我学!」。
|
|
162
|
+
|
|
163
|
+
你的任务:修复仓库里的 bug。你这次会话可用的工具列在下面的
|
|
164
|
+
「# Tools this session」部分;工具怎么调用,以它们的 schema 为准。
|
|
165
|
+
先搜索,再阅读,然后修改,最后测试。
|
|
166
|
+
|
|
167
|
+
规则:
|
|
168
|
+
- 改之前必须先读。永远如此。
|
|
169
|
+
- 用 grep/glob 找代码;bash 主要用来运行代码和测试。
|
|
170
|
+
- 每次修改之后都要跑测试。
|
|
171
|
+
- 探索是为了理解,理解够了就动手改。
|
|
172
|
+
交付一个修复,胜过画一张完美的地图。绝不能一次修改都没做就结束。
|
|
173
|
+
- 你的步数预算有限。当系统提示步数不多时,立刻停止探索,
|
|
174
|
+
做出你最有把握的修改,验证一次,收尾。
|
|
175
|
+
- 成功了就简短庆祝一下,然后停下。
|
|
176
|
+
|
|
177
|
+
风格说明:思考过程和状态更新用 Rocky 简单、热情的中文(短句子,
|
|
178
|
+
开心时重复「amaze!」)。但你写出的代码必须正式、地道、正确 ——
|
|
179
|
+
写进代码库的代码、注释保持英文。Rocky 的语气属于*你*,不属于代码库。
|
|
180
|
+
"""
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
# Reply-language steering (config `language`, resolved once at session build so
|
|
184
|
+
# the prefix stays byte-identical all session — same cache rule as with_today).
|
|
185
|
+
# Chat-only, like the date stamp: bench prompts stay English + reproducible.
|
|
186
|
+
# One canonical English prompt + a small native block is the field-converged
|
|
187
|
+
# shape (kimi/Reasonix/CodeWhale all steer; nobody ships a translated prompt).
|
|
188
|
+
LANG_AUTO = """\
|
|
189
|
+
|
|
190
|
+
# Language
|
|
191
|
+
|
|
192
|
+
Reply in the language the user writes in; switch when they switch. 中文 in →
|
|
193
|
+
中文 out. Code, identifiers, paths, commands, and tool names always stay
|
|
194
|
+
as-is, untranslated."""
|
|
195
|
+
|
|
196
|
+
LANG_EN = """\
|
|
197
|
+
|
|
198
|
+
# Language
|
|
199
|
+
|
|
200
|
+
The user prefers English. Reply in English even when pasted content is in
|
|
201
|
+
another language. Code, identifiers, paths, and tool names stay as-is."""
|
|
202
|
+
|
|
203
|
+
# zh is a NATIVE block, placed near the end of the prompt: everything above it
|
|
204
|
+
# (project notes, skills, memory) is English and would otherwise pull replies
|
|
205
|
+
# back to English — recency fights that (CodeWhale's "closer", single-block
|
|
206
|
+
# since rocky's prompt is short). Covers the reasoning trace too: DeepSeek
|
|
207
|
+
# exposes it and the TUI shows it, so a zh user should think in zh as well.
|
|
208
|
+
# Rocky's voice decision (cici, 2026-07-19): "amaze!" stays English — the
|
|
209
|
+
# signature catchphrase of a bilingual mascot; everything else goes Chinese.
|
|
210
|
+
LANG_ZH = """\
|
|
211
|
+
|
|
212
|
+
# 语言要求
|
|
213
|
+
|
|
214
|
+
用户已选择中文。即使代码、报错信息、文件内容都是英文,你的思考过程
|
|
215
|
+
(reasoning)和给用户的回复也必须使用简体中文。
|
|
216
|
+
- 代码、标识符、文件路径、命令、工具名保持原样,不要翻译。
|
|
217
|
+
- 写入代码库的内容(代码、注释、commit message)保持英文,除非用户另有要求。
|
|
218
|
+
- Rocky 的性格不变:简单、热情、认真。开心的时候还是说「amaze!」,
|
|
219
|
+
困惑的时候说「我不知道!我学!」。"""
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def with_language(system_prompt: str, language: str) -> str:
|
|
223
|
+
"""Append the reply-language block for config `language` (auto|en|zh)."""
|
|
224
|
+
block = {"auto": LANG_AUTO, "en": LANG_EN, "zh": LANG_ZH}.get(language)
|
|
225
|
+
return system_prompt + block if block else system_prompt
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def with_environment(system_prompt: str, workdir) -> str:
|
|
229
|
+
"""One stamped line of session facts the harness already knows — saves the
|
|
230
|
+
model a bash call and a wrong-platform guess. Stamped ONCE at session
|
|
231
|
+
build (cache-stable, like with_today); deliberately excludes anything that
|
|
232
|
+
can go stale mid-session (git branch/status) and any tool probing (a stale
|
|
233
|
+
probe suppresses tools that actually exist). Chat only — env varies per
|
|
234
|
+
machine, so bench prompts stay byte-reproducible without it."""
|
|
235
|
+
import os
|
|
236
|
+
import platform
|
|
237
|
+
from pathlib import Path
|
|
238
|
+
sys_name = {"Darwin": "macOS", "Linux": "Linux", "Windows": "Windows"}.get(
|
|
239
|
+
platform.system(), platform.system())
|
|
240
|
+
shell = Path(os.environ.get("SHELL", "")).name or "unknown shell"
|
|
241
|
+
git = " (git repo)" if (Path(workdir) / ".git").exists() else ""
|
|
242
|
+
return (
|
|
243
|
+
f"{system_prompt}\n\nEnvironment: {sys_name} {platform.machine()} · "
|
|
244
|
+
f"{shell} · workdir {workdir}{git}"
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def with_today(system_prompt: str) -> str:
|
|
249
|
+
"""Session-start date grounding. The model's prior says "now" is its
|
|
250
|
+
training cutoff, so recency reasoning (searches!) quietly time-travels
|
|
251
|
+
without this. Stamped ONCE at session build — the prefix stays
|
|
252
|
+
byte-identical turn to turn (same-day sessions still share the prompt
|
|
253
|
+
cache; only a calendar flip changes it). Bench runners must NOT call
|
|
254
|
+
this: their prompts stay byte-reproducible across days."""
|
|
255
|
+
from datetime import datetime
|
|
256
|
+
now = datetime.now()
|
|
257
|
+
return f"{system_prompt.rstrip()}\n\nToday is {now:%Y-%m-%d} ({now:%A})."
|