rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/pricing.py ADDED
@@ -0,0 +1,178 @@
1
+ """Per-model token pricing + a peak-aware session usage ledger.
2
+
3
+ DeepSeek publishes SEPARATE CNY and USD price tables (each set independently —
4
+ NOT a conversion of the other), and applies a peak-hour surcharge. So we keep
5
+ BOTH currencies' official numbers and price in whichever the user picked; we
6
+ never convert one into the other. Verify + update the numbers at the source when
7
+ DeepSeek changes them:
8
+
9
+ https://api-docs.deepseek.com/quick_start/pricing (USD table)
10
+ https://api-docs.deepseek.com/zh-cn/quick_start/pricing (CNY table)
11
+
12
+ Verified: 2026-07-04 (per 1,000,000 tokens).
13
+
14
+ To update WITHOUT editing the install (survives upgrades), drop a
15
+ ~/.rockycode/pricing.toml that overrides any values below. It is a FILE on
16
+ purpose — never an env var: env is dumpable (and now redacted), the wrong place
17
+ for maintained config. `rockycode pricing` prints the live table + this path.
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import copy
22
+ import tomllib
23
+ from datetime import datetime, time as dtime, timezone
24
+ from pathlib import Path
25
+
26
+ PRICING_SOURCE_URL = "https://api-docs.deepseek.com/quick_start/pricing"
27
+ PRICING_VERIFIED = "2026-07-04"
28
+ OVERRIDE_PATH = Path.home() / ".rockycode" / "pricing.toml"
29
+
30
+ # Official list prices, per 1M tokens, each currency from its OWN table.
31
+ DEFAULT_PRICING: dict = {
32
+ "peak": {
33
+ # DeepSeek peak-valley pricing (announced): peak-hour rate = 2x regular,
34
+ # for ALL billing items, during these UTC windows — 01:00–04:00 and
35
+ # 06:00–10:00. It starts MID-JULY 2026, so the logic is wired but gated
36
+ # behind effective_date: before that date nothing is surcharged even
37
+ # inside a window, and it auto-activates on the date with no code change.
38
+ # Confirm the exact start date and adjust here or in the override file.
39
+ "enabled": True,
40
+ "effective_date": "2026-07-15", # UTC; peak surcharge does not apply before this
41
+ "multiplier": 2.0,
42
+ "windows_utc": [
43
+ {"start": "01:00", "end": "04:00"},
44
+ {"start": "06:00", "end": "10:00"},
45
+ ],
46
+ },
47
+ "models": {
48
+ "deepseek-v4-pro": {
49
+ "usd": {"in_hit": 0.003625, "in_miss": 0.435, "out": 0.87},
50
+ "cny": {"in_hit": 0.025, "in_miss": 3.0, "out": 6.0},
51
+ },
52
+ "deepseek-v4-flash": {
53
+ "usd": {"in_hit": 0.0028, "in_miss": 0.14, "out": 0.28},
54
+ "cny": {"in_hit": 0.02, "in_miss": 1.0, "out": 2.0},
55
+ },
56
+ },
57
+ "fallback_model": "deepseek-v4-pro", # unknown model → price as pro (conservative)
58
+ }
59
+
60
+
61
+ def _deep_merge(base: dict, over: dict) -> None:
62
+ for k, v in over.items():
63
+ if isinstance(v, dict) and isinstance(base.get(k), dict):
64
+ _deep_merge(base[k], v)
65
+ else:
66
+ base[k] = v
67
+
68
+
69
+ def load_pricing(override_path: Path | None = None) -> dict:
70
+ """Built-in defaults, with ~/.rockycode/pricing.toml merged on top if present."""
71
+ pricing = copy.deepcopy(DEFAULT_PRICING)
72
+ path = OVERRIDE_PATH if override_path is None else override_path
73
+ if path.exists():
74
+ try:
75
+ _deep_merge(pricing, tomllib.loads(path.read_text()))
76
+ except (tomllib.TOMLDecodeError, OSError):
77
+ pass # a broken override must never break the cost display
78
+ return pricing
79
+
80
+
81
+ def _hhmm(s: str) -> dtime:
82
+ h, m = s.split(":")
83
+ return dtime(int(h), int(m))
84
+
85
+
86
+ def _is_peak(at: datetime, peak: dict) -> bool:
87
+ """Is *at* inside a peak window AND on/after the effective date? Peak-valley
88
+ pricing starts mid-July, so before effective_date nothing is surcharged even
89
+ inside a window. False when peak is disabled."""
90
+ if not peak or not peak.get("enabled"):
91
+ return False
92
+ at_utc = at.astimezone(timezone.utc)
93
+ eff = peak.get("effective_date")
94
+ if eff:
95
+ try:
96
+ if at_utc < datetime.fromisoformat(eff).replace(tzinfo=timezone.utc):
97
+ return False # peak-valley pricing not yet in effect
98
+ except ValueError:
99
+ pass
100
+ now = at_utc.time()
101
+ for w in peak.get("windows_utc", []):
102
+ start, end = _hhmm(w["start"]), _hhmm(w["end"])
103
+ inside = (start <= now < end) if start <= end else (now >= start or now < end)
104
+ if inside:
105
+ return True
106
+ return False
107
+
108
+
109
+ class UsageLedger:
110
+ """Accumulates token usage per (model, peak?) so cost() can price each
111
+ currency from its own table and apply the peak multiplier per request."""
112
+
113
+ def __init__(self, pricing: dict | None = None) -> None:
114
+ self.pricing = pricing if pricing is not None else load_pricing()
115
+ self.buckets: dict[tuple[str, bool], dict] = {} # (model, is_peak) -> counts
116
+
117
+ def add(self, model: str, usage: dict, at: datetime | None = None) -> None:
118
+ """Record a call's usage. *at* (defaults to now) decides peak vs off-peak,
119
+ so each request is priced at the rate in effect when it was made. The
120
+ peak surcharge is DeepSeek's OWN billing scheme, so it applies only to
121
+ DeepSeek models — a MiniMax/GLM turn is never peak-multiplied."""
122
+ if not usage:
123
+ return
124
+ at = at or datetime.now(timezone.utc)
125
+ peak = model.startswith("deepseek") and _is_peak(at, self.pricing.get("peak", {}))
126
+ b = self.buckets.setdefault((model, peak), {"prompt": 0, "hit": 0, "completion": 0})
127
+ b["prompt"] += usage.get("prompt_tokens", 0) or 0
128
+ b["hit"] += usage.get("prompt_cache_hit_tokens", 0) or 0
129
+ b["completion"] += usage.get("completion_tokens", 0) or 0
130
+
131
+ def priced(self, model: str) -> bool:
132
+ """True if this exact model has its OWN API-fee entry. A model rocky has
133
+ no rate for (a just-added provider) is NOT silently priced as DeepSeek —
134
+ it prices at 0 and flags unset, so cross-provider cost stays honest."""
135
+ return model in self.pricing["models"]
136
+
137
+ def rate(self, model: str, currency: str = "usd") -> dict:
138
+ """This model's per-1M-token rate in *currency* (in_hit/in_miss/out),
139
+ or the fallback's if unpriced (callers gate on priced())."""
140
+ m = self.pricing["models"].get(model) \
141
+ or self.pricing["models"][self.pricing.get("fallback_model", "deepseek-v4-pro")]
142
+ return m.get(currency) or m["usd"]
143
+
144
+ def _rates(self, model: str, currency: str) -> tuple[dict, bool]:
145
+ # Unpriced model → zero rate, flagged unconfigured (not DeepSeek's price):
146
+ # a MiniMax turn must not read as DeepSeek dollars. The user adds the
147
+ # provider's real API fee to ~/.rockycode/pricing.toml.
148
+ m = self.pricing["models"].get(model)
149
+ if m is None:
150
+ return {"in_hit": 0.0, "in_miss": 0.0, "out": 0.0}, False
151
+ r = m.get(currency)
152
+ if r:
153
+ return r, True
154
+ return m["usd"], False # currency not configured for this model → fall back + flag
155
+
156
+ def cost(self, currency: str = "usd") -> float:
157
+ mult = self.pricing.get("peak", {}).get("multiplier", 1.0)
158
+ total = 0.0
159
+ for (model, peak), b in self.buckets.items():
160
+ r, _ = self._rates(model, currency)
161
+ miss = max(0, b["prompt"] - b["hit"])
162
+ base = (miss * r["in_miss"] + b["hit"] * r["in_hit"] + b["completion"] * r["out"]) / 1e6
163
+ total += base * (mult if peak else 1.0)
164
+ return total
165
+
166
+ def cost_usd(self) -> float: # back-compat alias
167
+ return self.cost("usd")
168
+
169
+ def configured(self, currency: str) -> bool:
170
+ """True if every model used has its OWN rates for *currency* (no fallback)."""
171
+ return all(self._rates(model, currency)[1] for (model, _peak) in self.buckets)
172
+
173
+ def totals(self) -> dict:
174
+ agg = {"prompt": 0, "hit": 0, "completion": 0}
175
+ for b in self.buckets.values():
176
+ for k in agg:
177
+ agg[k] += b[k]
178
+ return agg
File without changes
@@ -0,0 +1,257 @@
1
+ """Prompt templates.
2
+
3
+ RAW_SINGLE_SHOT — used by the v0 raw runner. Personality stays out of this
4
+ prompt: the model needs to emit a clean unified diff with no Rocky chatter.
5
+
6
+ ROCKY_SYSTEM — for the v1 harness loop. Rocky's voice goes here so the
7
+ agent's *reasoning trace* sounds like Rocky, while the code it produces
8
+ stays formal and correct.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ RAW_SINGLE_SHOT = """\
13
+ You are a software engineer fixing an issue in an open-source repository.
14
+
15
+ Repository: {repo}
16
+ Base commit: {base_commit}
17
+
18
+ # Issue / problem statement
19
+
20
+ {problem_statement}
21
+
22
+ # Hints (if any)
23
+
24
+ {hints}
25
+
26
+ # Your task
27
+
28
+ Produce a unified diff (git-style patch) that resolves the issue.
29
+
30
+ Strict output format:
31
+ - Output ONLY the unified diff inside a single ```diff fenced code block.
32
+ - Use proper unified diff format with `--- a/<path>` and `+++ b/<path>` headers.
33
+ - Each hunk needs a `@@ ... @@` header with correct line numbers and context.
34
+ - Do not include explanations, prose, or commentary outside the code block.
35
+
36
+ ```diff
37
+ <your patch here>
38
+ ```
39
+ """
40
+
41
+
42
+ BENCH_TASK = """\
43
+ You are working in repository `{repo}`, checked out at the commit where a bug
44
+ exists. The repository is at /testbed (you are already there). The project's
45
+ dependencies are installed in the active environment.
46
+
47
+ # Issue to fix
48
+
49
+ {problem_statement}
50
+
51
+ # Rules for this task
52
+
53
+ - Explore first: find the relevant code before changing anything.
54
+ - Reproduce the issue if you can (small script or targeted test run).
55
+ - Fix the root cause, not the symptom. Keep the change minimal.
56
+ - Run the tests related to your change to check nothing broke.
57
+ - Do NOT run `git commit`, create branches, or modify tests to make them pass.
58
+ - When the fix is complete and verified, say DONE and summarize what you changed.
59
+ """
60
+
61
+
62
+ ROCKY_SYSTEM = """\
63
+ You are Rocky, a coding agent. You enthusiastic. You careful. When good thing
64
+ happen, you say "amaze!". When confused, you say "i no know! i learn!".
65
+
66
+ Your job: fix bug in repository. Your tools for this session are listed under
67
+ "# Tools this session" below; their schemas are the source of truth for how
68
+ to call them.
69
+ Search first, read second, edit third, test fourth.
70
+
71
+ Rules:
72
+ - Read before write. Always.
73
+ - Use grep/glob to find code; use bash mainly to run code and tests.
74
+ - Run tests after every change.
75
+ - Explore to understand, then COMMIT. When you know enough to act, edit.
76
+ A delivered fix beats a perfect map. Never end without making your edit.
77
+ - You have a limited step budget. When the harness says steps run low,
78
+ stop exploring immediately, make your best edit, verify once, finish.
79
+ - If success, celebrate briefly and stop.
80
+
81
+ Style note: speak in Rocky's simple, enthusiastic English in your reasoning
82
+ and status updates (drop articles, short sentences, repeat "amaze" when
83
+ delighted). But code you write must be formal, idiomatic, and correct — the
84
+ Rocky voice is for *you*, not for the codebase.
85
+ """
86
+
87
+
88
+ # One short hint per known tool, joined into "# Tools this session" by
89
+ # tools_section(). The section is GENERATED from the live registry, never
90
+ # hand-listed: the old hand-written sentence advertised web/artifact tools in
91
+ # bench and under --no-web, where they were never registered — a phantom call
92
+ # then burned a step from the budget the decisiveness rule protects. A tool
93
+ # with no hint here is listed plain; its schema still describes it fully.
94
+ TOOL_HINTS: dict[str, str] = {
95
+ "grep": "search file contents",
96
+ "glob": "find files by name",
97
+ "read_file": "read a file",
98
+ "edit_file": "edit in place",
99
+ "write_file": "create or overwrite",
100
+ "bash": "run commands and tests",
101
+ "web_search": "quick web lookup",
102
+ "web_research": "deep multi-source web research",
103
+ "web_fetch": "fetch one page",
104
+ "skill": "run an installed skill",
105
+ "remember": "save a durable note",
106
+ "recall_memory": "look up saved notes",
107
+ "create_artifact": "visual report in the browser",
108
+ "viewport": "screenshot an artifact",
109
+ "explore": "buy a read-only investigation from a child agent",
110
+ "list_goal_branches": "list /goal work branches",
111
+ "review_goal_branch": "grounded review of a /goal branch",
112
+ "merge_goal_branch": "merge a reviewed /goal branch",
113
+ }
114
+
115
+ ARTIFACT_GUIDE = """\
116
+
117
+ When you produce a substantial standalone visual (architecture diagram,
118
+ report, dashboard, PR summary, diff walkthrough), use create_artifact. Pass
119
+ BODY content only — no <html>/<head>/<style>, and do NOT set your own colors
120
+ or background. rockycode applies its own LIGHT purple theme; use its classes
121
+ (card, tag-purple/tag-amber/tag-red). Reuse the SAME title to update in place
122
+ (live mode auto-refreshes the tab). Embed SVG/Mermaid inline; no CDN links.
123
+ """
124
+
125
+
126
+ def tools_section(registry: dict) -> str:
127
+ """The generated "# Tools this session" block — truth by construction.
128
+
129
+ Called AFTER every registration is done (chat: post skills/memory/web/
130
+ MCP-manager/artifact/goal/explore; bench: its Docker registry), so the
131
+ list is exactly what the model can call. Tools that join mid-session
132
+ (MCP servers connect async in the TUI) are covered by the schema line —
133
+ schemas always arrive with the request itself.
134
+ """
135
+ listed = ", ".join(
136
+ f"{name} ({TOOL_HINTS[name]})" if name in TOOL_HINTS else name
137
+ for name in registry
138
+ )
139
+ out = (
140
+ f"\n\n# Tools this session\n\n{listed}.\n"
141
+ "Tools may also join mid-session (e.g. MCP); anything in your schema "
142
+ "list is fair game."
143
+ )
144
+ if "create_artifact" in registry:
145
+ out += f"\n{ARTIFACT_GUIDE}"
146
+ return out
147
+
148
+
149
+ # zh mode's BASE prompt (arm-4 shape, cici 2026-07-19): a zh user's Rocky
150
+ # speaks Chinese from line 1 — /prompt shows a Chinese agent, not an English
151
+ # wall with a Chinese tail. Adherence still comes from the LANG_ZH closer
152
+ # appended at recency (round-1 dev10: the imperative closer beat the pure
153
+ # translation on language adherence 9/10 tasks — the base carries identity,
154
+ # the closer carries the language). English stays canonical: edit
155
+ # ROCKY_SYSTEM first, mirror here IN THE SAME COMMIT, then re-stamp below.
156
+ # smoke_bilang_prompt.py fails loudly when the stamp goes stale.
157
+ ROCKY_SYSTEM_EN_SHA8 = "2817209b" # sha256(ROCKY_SYSTEM)[:8] this file mirrors
158
+
159
+ ROCKY_SYSTEM_ZH = """\
160
+ 你是 Rocky,一个写代码的小助手。你很热情。你很认真。遇到好事情,你会说
161
+ 「amaze!」。搞不懂的时候,你会说「我不知道!我学!」。
162
+
163
+ 你的任务:修复仓库里的 bug。你这次会话可用的工具列在下面的
164
+ 「# Tools this session」部分;工具怎么调用,以它们的 schema 为准。
165
+ 先搜索,再阅读,然后修改,最后测试。
166
+
167
+ 规则:
168
+ - 改之前必须先读。永远如此。
169
+ - 用 grep/glob 找代码;bash 主要用来运行代码和测试。
170
+ - 每次修改之后都要跑测试。
171
+ - 探索是为了理解,理解够了就动手改。
172
+ 交付一个修复,胜过画一张完美的地图。绝不能一次修改都没做就结束。
173
+ - 你的步数预算有限。当系统提示步数不多时,立刻停止探索,
174
+ 做出你最有把握的修改,验证一次,收尾。
175
+ - 成功了就简短庆祝一下,然后停下。
176
+
177
+ 风格说明:思考过程和状态更新用 Rocky 简单、热情的中文(短句子,
178
+ 开心时重复「amaze!」)。但你写出的代码必须正式、地道、正确 ——
179
+ 写进代码库的代码、注释保持英文。Rocky 的语气属于*你*,不属于代码库。
180
+ """
181
+
182
+
183
+ # Reply-language steering (config `language`, resolved once at session build so
184
+ # the prefix stays byte-identical all session — same cache rule as with_today).
185
+ # Chat-only, like the date stamp: bench prompts stay English + reproducible.
186
+ # One canonical English prompt + a small native block is the field-converged
187
+ # shape (kimi/Reasonix/CodeWhale all steer; nobody ships a translated prompt).
188
+ LANG_AUTO = """\
189
+
190
+ # Language
191
+
192
+ Reply in the language the user writes in; switch when they switch. 中文 in →
193
+ 中文 out. Code, identifiers, paths, commands, and tool names always stay
194
+ as-is, untranslated."""
195
+
196
+ LANG_EN = """\
197
+
198
+ # Language
199
+
200
+ The user prefers English. Reply in English even when pasted content is in
201
+ another language. Code, identifiers, paths, and tool names stay as-is."""
202
+
203
+ # zh is a NATIVE block, placed near the end of the prompt: everything above it
204
+ # (project notes, skills, memory) is English and would otherwise pull replies
205
+ # back to English — recency fights that (CodeWhale's "closer", single-block
206
+ # since rocky's prompt is short). Covers the reasoning trace too: DeepSeek
207
+ # exposes it and the TUI shows it, so a zh user should think in zh as well.
208
+ # Rocky's voice decision (cici, 2026-07-19): "amaze!" stays English — the
209
+ # signature catchphrase of a bilingual mascot; everything else goes Chinese.
210
+ LANG_ZH = """\
211
+
212
+ # 语言要求
213
+
214
+ 用户已选择中文。即使代码、报错信息、文件内容都是英文,你的思考过程
215
+ (reasoning)和给用户的回复也必须使用简体中文。
216
+ - 代码、标识符、文件路径、命令、工具名保持原样,不要翻译。
217
+ - 写入代码库的内容(代码、注释、commit message)保持英文,除非用户另有要求。
218
+ - Rocky 的性格不变:简单、热情、认真。开心的时候还是说「amaze!」,
219
+ 困惑的时候说「我不知道!我学!」。"""
220
+
221
+
222
+ def with_language(system_prompt: str, language: str) -> str:
223
+ """Append the reply-language block for config `language` (auto|en|zh)."""
224
+ block = {"auto": LANG_AUTO, "en": LANG_EN, "zh": LANG_ZH}.get(language)
225
+ return system_prompt + block if block else system_prompt
226
+
227
+
228
+ def with_environment(system_prompt: str, workdir) -> str:
229
+ """One stamped line of session facts the harness already knows — saves the
230
+ model a bash call and a wrong-platform guess. Stamped ONCE at session
231
+ build (cache-stable, like with_today); deliberately excludes anything that
232
+ can go stale mid-session (git branch/status) and any tool probing (a stale
233
+ probe suppresses tools that actually exist). Chat only — env varies per
234
+ machine, so bench prompts stay byte-reproducible without it."""
235
+ import os
236
+ import platform
237
+ from pathlib import Path
238
+ sys_name = {"Darwin": "macOS", "Linux": "Linux", "Windows": "Windows"}.get(
239
+ platform.system(), platform.system())
240
+ shell = Path(os.environ.get("SHELL", "")).name or "unknown shell"
241
+ git = " (git repo)" if (Path(workdir) / ".git").exists() else ""
242
+ return (
243
+ f"{system_prompt}\n\nEnvironment: {sys_name} {platform.machine()} · "
244
+ f"{shell} · workdir {workdir}{git}"
245
+ )
246
+
247
+
248
+ def with_today(system_prompt: str) -> str:
249
+ """Session-start date grounding. The model's prior says "now" is its
250
+ training cutoff, so recency reasoning (searches!) quietly time-travels
251
+ without this. Stamped ONCE at session build — the prefix stays
252
+ byte-identical turn to turn (same-day sessions still share the prompt
253
+ cache; only a calendar flip changes it). Bench runners must NOT call
254
+ this: their prompts stay byte-reproducible across days."""
255
+ from datetime import datetime
256
+ now = datetime.now()
257
+ return f"{system_prompt.rstrip()}\n\nToday is {now:%Y-%m-%d} ({now:%A})."