rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,479 @@
1
+ """v1 tool set: bash, read_file, write_file, edit_file, grep, glob.
2
+
3
+ Small and orthogonal on purpose. Each tool is an async function plus an
4
+ OpenAI-shape schema; the registry is what the loop hands to the model.
5
+ Output is truncated before it reaches the model so one noisy command can't
6
+ blow the context.
7
+
8
+ grep/glob exist even though bash could do both: dedicated tools with tight
9
+ schemas are easier for smaller models to call correctly, and their outputs
10
+ are capped + cleaned (no junk dirs, no binary files).
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import asyncio
15
+ import fnmatch
16
+ import json
17
+ import os
18
+ import re
19
+ import signal
20
+ from dataclasses import dataclass
21
+ from pathlib import Path
22
+ from typing import Awaitable, Callable, Optional
23
+
24
+ from rockycode.engine.redact import redact
25
+
26
+ MAX_OUTPUT_CHARS = 30_000
27
+ # Seconds before a bash command is killed. Raise it for long jobs (e.g. many
28
+ # file downloads) via ROCKYCODE_BASH_TIMEOUT.
29
+ BASH_TIMEOUT_S = int(os.environ.get("ROCKYCODE_BASH_TIMEOUT", "300"))
30
+ GREP_MAX_MATCHES = 200
31
+ GLOB_MAX_PATHS = 500
32
+
33
+ SKIP_DIRS = {".git", "__pycache__", ".venv", "venv", "node_modules", ".rockycode", ".cache"}
34
+
35
+
36
+ def _truncate(text: str, limit: int = MAX_OUTPUT_CHARS) -> str:
37
+ if len(text) <= limit:
38
+ return text
39
+ head = text[: limit // 2]
40
+ tail = text[-limit // 2 :]
41
+ dropped = len(text) - limit
42
+ return f"{head}\n... [{dropped} chars truncated] ...\n{tail}"
43
+
44
+
45
+ @dataclass
46
+ class Tool:
47
+ name: str
48
+ schema: dict
49
+ fn: Callable[..., Awaitable[str]]
50
+ # Approval tier for the permission layer. "risky" is the fail-safe default so a
51
+ # tool that forgets to classify itself (incl. dynamic MCP tools) gets gated
52
+ # rather than silently allowed. See engine/permission.py for tier -> policy.
53
+ risk: str = "risky"
54
+
55
+
56
+ def _fn_schema(name: str, description: str, params: dict, required: list[str]) -> dict:
57
+ return {
58
+ "type": "function",
59
+ "function": {
60
+ "name": name,
61
+ "description": description,
62
+ "parameters": {"type": "object", "properties": params, "required": required},
63
+ },
64
+ }
65
+
66
+
67
+ # Shared by the local registry (chat) and the container registry (bench) —
68
+ # the model sees identical tools either way, only the execution target differs.
69
+ SCHEMAS: dict[str, dict] = {
70
+ "bash": _fn_schema(
71
+ "bash",
72
+ "Run a shell command in the repository working directory. "
73
+ "Stdout+stderr combined, exit code on the first line.",
74
+ {"command": {"type": "string", "description": "The shell command to run."}},
75
+ ["command"],
76
+ ),
77
+ # offset/limit exist because the model keeps asking for them: 128 chat +
78
+ # ~60 bench "[error] … unexpected keyword argument 'offset'" trajectories
79
+ # show DeepSeek's prior wants windowed reads. Line numbers stay ABSOLUTE
80
+ # (numbered before slicing) so follow-up edits reference real positions.
81
+ "read_file": _fn_schema(
82
+ "read_file",
83
+ "Read a file (line-numbered) or list a directory.",
84
+ {
85
+ "path": {"type": "string", "description": "File or directory path."},
86
+ "offset": {"type": "integer", "description": "1-based line number to start from (optional)."},
87
+ "limit": {"type": "integer", "description": "Max lines to return from offset (optional)."},
88
+ },
89
+ ["path"],
90
+ ),
91
+ "write_file": _fn_schema(
92
+ "write_file",
93
+ "Create or overwrite a file with the given content.",
94
+ {"path": {"type": "string"}, "content": {"type": "string"}},
95
+ ["path", "content"],
96
+ ),
97
+ "edit_file": _fn_schema(
98
+ "edit_file",
99
+ "Replace one exact occurrence of old_string with new_string in a file. "
100
+ "old_string must be unique in the file — include surrounding context.",
101
+ {
102
+ "path": {"type": "string"},
103
+ "old_string": {"type": "string"},
104
+ "new_string": {"type": "string"},
105
+ },
106
+ ["path", "old_string", "new_string"],
107
+ ),
108
+ "grep": _fn_schema(
109
+ "grep",
110
+ "Search file contents with an extended regular expression. Returns up to "
111
+ f"{GREP_MAX_MATCHES} matches as path:line: text. Prefer this over `grep` in bash — "
112
+ "it skips binaries and junk directories.",
113
+ {
114
+ "pattern": {"type": "string", "description": "Extended regex to search for."},
115
+ "path": {"type": "string", "description": "Directory to search. Default: working dir."},
116
+ "include": {
117
+ "type": "string",
118
+ "description": "Only search files whose name matches this glob, e.g. '*.py'.",
119
+ },
120
+ },
121
+ ["pattern"],
122
+ ),
123
+ "glob": _fn_schema(
124
+ "glob",
125
+ "List files matching a path pattern, relative to the working directory. "
126
+ "Supports ** for recursion, e.g. '**/*.py' or 'lib/**/test_*.py'. "
127
+ f"Returns up to {GLOB_MAX_PATHS} sorted paths.",
128
+ {
129
+ "pattern": {"type": "string", "description": "Path glob, ** allowed."},
130
+ },
131
+ ["pattern"],
132
+ ),
133
+ }
134
+
135
+
136
+ async def _bash(command: str, *, workdir: Path) -> str:
137
+ # start_new_session=True gives the shell its own process group, so on a
138
+ # timeout we can kill the whole tree (curl/wget/… children) — not just the
139
+ # shell, which proc.kill() alone would leave running as orphans.
140
+ proc = await asyncio.create_subprocess_shell(
141
+ command,
142
+ stdout=asyncio.subprocess.PIPE,
143
+ stderr=asyncio.subprocess.STDOUT,
144
+ cwd=workdir,
145
+ start_new_session=True,
146
+ )
147
+ try:
148
+ out, _ = await asyncio.wait_for(proc.communicate(), timeout=BASH_TIMEOUT_S)
149
+ except asyncio.TimeoutError:
150
+ try:
151
+ os.killpg(proc.pid, signal.SIGKILL) # whole group (pgid == shell pid)
152
+ except ProcessLookupError:
153
+ pass
154
+ try:
155
+ await proc.wait() # reap the shell so it doesn't linger as a zombie
156
+ except Exception: # noqa: BLE001
157
+ pass
158
+ # No retry: re-running as-is would just time out again. Tell the model/
159
+ # user what happened and how to recover.
160
+ return (
161
+ f"[timeout] command exceeded {BASH_TIMEOUT_S}s and was killed "
162
+ f"(whole process group terminated). For a longer command, raise "
163
+ f"ROCKYCODE_BASH_TIMEOUT; or split the work into smaller steps."
164
+ )
165
+ except asyncio.CancelledError:
166
+ # User interrupted (Esc / new message). asyncio.wait_for cancels
167
+ # communicate() but that does NOT kill the OS process — SIGKILL the whole
168
+ # group so the shell and its curl/build/… children don't survive as
169
+ # orphans (the reason for start_new_session above). Then re-raise: the
170
+ # loop's cancellation backfill relies on CancelledError propagating.
171
+ try:
172
+ os.killpg(proc.pid, signal.SIGKILL)
173
+ except ProcessLookupError:
174
+ pass
175
+ raise
176
+ text = out.decode(errors="replace")
177
+ status = f"[exit {proc.returncode}]"
178
+ return _truncate(f"{status}\n{text}" if text else status)
179
+
180
+
181
+ # Filenames that almost always hold secrets. Reading them just funnels
182
+ # credentials into history/the trajectory, so read_file refuses outright (any
183
+ # mode) — the model should ask the user for a specific value it needs.
184
+ _SECRET_FILE = re.compile(
185
+ r"(?i)(^|/)("
186
+ r"\.env(\.[\w.-]+)?|\.netrc|\.npmrc|\.pypirc|\.git-credentials|"
187
+ r"id_(rsa|dsa|ecdsa|ed25519)|[\w.-]+\.pem|[\w.-]+\.key"
188
+ r")$"
189
+ )
190
+
191
+
192
+ def _is_secret_file(p: Path) -> bool:
193
+ s = str(p)
194
+ if "/.ssh/" in s or "/.aws/credentials" in s or "/.gnupg/" in s:
195
+ return True
196
+ return bool(_SECRET_FILE.search("/" + p.name))
197
+
198
+
199
+ def _jail(
200
+ path: str, workdir: Path, allowed_roots: tuple[Path, ...] = (), grants=(),
201
+ ) -> tuple[Optional[Path], Optional[str]]:
202
+ """Resolve *path* and confine it to *workdir* (plus any *allowed_roots*).
203
+ Returns (resolved_path, None) on success or (None, error_string) if it escapes.
204
+
205
+ *grants* are resolved paths a live human approval widened the jail to — READS
206
+ ONLY (write/edit never pass them). Like allowed_roots they can't come from a
207
+ file, only a launch flag or an in-session approval click, so an untrusted repo
208
+ can't grant itself out.
209
+
210
+ This is a HARD jail inside the tool, not the advisory permission layer:
211
+ bench/serve/yolo all bypass that layer, so read/write/edit must refuse an
212
+ escape on their own. resolve() collapses `..` and follows symlink components
213
+ (a symlink inside workdir that points out, or an absolute path, is caught).
214
+ strict=False so a not-yet-created write target still resolves. Best-effort
215
+ against a symlink swapped in after the check (a local race, out of scope for
216
+ the hostile-model threat model); it fully blocks model-driven `..`/abs/symlink
217
+ escapes. Reads/writes outside these roots must go through the (gated) bash tool.
218
+
219
+ *allowed_roots* are extra in-bounds directories the HUMAN declared at launch
220
+ (`--allow-dir`), already resolved. Deliberately NOT sourced from a project's
221
+ `.rockycode/config.toml` — an untrusted repo must never be able to widen its
222
+ own jail; only an explicit launch-time flag can.
223
+ """
224
+ p = Path(path)
225
+ if not p.is_absolute():
226
+ p = workdir / p
227
+ try:
228
+ resolved = p.resolve()
229
+ wd = workdir.resolve()
230
+ except OSError as e:
231
+ return None, f"[error] {e}"
232
+ roots = (wd, *allowed_roots, *grants)
233
+ if any(resolved == r or r in resolved.parents for r in roots):
234
+ return resolved, None
235
+ extra = (
236
+ f" or a declared --allow-dir root ({', '.join(str(r) for r in allowed_roots)})"
237
+ if allowed_roots else ""
238
+ )
239
+ return None, (
240
+ f"[blocked] path escapes the working directory: {path}. "
241
+ f"read/write/edit are confined to {wd}{extra}."
242
+ )
243
+
244
+
245
+ async def _read_file(path: str, *, workdir: Path, allowed_roots: tuple[Path, ...] = (), read_grants=None,
246
+ offset=None, limit=None) -> str:
247
+ p, err = _jail(path, workdir, allowed_roots, grants=tuple(read_grants or ()))
248
+ if err:
249
+ return err
250
+ if _is_secret_file(p):
251
+ return (
252
+ f"[blocked] refusing to read '{p.name}' — it likely holds secrets "
253
+ f"(.env / credentials / private key). Ask the user for the specific "
254
+ f"value you need instead of reading the file."
255
+ )
256
+ if not p.exists():
257
+ return f"[error] file not found: {p}"
258
+ if p.is_dir():
259
+ entries = "\n".join(sorted(e.name + ("/" if e.is_dir() else "") for e in p.iterdir()))
260
+ return _truncate(f"[directory] {p}\n{entries}")
261
+ try:
262
+ text = p.read_text(errors="replace")
263
+ except OSError as e:
264
+ return f"[error] {e}"
265
+ lines = text.splitlines()
266
+ start = max(int(offset) - 1, 0) if offset else 0
267
+ if start >= len(lines) > 0:
268
+ return f"[error] offset {offset} is past the end of the file ({len(lines)} lines)"
269
+ end = start + int(limit) if limit else len(lines)
270
+ numbered = "\n".join(f"{i + 1}\t{lines[i]}" for i in range(start, min(end, len(lines))))
271
+ return _truncate(numbered)
272
+
273
+
274
+ def _secret_block(p: Path) -> Optional[str]:
275
+ """Error string if writing/editing *p* would clobber a likely-secret file,
276
+ else None. Symmetric with read_file's refusal — a model shouldn't be able to
277
+ overwrite ~/.ssh keys, .env, .npmrc, etc. any more than it can read them."""
278
+ if _is_secret_file(p):
279
+ return (
280
+ f"[blocked] refusing to write '{p.name}' — it looks like a secrets / "
281
+ f"credentials file (.env / key / .ssh / .aws). If you truly need this, "
282
+ f"ask the user to do it."
283
+ )
284
+ return None
285
+
286
+
287
+ async def _write_file(path: str, content: str, *, workdir: Path, allowed_roots: tuple[Path, ...] = ()) -> str:
288
+ p, err = _jail(path, workdir, allowed_roots)
289
+ if err:
290
+ return err
291
+ if (blocked := _secret_block(p)) is not None:
292
+ return blocked
293
+ try:
294
+ p.parent.mkdir(parents=True, exist_ok=True)
295
+ p.write_text(content)
296
+ except OSError as e:
297
+ return f"[error] {e}"
298
+ return f"[ok] wrote {len(content)} chars to {p}"
299
+
300
+
301
+ async def _edit_file(path: str, old_string: str, new_string: str, *, workdir: Path, allowed_roots: tuple[Path, ...] = ()) -> str:
302
+ p, err = _jail(path, workdir, allowed_roots)
303
+ if err:
304
+ return err
305
+ if (blocked := _secret_block(p)) is not None:
306
+ return blocked
307
+ if not p.exists():
308
+ return f"[error] file not found: {p}"
309
+ try:
310
+ raw = p.read_bytes()
311
+ except OSError as e:
312
+ return f"[error] {e}"
313
+ if b"\0" in raw[:8192]: # don't corrupt binaries with a text round-trip
314
+ return f"[error] {p} looks binary — refusing to edit."
315
+ text = raw.decode(errors="replace")
316
+ n = text.count(old_string)
317
+ if n == 0:
318
+ return "[error] old_string not found in file. read the file again — it may have changed."
319
+ if n > 1:
320
+ return f"[error] old_string appears {n} times; include more surrounding context to make it unique."
321
+ try:
322
+ p.write_text(text.replace(old_string, new_string, 1))
323
+ except OSError as e:
324
+ return f"[error] {e}"
325
+ return f"[ok] edited {p}"
326
+
327
+
328
+ def _walk_files(root: Path):
329
+ """Files under root, sorted, skipping junk dirs."""
330
+ stack = [root]
331
+ while stack:
332
+ d = stack.pop()
333
+ try:
334
+ entries = sorted(d.iterdir(), reverse=True)
335
+ except OSError:
336
+ continue
337
+ for e in entries:
338
+ if e.is_dir():
339
+ if e.name not in SKIP_DIRS:
340
+ stack.append(e)
341
+ elif e.is_file():
342
+ yield e
343
+
344
+
345
+ def _grep_sync(pattern: str, path: str, include: Optional[str], workdir: Path) -> str:
346
+ try:
347
+ rx = re.compile(pattern)
348
+ except re.error as e:
349
+ return f"[error] bad regex: {e}"
350
+ root = Path(path) if Path(path).is_absolute() else workdir / path
351
+ if not root.exists():
352
+ return f"[error] path not found: {root}"
353
+ matches: list[str] = []
354
+ for f in _walk_files(root):
355
+ if include and not fnmatch.fnmatch(f.name, include):
356
+ continue
357
+ try:
358
+ # one read: the handle is closed (no fd leak across many files),
359
+ # binary detection and decode share the same bytes (no double read)
360
+ raw = f.read_bytes()
361
+ if b"\0" in raw[:8192]:
362
+ continue # binary
363
+ text = raw.decode(errors="replace")
364
+ except OSError:
365
+ continue
366
+ rel = f.relative_to(root)
367
+ for i, line in enumerate(text.splitlines()):
368
+ if rx.search(line):
369
+ matches.append(f"{rel}:{i + 1}: {line.strip()[:200]}")
370
+ if len(matches) >= GREP_MAX_MATCHES:
371
+ matches.append(f"[truncated at {GREP_MAX_MATCHES} matches — narrow the pattern]")
372
+ return "\n".join(matches)
373
+ return "\n".join(matches) if matches else "[no matches]"
374
+
375
+
376
+ def _glob_sync(pattern: str, workdir: Path) -> str:
377
+ try:
378
+ hits = sorted(
379
+ str(p.relative_to(workdir))
380
+ for p in workdir.glob(pattern)
381
+ if not any(part in SKIP_DIRS for part in p.parts)
382
+ )
383
+ except (ValueError, OSError) as e:
384
+ return f"[error] bad pattern: {e}"
385
+ if not hits:
386
+ return "[no matches]"
387
+ out = hits[:GLOB_MAX_PATHS]
388
+ if len(hits) > GLOB_MAX_PATHS:
389
+ out.append(f"[truncated: {len(hits)} total — narrow the pattern]")
390
+ return "\n".join(out)
391
+
392
+
393
+ async def _grep(pattern: str, path: str = ".", include: Optional[str] = None, *, workdir: Path) -> str:
394
+ return _truncate(await asyncio.to_thread(_grep_sync, pattern, path, include, workdir))
395
+
396
+
397
+ async def _glob(pattern: str, *, workdir: Path) -> str:
398
+ return _truncate(await asyncio.to_thread(_glob_sync, pattern, workdir))
399
+
400
+
401
+ # Static risk tier per built-in tool, shared by build_registry and
402
+ # container.build_session_registry. Read-only = safe; filesystem mutation =
403
+ # moderate; arbitrary shell = risky. engine/permission.py maps (tier, mode) ->
404
+ # allow/ask. Anything not listed falls back to the Tool default ("risky").
405
+ RISK = {
406
+ "read_file": "safe",
407
+ "grep": "safe",
408
+ "glob": "safe",
409
+ "write_file": "moderate",
410
+ "edit_file": "moderate",
411
+ "bash": "risky",
412
+ }
413
+
414
+
415
+ def build_registry(workdir: Path, allowed_roots: tuple[Path, ...] = (), read_grants=None) -> dict[str, Tool]:
416
+ """Tools bound to a local working directory (the chat TUI's registry).
417
+
418
+ *allowed_roots* are extra in-bounds directories the human declared at launch
419
+ (`--allow-dir`); read/write/edit accept paths inside them as well as workdir.
420
+ *read_grants* is a live-mutable set of resolved paths a session approval
421
+ widened the READ jail to (read_file only; never writes) — the read_file
422
+ closure holds the set by reference, so approvals take effect immediately.
423
+ """
424
+ fns = {
425
+ "bash": lambda command: _bash(command, workdir=workdir),
426
+ "read_file": lambda path, offset=None, limit=None: _read_file(
427
+ path, workdir=workdir, allowed_roots=allowed_roots, read_grants=read_grants,
428
+ offset=offset, limit=limit),
429
+ "write_file": lambda path, content: _write_file(path, content, workdir=workdir, allowed_roots=allowed_roots),
430
+ "edit_file": lambda path, old_string, new_string: _edit_file(
431
+ path, old_string, new_string, workdir=workdir, allowed_roots=allowed_roots
432
+ ),
433
+ "grep": lambda pattern, path=".", include=None: _grep(
434
+ pattern, path, include, workdir=workdir
435
+ ),
436
+ "glob": lambda pattern: _glob(pattern, workdir=workdir),
437
+ }
438
+ reg = {
439
+ name: Tool(name=name, schema=SCHEMAS[name], fn=fn, risk=RISK.get(name, "risky"))
440
+ for name, fn in fns.items()
441
+ }
442
+ # check_code: run the project's own linters (see engine/checks.py). Lazy
443
+ # import to avoid a tools<->checks import cycle. Chat registry only — bench
444
+ # runs in the container registry, so published scores are unaffected.
445
+ from rockycode.engine.checks import build_check_tool
446
+ reg.update(build_check_tool(workdir))
447
+ return reg
448
+
449
+
450
+ async def execute(registry: dict[str, Tool], name: str, arguments_json: str) -> tuple[str, bool]:
451
+ """Run a tool by name with JSON-encoded args. Returns (output, ok).
452
+
453
+ Never raises: malformed args / unknown tools / tool crashes all come back
454
+ as error strings so the model can read them and recover.
455
+ """
456
+ tool = registry.get(name)
457
+ if tool is None:
458
+ return f"[error] unknown tool: {name}", False
459
+ try:
460
+ args = json.loads(arguments_json) if arguments_json.strip() else {}
461
+ except json.JSONDecodeError as e:
462
+ return f"[error] malformed tool arguments (invalid JSON): {e}", False
463
+ if not isinstance(args, dict):
464
+ return "[error] tool arguments must be a JSON object", False
465
+ try:
466
+ out = await tool.fn(**args)
467
+ ok = not out.startswith(("[error]", "[timeout]"))
468
+ # Redact secrets before the output enters history: history is what goes
469
+ # to the API prompt AND the trajectory log, so masking here covers both,
470
+ # and the model never sees a raw key it could echo later.
471
+ return redact(out), ok
472
+ except TypeError:
473
+ # Name the accepted params instead of leaking the impl's lambda repr
474
+ # ("build_session_registry.<locals>.<lambda>() got an unexpected …").
475
+ props = tool.schema.get("function", {}).get("parameters", {}).get("properties", {})
476
+ accepted = ", ".join(props) or "see the tool schema"
477
+ return f"[error] bad arguments for {name} — accepted parameters: {accepted}", False
478
+ except Exception as e: # noqa: BLE001 — tool failures go back to the model
479
+ return f"[error] {type(e).__name__}: {e}", False
@@ -0,0 +1,131 @@
1
+ """Trajectory logging in a training-ready format.
2
+
3
+ Every message the engine appends to its conversation history is mirrored to
4
+ a session JSONL file, OpenAI message shape, so a session can later become
5
+ SFT data or an RL rollout with zero retrofitting. This is the bridge from
6
+ "code tool" to "RL environment" — keep it append-only and boring.
7
+
8
+ Line shapes:
9
+ {"t": <unix>, "kind": "meta", "data": {model, thinking, ...}}
10
+ {"t": <unix>, "kind": "message", "data": {role, content, ...}} # as sent to / received from the API
11
+ {"t": <unix>, "kind": "usage", "data": {prompt_tokens, ...}} # per API call
12
+ {"t": <unix>, "kind": "compaction", "data": {strategy, ...}} # history rewrite (see loop._maybe_compact)
13
+ {"t": <unix>, "kind": "outcome", "data": {...}} # reward signal, when known
14
+ {"t": <unix>, "kind": "feedback", "data": {mood, text, local_only}} # user's exit sheet — LOCAL ONLY (see feedback())
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import json
19
+ import os
20
+ import time
21
+ import uuid
22
+ from pathlib import Path
23
+
24
+
25
+ def trajectory_dir() -> Path:
26
+ """The one GLOBAL trajectory store. Honors $ROCKYCODE_HOME (tests and users
27
+ redirect it); defaults to ~/.rockycode/trajectories.
28
+
29
+ Every session lands here regardless of which project/cwd rocky is launched
30
+ from. Each line's meta carries project_id/project_name/workdir, so readers
31
+ (resume picker, dream) filter by project — deliberate global, not the old
32
+ behaviour where a relative path dumped sessions wherever rocky was launched.
33
+ """
34
+ base = os.environ.get("ROCKYCODE_HOME")
35
+ root = Path(base).expanduser() if base else Path.home() / ".rockycode"
36
+ return root / "trajectories"
37
+
38
+
39
+ TRAJECTORY_DIR = trajectory_dir()
40
+
41
+
42
+ class TrajectoryLogger:
43
+ def __init__(self, meta: dict, directory: Path | None = None) -> None:
44
+ if directory is None:
45
+ directory = trajectory_dir()
46
+ # Bench rollouts live in a subdir so they don't mix with chat
47
+ # sessions — the resume picker and dream read only the top level.
48
+ if meta.get("runner") == "rockycode" or meta.get("instance_id"):
49
+ directory = directory / "bench"
50
+ stamp = time.strftime("%Y%m%d-%H%M%S")
51
+ self.session_id = f"{stamp}-{uuid.uuid4().hex[:8]}"
52
+ # Logging is best-effort: an unwritable location (read-only cwd/home,
53
+ # full disk) must NOT crash the session at startup. On failure we set
54
+ # path=None and every write below no-ops. disabled_reason lets a caller
55
+ # surface it if it wants.
56
+ self.path: Path | None = None
57
+ self.disabled_reason: str | None = None
58
+ try:
59
+ directory.mkdir(parents=True, exist_ok=True)
60
+ self.path = directory / f"{self.session_id}.jsonl"
61
+ self._write("meta", meta)
62
+ except OSError as e:
63
+ self.path = None
64
+ self.disabled_reason = str(e)
65
+
66
+ def _write(self, kind: str, data: dict) -> None:
67
+ if self.path is None:
68
+ return # logging disabled (unwritable location) — silently skip
69
+ line = json.dumps({"t": time.time(), "kind": kind, "data": data}, ensure_ascii=False)
70
+ # encoding="utf-8" is required: ensure_ascii=False emits raw non-ASCII, and
71
+ # the platform default (cp1252 on Windows) would raise UnicodeEncodeError on
72
+ # the first emoji/CJK char and kill the session's logging.
73
+ try:
74
+ with self.path.open("a", encoding="utf-8") as f:
75
+ f.write(line + "\n")
76
+ except OSError as e:
77
+ # Went unwritable mid-session (disk full, unmounted) — stop quietly
78
+ # rather than take the turn down with us.
79
+ self.path = None
80
+ self.disabled_reason = str(e)
81
+
82
+ def message(self, msg: dict) -> None:
83
+ self._write("message", msg)
84
+
85
+ def reasoning(self, text: str) -> None:
86
+ """The turn's reasoning_content. Never part of history (DeepSeek 400s
87
+ if it is sent back), so without this record the trace exists only as
88
+ ephemeral stream deltas — but language-adherence analysis and the
89
+ RL-export path (self-evolve phase 3) both need it."""
90
+ if text:
91
+ self._write("reasoning", {"text": text})
92
+
93
+ def usage(self, usage: dict) -> None:
94
+ if usage:
95
+ self._write("usage", usage)
96
+
97
+ def compaction(self, data: dict) -> None:
98
+ self._write("compaction", data)
99
+
100
+ def outcome(self, data: dict) -> None:
101
+ self._write("outcome", data)
102
+
103
+ def feedback(self, data: dict) -> None:
104
+ # The user's exit-sheet rating. LOCAL ONLY by contract: the sheet
105
+ # promises "never sent to the model provider", so readers must never
106
+ # place feedback records in any cloud-bound prompt — only the local
107
+ # dream (Ollama) consumes them. See the self-evolve design.
108
+ self._write("feedback", data)
109
+
110
+ def note(self, data: dict) -> None:
111
+ self._write("note", data)
112
+
113
+ def title(self, text: str) -> None:
114
+ # Appended once the session has a name (see engine/titler.py); readers
115
+ # take the LAST title record, so regenerating just appends a fresher one.
116
+ if text and text.strip():
117
+ self._write("title", {"title": text.strip()})
118
+
119
+
120
+ def append_record(path: Path, kind: str, data: dict) -> bool:
121
+ """Append one record to an EXISTING trajectory file — for post-hoc writers
122
+ like the dream-time judge, which grades sessions long after their logger
123
+ is gone. Same line shape and best-effort contract as TrajectoryLogger:
124
+ returns False instead of raising."""
125
+ try:
126
+ line = json.dumps({"t": time.time(), "kind": kind, "data": data}, ensure_ascii=False)
127
+ with path.open("a", encoding="utf-8") as f:
128
+ f.write(line + "\n")
129
+ return True
130
+ except OSError:
131
+ return False