rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,176 @@
1
+ """Raw single-shot runner: model gets the problem statement and emits a unified
2
+ diff. No tools, no iteration. This is the lower-bound baseline against which
3
+ the rockycode harness's value-add is measured.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ import json
8
+ import re
9
+ from datetime import datetime
10
+ from pathlib import Path
11
+ from typing import Optional
12
+
13
+ from openai import OpenAI
14
+ from rich.console import Console
15
+ from rich.progress import (
16
+ BarColumn,
17
+ Progress,
18
+ SpinnerColumn,
19
+ TextColumn,
20
+ TimeElapsedColumn,
21
+ )
22
+ from rich.table import Table
23
+
24
+ from rockycode.banner import amaze, confused, fail
25
+ from rockycode.engine.effort import build_extra_body
26
+ from rockycode.onboarding import require_base_url, require_key
27
+ from rockycode.palette import PURPLE, VIOLET
28
+ from rockycode.prompts.rocky import RAW_SINGLE_SHOT
29
+ from rockycode.runners.data import load_verified
30
+
31
+ PREDICTIONS_DIR = Path("results") / "predictions"
32
+
33
+ # Pull the diff out of ```diff …``` first, then any fenced block, else raw.
34
+ _DIFF_FENCE = re.compile(r"```diff\s*\n(.*?)```", re.DOTALL)
35
+ _ANY_FENCE = re.compile(r"```\w*\s*\n(.*?)```", re.DOTALL)
36
+
37
+
38
+ def _extract_diff(text: str) -> str:
39
+ if m := _DIFF_FENCE.search(text):
40
+ return m.group(1)
41
+ if m := _ANY_FENCE.search(text):
42
+ return m.group(1)
43
+ return text
44
+
45
+
46
+ def _build_prompt(instance: dict) -> str:
47
+ return RAW_SINGLE_SHOT.format(
48
+ repo=instance.get("repo", "unknown"),
49
+ base_commit=instance.get("base_commit", "unknown"),
50
+ problem_statement=instance.get("problem_statement", ""),
51
+ hints=instance.get("hints_text", "") or "(none)",
52
+ )
53
+
54
+
55
+ def _extract_usage(resp) -> dict:
56
+ """Flatten resp.usage including DeepSeek-specific extras (cache hit/miss).
57
+
58
+ DeepSeek surfaces `prompt_cache_hit_tokens` and `prompt_cache_miss_tokens`
59
+ in usage, which aren't in the openai-python SDK's typed Usage model.
60
+ Pull them via model_dump so we don't lose them.
61
+ """
62
+ if resp.usage is None:
63
+ return {}
64
+ try:
65
+ return resp.usage.model_dump()
66
+ except AttributeError:
67
+ return dict(resp.usage)
68
+
69
+
70
+ def run(
71
+ model: str,
72
+ instance_ids: Optional[list[str]],
73
+ console: Console,
74
+ *,
75
+ thinking: bool = True,
76
+ reasoning_effort: str = "max",
77
+ max_tokens: int = 16384,
78
+ task_label: str = "tasks",
79
+ output_dir: Path = PREDICTIONS_DIR,
80
+ ) -> Path:
81
+ """Generate single-shot predictions for the requested instances.
82
+
83
+ Returns the path to a SWE-bench-format predictions JSONL file.
84
+ """
85
+ ds = load_verified(console, instance_ids)
86
+ console.print(f" → {len(ds)} instances to predict\n")
87
+
88
+ output_dir.mkdir(parents=True, exist_ok=True)
89
+ safe_model = model.replace("/", "-")
90
+ # task-set + timestamp: no run ever overwrites another
91
+ stamp = datetime.now().strftime("%Y%m%d-%H%M")
92
+ out_path = output_dir / f"raw-{safe_model}-{task_label}-{stamp}.jsonl"
93
+
94
+ # Key AND endpoint from rocky's credential chain — never the SDK's ambient
95
+ # OPENAI_API_KEY / OPENAI_BASE_URL fallbacks.
96
+ # max_retries=5: DeepSeek is flakier than OpenAI; SDK default of 2 is too low.
97
+ client = OpenAI(api_key=require_key(), base_url=require_base_url(),
98
+ max_retries=5, timeout=120.0)
99
+ extra_body = build_extra_body(thinking, reasoning_effort)
100
+
101
+ n_ok = n_fail = 0
102
+ totals = {
103
+ "prompt_tokens": 0,
104
+ "completion_tokens": 0,
105
+ "prompt_cache_hit_tokens": 0,
106
+ "prompt_cache_miss_tokens": 0,
107
+ }
108
+
109
+ with out_path.open("w") as f, Progress(
110
+ SpinnerColumn(style=PURPLE),
111
+ TextColumn("[progress.description]{task.description}"),
112
+ BarColumn(complete_style=PURPLE, finished_style=VIOLET),
113
+ TextColumn("{task.completed}/{task.total}"),
114
+ TimeElapsedColumn(),
115
+ console=console,
116
+ ) as progress:
117
+ bar = progress.add_task("predicting", total=len(ds))
118
+ for instance in ds:
119
+ iid = instance["instance_id"]
120
+ try:
121
+ resp = client.chat.completions.create(
122
+ model=model,
123
+ messages=[{"role": "user", "content": _build_prompt(instance)}],
124
+ temperature=0.0,
125
+ max_tokens=max_tokens,
126
+ extra_body=extra_body,
127
+ )
128
+ content = resp.choices[0].message.content or ""
129
+ patch = _extract_diff(content).strip()
130
+ if patch and not patch.endswith("\n"):
131
+ patch += "\n"
132
+
133
+ usage = _extract_usage(resp)
134
+ for k in totals:
135
+ totals[k] += usage.get(k, 0) or 0
136
+
137
+ n_ok += 1
138
+ except Exception as e: # noqa: BLE001 — log + record empty patch, keep going
139
+ fail(console, f"{iid}: {e}")
140
+ patch = ""
141
+ n_fail += 1
142
+
143
+ f.write(json.dumps({
144
+ "instance_id": iid,
145
+ "model_name_or_path": f"raw-{safe_model}",
146
+ "model_patch": patch,
147
+ }) + "\n")
148
+ f.flush()
149
+ progress.update(bar, advance=1)
150
+
151
+ if n_fail == 0:
152
+ amaze(console, f"all {n_ok} predictions written to {out_path}!")
153
+ else:
154
+ confused(console, f"{n_ok} ok, {n_fail} failed. predictions at {out_path}")
155
+
156
+ _print_usage_summary(console, totals)
157
+ return out_path
158
+
159
+
160
+ def _print_usage_summary(console: Console, totals: dict) -> None:
161
+ prompt = totals["prompt_tokens"]
162
+ completion = totals["completion_tokens"]
163
+ cache_hit = totals["prompt_cache_hit_tokens"]
164
+ cache_miss = totals["prompt_cache_miss_tokens"]
165
+ cache_seen = cache_hit + cache_miss
166
+ hit_rate = (cache_hit / cache_seen * 100) if cache_seen else 0.0
167
+
168
+ table = Table(title="token usage", show_header=True, header_style=f"bold {VIOLET}")
169
+ table.add_column("metric")
170
+ table.add_column("value", justify="right")
171
+ table.add_row("prompt tokens", f"{prompt:,}")
172
+ table.add_row("completion tokens", f"{completion:,}")
173
+ table.add_row("cache hit tokens", f"{cache_hit:,}")
174
+ table.add_row("cache miss tokens", f"{cache_miss:,}")
175
+ table.add_row("cache hit rate", f"{hit_rate:.1f}%")
176
+ console.print(table)
rockycode/score.py ADDED
@@ -0,0 +1,114 @@
1
+ """Wraps SWE-bench's official evaluation harness as a subprocess.
2
+
3
+ We call `python -m swebench.harness.run_evaluation` rather than importing
4
+ internals — the harness's public CLI is its stable contract; its Python API
5
+ moves between releases.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import subprocess
11
+ import sys
12
+ from pathlib import Path
13
+ from typing import Optional
14
+
15
+ from rich.console import Console
16
+ from rich.table import Table
17
+
18
+ from rockycode.banner import amaze, fail, info
19
+ from rockycode.palette import VIOLET
20
+
21
+ DATASET = "princeton-nlp/SWE-bench_Verified"
22
+
23
+
24
+ def score(
25
+ predictions_path: Path,
26
+ run_id: str,
27
+ instance_ids: Optional[list[str]],
28
+ console: Console,
29
+ max_workers: int = 4,
30
+ timeout: int = 1800,
31
+ ) -> dict:
32
+ """Run swebench eval and parse the report. Returns the parsed report dict."""
33
+ info(console, f"scoring run_id={run_id}")
34
+ info(console, f"predictions={predictions_path}")
35
+
36
+ cmd = [
37
+ sys.executable, "-m", "swebench.harness.run_evaluation",
38
+ "--dataset_name", DATASET,
39
+ "--predictions_path", str(predictions_path),
40
+ "--max_workers", str(max_workers),
41
+ "--run_id", run_id,
42
+ "--timeout", str(timeout),
43
+ ]
44
+ if instance_ids:
45
+ cmd += ["--instance_ids", *instance_ids]
46
+
47
+ info(console, f"$ {' '.join(cmd)}")
48
+ proc = subprocess.run(cmd, check=False)
49
+ if proc.returncode != 0:
50
+ fail(console, f"swebench harness exited {proc.returncode}")
51
+ raise SystemExit(proc.returncode)
52
+
53
+ report = _find_report(run_id)
54
+ if report is None:
55
+ fail(console, "no report file found after eval. check logs/run_evaluation/")
56
+ raise SystemExit(1)
57
+
58
+ _print_report(console, report, run_id)
59
+ return report
60
+
61
+
62
+ def _find_report(run_id: str) -> Optional[dict]:
63
+ """The harness writes its summary as <model_name>.<run_id>.json in cwd
64
+ (older versions) or under logs/run_evaluation/<run_id>/ (newer). Match run_id
65
+ EXACTLY (delimited) and take the NEWEST match: a loose `*run_id*` glob matched
66
+ 'v1' inside a 'v12' report, and with no freshness order a re-run of the same
67
+ run_id could pick up a stale report."""
68
+ candidates: list[Path] = []
69
+ candidates += list(Path(".").glob(f"*.{run_id}.json"))
70
+ candidates += list(Path("logs/run_evaluation").glob(f"{run_id}/**/results.json"))
71
+ candidates += list(Path("logs/run_evaluation").glob(f"{run_id}/**/*.json"))
72
+ candidates.sort(key=lambda p: p.stat().st_mtime if p.exists() else 0.0, reverse=True)
73
+ for path in candidates:
74
+ try:
75
+ data = json.loads(path.read_text(encoding="utf-8"))
76
+ if isinstance(data, dict) and ("resolved_ids" in data or "resolved_instances" in data):
77
+ return data
78
+ except (json.JSONDecodeError, OSError):
79
+ continue
80
+ return None
81
+
82
+
83
+ def _count(report: dict, ids_key: str, count_key: str) -> int:
84
+ v = report.get(ids_key)
85
+ if isinstance(v, list):
86
+ return len(v)
87
+ c = report.get(count_key)
88
+ return int(c) if isinstance(c, (int, float)) else 0
89
+
90
+
91
+ def _print_report(console: Console, report: dict, run_id: str) -> None:
92
+ resolved = _count(report, "resolved_ids", "resolved_instances")
93
+ submitted = _count(report, "submitted_ids", "submitted_instances")
94
+ error = _count(report, "error_ids", "error_instances")
95
+ total = report.get("total_instances") or 0
96
+ # Score over what we actually submitted. swebench's total_instances is the
97
+ # whole Verified set (500) — dividing by it made a 1-task run read as 0.2%.
98
+ pct = (resolved / submitted * 100) if submitted else 0.0
99
+
100
+ info(console, f"run: {run_id}")
101
+ table = Table(show_header=True, header_style=f"bold {VIOLET}")
102
+ table.add_column("metric")
103
+ table.add_column("value", justify="right")
104
+ table.add_row("submitted", str(submitted))
105
+ table.add_row("resolved", str(resolved))
106
+ table.add_row("errors", str(error))
107
+ table.add_row("score (resolved / submitted)", f"[bold]{pct:.1f}%[/bold]")
108
+ table.add_row("[dim]verified set size[/dim]", f"[dim]{total}[/dim]")
109
+ console.print(table)
110
+
111
+ if pct > 0:
112
+ amaze(console, f"score {pct:.1f}%! amaze!")
113
+ else:
114
+ console.print("[dim]we no resolve any. we try harness next, learn more.[/dim]")
rockycode/session.py ADDED
@@ -0,0 +1,298 @@
1
+ """Session storage: stable project identity + cross-folder discovery.
2
+
3
+ Claude Code / Codex key sessions by a folder's absolute path in a global
4
+ store, so renaming or moving the folder ORPHANS its sessions. rockycode does
5
+ it differently:
6
+
7
+ - Each project gets a STABLE id in `.rockycode/project.json`. The file lives
8
+ in the folder, so it travels on rename/move — the project keeps its identity
9
+ no matter where it is or what it's called.
10
+ - Trajectories live in ONE global store (`~/.rockycode/trajectories`); each
11
+ file's meta records which project it belongs to, so sessions follow the
12
+ project identity, not the folder path.
13
+ - A tiny global registry (`~/.rockycode/projects.json`) records where each
14
+ project currently lives, so sessions are discoverable ACROSS folders — the
15
+ resume picker can list "this folder" or "all folders", and `--resume <id>`
16
+ can land back in a project's CURRENT folder even after a rename.
17
+
18
+ So: sessions survive folder renames AND are searchable across the machine —
19
+ which neither Claude Code nor Codex manage.
20
+
21
+ Public session ids are `rk_<hash>` — the uuid tail of the trajectory stem
22
+ (filenames keep their sortable timestamp form on disk; humans get the short
23
+ hash, like opencode's `ses_…`).
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import json
28
+ import os
29
+ import time
30
+ import uuid
31
+ from dataclasses import dataclass
32
+ from pathlib import Path
33
+ from typing import Optional
34
+
35
+ _HOME_ENV = os.environ.get("ROCKYCODE_HOME")
36
+ HOME_ROOT = Path(_HOME_ENV).expanduser() if _HOME_ENV else Path.home() / ".rockycode"
37
+ REGISTRY = HOME_ROOT / "projects.json"
38
+ PROJECT_REL = Path(".rockycode") / "project.json"
39
+ TRAJ_REL = Path(".rockycode") / "trajectories"
40
+
41
+
42
+ @dataclass
43
+ class Project:
44
+ id: str
45
+ name: str
46
+ root: Path
47
+
48
+
49
+ @dataclass
50
+ class SessionInfo:
51
+ session_id: str
52
+ path: Path
53
+ project_id: str
54
+ project_name: str
55
+ project_path: str
56
+ model: str
57
+ started_at: float
58
+ n_messages: int
59
+ summary: str
60
+ title: str = "" # flash-generated; empty on old/offline sessions
61
+
62
+ @property
63
+ def display_title(self) -> str:
64
+ return self.title or self.summary
65
+
66
+
67
+ # ---- project identity + registry -------------------------------------------
68
+
69
+ def get_project(workdir: Path) -> Project:
70
+ """Stable identity for a folder. Creates `.rockycode/project.json` on first
71
+ use; that file travels with the folder, so the id survives renames."""
72
+ workdir = Path(workdir).resolve()
73
+ pf = workdir / PROJECT_REL
74
+ if pf.exists():
75
+ try:
76
+ data = json.loads(pf.read_text(encoding="utf-8"))
77
+ proj = Project(id=data["project_id"], name=data.get("name", workdir.name), root=workdir)
78
+ except (json.JSONDecodeError, OSError, KeyError):
79
+ proj = _new_project(workdir, pf)
80
+ else:
81
+ proj = _new_project(workdir, pf)
82
+ _register(proj)
83
+ return proj
84
+
85
+
86
+ def _new_project(workdir: Path, pf: Path) -> Project:
87
+ proj = Project(id=uuid.uuid4().hex, name=workdir.name, root=workdir)
88
+ try:
89
+ pf.parent.mkdir(parents=True, exist_ok=True)
90
+ pf.write_text(
91
+ json.dumps({"project_id": proj.id, "name": proj.name, "created": time.time()}, indent=2),
92
+ encoding="utf-8",
93
+ )
94
+ except OSError:
95
+ # Read-only / unwritable cwd: use an ephemeral id for this run instead of
96
+ # crashing chat at startup. It just won't persist across launches (so
97
+ # resume-by-project won't find it), which is the right degradation.
98
+ pass
99
+ return proj
100
+
101
+
102
+ def _load_registry() -> dict:
103
+ if REGISTRY.exists():
104
+ try:
105
+ return json.loads(REGISTRY.read_text(encoding="utf-8"))
106
+ except (json.JSONDecodeError, OSError):
107
+ return {}
108
+ return {}
109
+
110
+
111
+ def _register(proj: Project) -> None:
112
+ """Record (and self-heal) where this project lives. `current` is the
113
+ authoritative path; old paths linger but are harmless. Best-effort: the
114
+ registry only powers cross-folder resume, so an unwritable home must never
115
+ block startup."""
116
+ try:
117
+ HOME_ROOT.mkdir(parents=True, exist_ok=True)
118
+ reg = _load_registry()
119
+ entry = reg.get(proj.id, {})
120
+ paths = set(entry.get("paths", []))
121
+ paths.add(str(proj.root))
122
+ reg[proj.id] = {
123
+ "name": proj.name,
124
+ "current": str(proj.root),
125
+ "paths": sorted(paths),
126
+ "last_seen": time.time(),
127
+ }
128
+ REGISTRY.write_text(json.dumps(reg, indent=2), encoding="utf-8")
129
+ except OSError:
130
+ pass
131
+
132
+
133
+ # ---- session discovery -----------------------------------------------------
134
+
135
+ def _is_chat_session(meta: dict) -> bool:
136
+ # bench, goal, and routine runs land in the same trajectories dir but
137
+ # carry a runner / instance_id; the resume picker only wants interactive
138
+ # chat sessions. (Goal/routine runs DO carry project_id now — that's for
139
+ # the dream, which grades them alongside chats, not for the picker.)
140
+ return "instance_id" not in meta and meta.get("runner") not in ("rockycode", "goal", "routine")
141
+
142
+
143
+ def global_traj_dir() -> Path:
144
+ """The single global trajectory store. Read at call time so a test that
145
+ redirects HOME_ROOT takes effect."""
146
+ return HOME_ROOT / "trajectories"
147
+
148
+
149
+ def _read_info(traj: Path) -> Optional[SessionInfo]:
150
+ """Build a SessionInfo from a trajectory; project identity comes from its
151
+ own meta (project_id/project_name/workdir), since all sessions share the
152
+ one global dir now."""
153
+ try:
154
+ lines = traj.read_text(encoding="utf-8", errors="replace").splitlines()
155
+ except OSError:
156
+ return None
157
+ if not lines:
158
+ return None
159
+ meta: dict = {}
160
+ summary = ""
161
+ title = ""
162
+ n_messages = 0
163
+ started_at = traj.stat().st_mtime
164
+ for ln in lines:
165
+ try:
166
+ rec = json.loads(ln)
167
+ except json.JSONDecodeError:
168
+ continue
169
+ kind, data = rec.get("kind"), rec.get("data", {})
170
+ if kind == "meta":
171
+ meta = data
172
+ started_at = rec.get("t", started_at)
173
+ elif kind == "title":
174
+ # last one wins — the trajectory is append-only, so regenerating a
175
+ # title later just appends a fresher record
176
+ t = data.get("title")
177
+ if isinstance(t, str) and t.strip():
178
+ title = t.strip()
179
+ elif kind == "message":
180
+ n_messages += 1
181
+ if not summary and data.get("role") == "user":
182
+ c = data.get("content")
183
+ if isinstance(c, str) and c.strip():
184
+ summary = c.strip().splitlines()[0][:80]
185
+ if not _is_chat_session(meta):
186
+ return None
187
+ workdir = meta.get("workdir", "")
188
+ return SessionInfo(
189
+ session_id=traj.stem,
190
+ path=traj,
191
+ project_id=meta.get("project_id", ""),
192
+ project_name=meta.get("project_name") or (Path(workdir).name if workdir else "?"),
193
+ project_path=workdir,
194
+ model=meta.get("model", "?"),
195
+ started_at=started_at,
196
+ n_messages=n_messages,
197
+ summary=summary or "(no message)",
198
+ title=title,
199
+ )
200
+
201
+
202
+ def list_sessions(
203
+ scope: str = "project",
204
+ *,
205
+ workdir: Optional[Path] = None,
206
+ query: Optional[str] = None,
207
+ limit: int = 50,
208
+ ) -> list[SessionInfo]:
209
+ """Chat sessions from the global store, newest first. scope='project'
210
+ keeps only this workdir's project (by project_id); scope='all' keeps every
211
+ project. query filters by summary/project-name substring."""
212
+ traj_dir = global_traj_dir()
213
+ if not traj_dir.is_dir():
214
+ return []
215
+ target_pid: Optional[str] = None
216
+ if scope != "all" and workdir is not None:
217
+ target_pid = get_project(workdir).id
218
+ infos: list[SessionInfo] = []
219
+ for f in traj_dir.glob("*.jsonl"):
220
+ info = _read_info(f)
221
+ if info is None:
222
+ continue
223
+ if target_pid is not None and info.project_id != target_pid:
224
+ continue
225
+ infos.append(info)
226
+ if query:
227
+ q = query.lower()
228
+ infos = [i for i in infos
229
+ if q in i.title.lower() or q in i.summary.lower() or q in i.project_name.lower()]
230
+ infos.sort(key=lambda i: i.started_at, reverse=True)
231
+ return infos[:limit]
232
+
233
+
234
+ # ---- public session ids ------------------------------------------------------
235
+
236
+ ID_PREFIX = "rk_"
237
+
238
+
239
+ def public_id(session_id: str) -> str:
240
+ """`rk_ab12cd34` — the uuid tail of the trajectory stem, rocky-prefixed.
241
+ The filename keeps its sortable stamp form on disk; humans get the hash."""
242
+ return ID_PREFIX + session_id.rsplit("-", 1)[-1]
243
+
244
+
245
+ def resolve_session(token: str) -> tuple[Optional[SessionInfo], str]:
246
+ """One session from a user-typed id. Accepted forms: `rk_ab12cd34`, bare
247
+ `ab12cd34`, a unique hash prefix (≥4 chars), or a full legacy stem.
248
+ Returns (info, error) — exactly one is set."""
249
+ t = token.strip().lower()
250
+ if t.startswith(ID_PREFIX):
251
+ t = t[len(ID_PREFIX):]
252
+ if not t:
253
+ return None, "empty session id"
254
+ sessions = list_sessions(scope="all", limit=100_000)
255
+ exact = [s for s in sessions if s.session_id.lower() == t]
256
+ if exact:
257
+ return exact[0], ""
258
+ if len(t) >= 4:
259
+ hits = [s for s in sessions
260
+ if s.session_id.rsplit("-", 1)[-1].lower().startswith(t)]
261
+ if len(hits) == 1:
262
+ return hits[0], ""
263
+ if len(hits) > 1:
264
+ opts = ", ".join(f"{public_id(s.session_id)} ({s.display_title[:30]})" for s in hits[:5])
265
+ return None, f"'{token}' is ambiguous — matches {opts}"
266
+ return None, f"no session matches '{token}' — run `rockycode --resume` to browse"
267
+
268
+
269
+ def project_current_path(project_id: str) -> Optional[Path]:
270
+ """Where a project lives NOW, per the registry — survives folder renames.
271
+ None when the registry has no entry (or the recorded path is gone)."""
272
+ entry = _load_registry().get(project_id) or {}
273
+ cur = entry.get("current")
274
+ if cur and Path(cur).is_dir():
275
+ return Path(cur)
276
+ for p in reversed(entry.get("paths", [])):
277
+ if Path(p).is_dir():
278
+ return Path(p)
279
+ return None
280
+
281
+
282
+ def load_history(traj: Path) -> list[dict]:
283
+ """Reconstruct an engine message history from a trajectory file — exactly
284
+ the {role, content, tool_calls, tool_call_id} dicts that were sent to the
285
+ API (reasoning_content was never stored, so this is clean to replay)."""
286
+ history: list[dict] = []
287
+ try:
288
+ lines = Path(traj).read_text(encoding="utf-8", errors="replace").splitlines()
289
+ except OSError:
290
+ return history
291
+ for ln in lines:
292
+ try:
293
+ rec = json.loads(ln)
294
+ except json.JSONDecodeError:
295
+ continue
296
+ if rec.get("kind") == "message":
297
+ history.append(rec["data"])
298
+ return history
@@ -0,0 +1,71 @@
1
+ ---
2
+ name: architecture-viz
3
+ description: Visualize a neural-net block (attention, transformer, MoE, diffusion-LLM) as an artifact three linked ways — the paper diagram, the tensor-shape ribbon, and the user's PyTorch — cross-highlighted so hovering one lights the other two. Grounded in the real paper/code, never drawn from memory.
4
+ ---
5
+
6
+ # architecture-viz — one block, three linked views
7
+
8
+ The register (LOCKED — the user chose it after rejecting two others):
9
+ **a recognizable paper diagram + a shape ribbon + the user's PyTorch code, all
10
+ cross-highlighted.** Hover any box, shape-chip, or code line and the matching
11
+ two light up. That link is the whole point — it's what turns an abstract figure
12
+ into something the user can tie to the code they actually write.
13
+
14
+ ## Why exactly these three (do not drop any)
15
+
16
+ - **Paper diagram** — the figure people recognize at a glance (e.g. Attention
17
+ Is All You Need's scaled-dot-product / multi-head). The anchor.
18
+ - **Shape ribbon** — how the tensor shape changes step by step
19
+ (`(n,d) → (n,dₖ)×3 → (n,n) → (n,n) → (n,dₖ)`). This is the real value.
20
+ - **PyTorch** — the language the user writes. NEVER show a formal language
21
+ (Lean/TorchLean syntax) as the teaching vehicle: a language they never
22
+ learned teaches nothing and frustrates. Proving stays a plain-English
23
+ footnote only: "want a shape proven for every n? `/research prove`".
24
+
25
+ The three share a `data-step` id per stage; that shared id IS the cross-link.
26
+
27
+ ## Ground it — never draw from memory
28
+
29
+ The diagram, shapes, and code must be DERIVED, not remembered — a subtly wrong
30
+ architecture is worse than none (the user acts on it). Risk scales with how
31
+ novel the block is:
32
+
33
+ - **Standard transformer / attention** — well-known; still label real-vs-example.
34
+ - **DeepSeek V3.2 MoE, diffusion-LLM, anything recent** — HIGH risk from memory.
35
+ First get ground truth: `read_file` the user's model code if they have it, or
36
+ `web_fetch` the paper / model card, and build the diagram + shapes + torch
37
+ from THAT. Diffusion-LLM is not autoregressive — its "forward" is a denoising
38
+ loop; don't force it into a transformer figure.
39
+ - Put a provenance line at the bottom: `◆ derived` (what came from code/paper)
40
+ vs `◇ illustrative` (example sizes). Be honest about which.
41
+
42
+ ## Build it — rocky artifact rules (important)
43
+
44
+ `create_artifact` takes BODY content only and STRIPS every `<style>` block, then
45
+ applies rocky's light theme. So:
46
+
47
+ - **No `<style>` block** — it will be deleted. Style with **inline `style=`**
48
+ attributes only (those survive), using rocky's light palette:
49
+ bg is themed for you; use `#efeafa` panels, `#9d7cd8`/`#7c5cba` strokes/accent,
50
+ `#6a4ca3` headings, `#2a2a38` text, `#ede6fb` for the lit/highlight state.
51
+ - The cross-highlight is done in a **`<script>`** (scripts survive): on
52
+ `mouseenter`/`focus` of any `[data-step]`, set inline styles on every element
53
+ with the same `data-step`; revert on `mouseleave`/`blur`. No CSS `:hover`.
54
+ - Self-contained: no CDN, no external fonts/images. Inline SVG for the diagram.
55
+ - Use rocky's classes where they fit: `card`, `tag`/`tag-purple`/`tag-amber`.
56
+ - Reuse the SAME artifact title to update in place on a re-run.
57
+
58
+ ## The scaffold
59
+
60
+ `template.html` in this skill's directory is a complete, working example (scaled
61
+ dot-product attention) in exactly this shape. **Copy it and adapt** the three
62
+ panels for the block at hand — swap the SVG figure, the ribbon chips, and the
63
+ torch lines, keeping the `data-step` ids matched across all three. Don't
64
+ re-derive the highlight script; reuse it.
65
+
66
+ ## Multi-panel architectures
67
+
68
+ For a whole model (MoE: router → experts → combine; a transformer layer:
69
+ attention + FFN + residual/norm; diffusion: the denoising steps), give each
70
+ sub-block its own diagram+ribbon+torch row, each internally linked, stacked top
71
+ to bottom — one story, every stage labeled with the shape it carries.