rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/runners/raw.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Raw single-shot runner: model gets the problem statement and emits a unified
|
|
2
|
+
diff. No tools, no iteration. This is the lower-bound baseline against which
|
|
3
|
+
the rockycode harness's value-add is measured.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import re
|
|
9
|
+
from datetime import datetime
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Optional
|
|
12
|
+
|
|
13
|
+
from openai import OpenAI
|
|
14
|
+
from rich.console import Console
|
|
15
|
+
from rich.progress import (
|
|
16
|
+
BarColumn,
|
|
17
|
+
Progress,
|
|
18
|
+
SpinnerColumn,
|
|
19
|
+
TextColumn,
|
|
20
|
+
TimeElapsedColumn,
|
|
21
|
+
)
|
|
22
|
+
from rich.table import Table
|
|
23
|
+
|
|
24
|
+
from rockycode.banner import amaze, confused, fail
|
|
25
|
+
from rockycode.engine.effort import build_extra_body
|
|
26
|
+
from rockycode.onboarding import require_base_url, require_key
|
|
27
|
+
from rockycode.palette import PURPLE, VIOLET
|
|
28
|
+
from rockycode.prompts.rocky import RAW_SINGLE_SHOT
|
|
29
|
+
from rockycode.runners.data import load_verified
|
|
30
|
+
|
|
31
|
+
PREDICTIONS_DIR = Path("results") / "predictions"
|
|
32
|
+
|
|
33
|
+
# Pull the diff out of ```diff …``` first, then any fenced block, else raw.
|
|
34
|
+
_DIFF_FENCE = re.compile(r"```diff\s*\n(.*?)```", re.DOTALL)
|
|
35
|
+
_ANY_FENCE = re.compile(r"```\w*\s*\n(.*?)```", re.DOTALL)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _extract_diff(text: str) -> str:
|
|
39
|
+
if m := _DIFF_FENCE.search(text):
|
|
40
|
+
return m.group(1)
|
|
41
|
+
if m := _ANY_FENCE.search(text):
|
|
42
|
+
return m.group(1)
|
|
43
|
+
return text
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _build_prompt(instance: dict) -> str:
|
|
47
|
+
return RAW_SINGLE_SHOT.format(
|
|
48
|
+
repo=instance.get("repo", "unknown"),
|
|
49
|
+
base_commit=instance.get("base_commit", "unknown"),
|
|
50
|
+
problem_statement=instance.get("problem_statement", ""),
|
|
51
|
+
hints=instance.get("hints_text", "") or "(none)",
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _extract_usage(resp) -> dict:
|
|
56
|
+
"""Flatten resp.usage including DeepSeek-specific extras (cache hit/miss).
|
|
57
|
+
|
|
58
|
+
DeepSeek surfaces `prompt_cache_hit_tokens` and `prompt_cache_miss_tokens`
|
|
59
|
+
in usage, which aren't in the openai-python SDK's typed Usage model.
|
|
60
|
+
Pull them via model_dump so we don't lose them.
|
|
61
|
+
"""
|
|
62
|
+
if resp.usage is None:
|
|
63
|
+
return {}
|
|
64
|
+
try:
|
|
65
|
+
return resp.usage.model_dump()
|
|
66
|
+
except AttributeError:
|
|
67
|
+
return dict(resp.usage)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def run(
|
|
71
|
+
model: str,
|
|
72
|
+
instance_ids: Optional[list[str]],
|
|
73
|
+
console: Console,
|
|
74
|
+
*,
|
|
75
|
+
thinking: bool = True,
|
|
76
|
+
reasoning_effort: str = "max",
|
|
77
|
+
max_tokens: int = 16384,
|
|
78
|
+
task_label: str = "tasks",
|
|
79
|
+
output_dir: Path = PREDICTIONS_DIR,
|
|
80
|
+
) -> Path:
|
|
81
|
+
"""Generate single-shot predictions for the requested instances.
|
|
82
|
+
|
|
83
|
+
Returns the path to a SWE-bench-format predictions JSONL file.
|
|
84
|
+
"""
|
|
85
|
+
ds = load_verified(console, instance_ids)
|
|
86
|
+
console.print(f" → {len(ds)} instances to predict\n")
|
|
87
|
+
|
|
88
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
89
|
+
safe_model = model.replace("/", "-")
|
|
90
|
+
# task-set + timestamp: no run ever overwrites another
|
|
91
|
+
stamp = datetime.now().strftime("%Y%m%d-%H%M")
|
|
92
|
+
out_path = output_dir / f"raw-{safe_model}-{task_label}-{stamp}.jsonl"
|
|
93
|
+
|
|
94
|
+
# Key AND endpoint from rocky's credential chain — never the SDK's ambient
|
|
95
|
+
# OPENAI_API_KEY / OPENAI_BASE_URL fallbacks.
|
|
96
|
+
# max_retries=5: DeepSeek is flakier than OpenAI; SDK default of 2 is too low.
|
|
97
|
+
client = OpenAI(api_key=require_key(), base_url=require_base_url(),
|
|
98
|
+
max_retries=5, timeout=120.0)
|
|
99
|
+
extra_body = build_extra_body(thinking, reasoning_effort)
|
|
100
|
+
|
|
101
|
+
n_ok = n_fail = 0
|
|
102
|
+
totals = {
|
|
103
|
+
"prompt_tokens": 0,
|
|
104
|
+
"completion_tokens": 0,
|
|
105
|
+
"prompt_cache_hit_tokens": 0,
|
|
106
|
+
"prompt_cache_miss_tokens": 0,
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
with out_path.open("w") as f, Progress(
|
|
110
|
+
SpinnerColumn(style=PURPLE),
|
|
111
|
+
TextColumn("[progress.description]{task.description}"),
|
|
112
|
+
BarColumn(complete_style=PURPLE, finished_style=VIOLET),
|
|
113
|
+
TextColumn("{task.completed}/{task.total}"),
|
|
114
|
+
TimeElapsedColumn(),
|
|
115
|
+
console=console,
|
|
116
|
+
) as progress:
|
|
117
|
+
bar = progress.add_task("predicting", total=len(ds))
|
|
118
|
+
for instance in ds:
|
|
119
|
+
iid = instance["instance_id"]
|
|
120
|
+
try:
|
|
121
|
+
resp = client.chat.completions.create(
|
|
122
|
+
model=model,
|
|
123
|
+
messages=[{"role": "user", "content": _build_prompt(instance)}],
|
|
124
|
+
temperature=0.0,
|
|
125
|
+
max_tokens=max_tokens,
|
|
126
|
+
extra_body=extra_body,
|
|
127
|
+
)
|
|
128
|
+
content = resp.choices[0].message.content or ""
|
|
129
|
+
patch = _extract_diff(content).strip()
|
|
130
|
+
if patch and not patch.endswith("\n"):
|
|
131
|
+
patch += "\n"
|
|
132
|
+
|
|
133
|
+
usage = _extract_usage(resp)
|
|
134
|
+
for k in totals:
|
|
135
|
+
totals[k] += usage.get(k, 0) or 0
|
|
136
|
+
|
|
137
|
+
n_ok += 1
|
|
138
|
+
except Exception as e: # noqa: BLE001 — log + record empty patch, keep going
|
|
139
|
+
fail(console, f"{iid}: {e}")
|
|
140
|
+
patch = ""
|
|
141
|
+
n_fail += 1
|
|
142
|
+
|
|
143
|
+
f.write(json.dumps({
|
|
144
|
+
"instance_id": iid,
|
|
145
|
+
"model_name_or_path": f"raw-{safe_model}",
|
|
146
|
+
"model_patch": patch,
|
|
147
|
+
}) + "\n")
|
|
148
|
+
f.flush()
|
|
149
|
+
progress.update(bar, advance=1)
|
|
150
|
+
|
|
151
|
+
if n_fail == 0:
|
|
152
|
+
amaze(console, f"all {n_ok} predictions written to {out_path}!")
|
|
153
|
+
else:
|
|
154
|
+
confused(console, f"{n_ok} ok, {n_fail} failed. predictions at {out_path}")
|
|
155
|
+
|
|
156
|
+
_print_usage_summary(console, totals)
|
|
157
|
+
return out_path
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _print_usage_summary(console: Console, totals: dict) -> None:
|
|
161
|
+
prompt = totals["prompt_tokens"]
|
|
162
|
+
completion = totals["completion_tokens"]
|
|
163
|
+
cache_hit = totals["prompt_cache_hit_tokens"]
|
|
164
|
+
cache_miss = totals["prompt_cache_miss_tokens"]
|
|
165
|
+
cache_seen = cache_hit + cache_miss
|
|
166
|
+
hit_rate = (cache_hit / cache_seen * 100) if cache_seen else 0.0
|
|
167
|
+
|
|
168
|
+
table = Table(title="token usage", show_header=True, header_style=f"bold {VIOLET}")
|
|
169
|
+
table.add_column("metric")
|
|
170
|
+
table.add_column("value", justify="right")
|
|
171
|
+
table.add_row("prompt tokens", f"{prompt:,}")
|
|
172
|
+
table.add_row("completion tokens", f"{completion:,}")
|
|
173
|
+
table.add_row("cache hit tokens", f"{cache_hit:,}")
|
|
174
|
+
table.add_row("cache miss tokens", f"{cache_miss:,}")
|
|
175
|
+
table.add_row("cache hit rate", f"{hit_rate:.1f}%")
|
|
176
|
+
console.print(table)
|
rockycode/score.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""Wraps SWE-bench's official evaluation harness as a subprocess.
|
|
2
|
+
|
|
3
|
+
We call `python -m swebench.harness.run_evaluation` rather than importing
|
|
4
|
+
internals — the harness's public CLI is its stable contract; its Python API
|
|
5
|
+
moves between releases.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import subprocess
|
|
11
|
+
import sys
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Optional
|
|
14
|
+
|
|
15
|
+
from rich.console import Console
|
|
16
|
+
from rich.table import Table
|
|
17
|
+
|
|
18
|
+
from rockycode.banner import amaze, fail, info
|
|
19
|
+
from rockycode.palette import VIOLET
|
|
20
|
+
|
|
21
|
+
DATASET = "princeton-nlp/SWE-bench_Verified"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def score(
|
|
25
|
+
predictions_path: Path,
|
|
26
|
+
run_id: str,
|
|
27
|
+
instance_ids: Optional[list[str]],
|
|
28
|
+
console: Console,
|
|
29
|
+
max_workers: int = 4,
|
|
30
|
+
timeout: int = 1800,
|
|
31
|
+
) -> dict:
|
|
32
|
+
"""Run swebench eval and parse the report. Returns the parsed report dict."""
|
|
33
|
+
info(console, f"scoring run_id={run_id}")
|
|
34
|
+
info(console, f"predictions={predictions_path}")
|
|
35
|
+
|
|
36
|
+
cmd = [
|
|
37
|
+
sys.executable, "-m", "swebench.harness.run_evaluation",
|
|
38
|
+
"--dataset_name", DATASET,
|
|
39
|
+
"--predictions_path", str(predictions_path),
|
|
40
|
+
"--max_workers", str(max_workers),
|
|
41
|
+
"--run_id", run_id,
|
|
42
|
+
"--timeout", str(timeout),
|
|
43
|
+
]
|
|
44
|
+
if instance_ids:
|
|
45
|
+
cmd += ["--instance_ids", *instance_ids]
|
|
46
|
+
|
|
47
|
+
info(console, f"$ {' '.join(cmd)}")
|
|
48
|
+
proc = subprocess.run(cmd, check=False)
|
|
49
|
+
if proc.returncode != 0:
|
|
50
|
+
fail(console, f"swebench harness exited {proc.returncode}")
|
|
51
|
+
raise SystemExit(proc.returncode)
|
|
52
|
+
|
|
53
|
+
report = _find_report(run_id)
|
|
54
|
+
if report is None:
|
|
55
|
+
fail(console, "no report file found after eval. check logs/run_evaluation/")
|
|
56
|
+
raise SystemExit(1)
|
|
57
|
+
|
|
58
|
+
_print_report(console, report, run_id)
|
|
59
|
+
return report
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _find_report(run_id: str) -> Optional[dict]:
|
|
63
|
+
"""The harness writes its summary as <model_name>.<run_id>.json in cwd
|
|
64
|
+
(older versions) or under logs/run_evaluation/<run_id>/ (newer). Match run_id
|
|
65
|
+
EXACTLY (delimited) and take the NEWEST match: a loose `*run_id*` glob matched
|
|
66
|
+
'v1' inside a 'v12' report, and with no freshness order a re-run of the same
|
|
67
|
+
run_id could pick up a stale report."""
|
|
68
|
+
candidates: list[Path] = []
|
|
69
|
+
candidates += list(Path(".").glob(f"*.{run_id}.json"))
|
|
70
|
+
candidates += list(Path("logs/run_evaluation").glob(f"{run_id}/**/results.json"))
|
|
71
|
+
candidates += list(Path("logs/run_evaluation").glob(f"{run_id}/**/*.json"))
|
|
72
|
+
candidates.sort(key=lambda p: p.stat().st_mtime if p.exists() else 0.0, reverse=True)
|
|
73
|
+
for path in candidates:
|
|
74
|
+
try:
|
|
75
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
76
|
+
if isinstance(data, dict) and ("resolved_ids" in data or "resolved_instances" in data):
|
|
77
|
+
return data
|
|
78
|
+
except (json.JSONDecodeError, OSError):
|
|
79
|
+
continue
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _count(report: dict, ids_key: str, count_key: str) -> int:
|
|
84
|
+
v = report.get(ids_key)
|
|
85
|
+
if isinstance(v, list):
|
|
86
|
+
return len(v)
|
|
87
|
+
c = report.get(count_key)
|
|
88
|
+
return int(c) if isinstance(c, (int, float)) else 0
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _print_report(console: Console, report: dict, run_id: str) -> None:
|
|
92
|
+
resolved = _count(report, "resolved_ids", "resolved_instances")
|
|
93
|
+
submitted = _count(report, "submitted_ids", "submitted_instances")
|
|
94
|
+
error = _count(report, "error_ids", "error_instances")
|
|
95
|
+
total = report.get("total_instances") or 0
|
|
96
|
+
# Score over what we actually submitted. swebench's total_instances is the
|
|
97
|
+
# whole Verified set (500) — dividing by it made a 1-task run read as 0.2%.
|
|
98
|
+
pct = (resolved / submitted * 100) if submitted else 0.0
|
|
99
|
+
|
|
100
|
+
info(console, f"run: {run_id}")
|
|
101
|
+
table = Table(show_header=True, header_style=f"bold {VIOLET}")
|
|
102
|
+
table.add_column("metric")
|
|
103
|
+
table.add_column("value", justify="right")
|
|
104
|
+
table.add_row("submitted", str(submitted))
|
|
105
|
+
table.add_row("resolved", str(resolved))
|
|
106
|
+
table.add_row("errors", str(error))
|
|
107
|
+
table.add_row("score (resolved / submitted)", f"[bold]{pct:.1f}%[/bold]")
|
|
108
|
+
table.add_row("[dim]verified set size[/dim]", f"[dim]{total}[/dim]")
|
|
109
|
+
console.print(table)
|
|
110
|
+
|
|
111
|
+
if pct > 0:
|
|
112
|
+
amaze(console, f"score {pct:.1f}%! amaze!")
|
|
113
|
+
else:
|
|
114
|
+
console.print("[dim]we no resolve any. we try harness next, learn more.[/dim]")
|
rockycode/session.py
ADDED
|
@@ -0,0 +1,298 @@
|
|
|
1
|
+
"""Session storage: stable project identity + cross-folder discovery.
|
|
2
|
+
|
|
3
|
+
Claude Code / Codex key sessions by a folder's absolute path in a global
|
|
4
|
+
store, so renaming or moving the folder ORPHANS its sessions. rockycode does
|
|
5
|
+
it differently:
|
|
6
|
+
|
|
7
|
+
- Each project gets a STABLE id in `.rockycode/project.json`. The file lives
|
|
8
|
+
in the folder, so it travels on rename/move — the project keeps its identity
|
|
9
|
+
no matter where it is or what it's called.
|
|
10
|
+
- Trajectories live in ONE global store (`~/.rockycode/trajectories`); each
|
|
11
|
+
file's meta records which project it belongs to, so sessions follow the
|
|
12
|
+
project identity, not the folder path.
|
|
13
|
+
- A tiny global registry (`~/.rockycode/projects.json`) records where each
|
|
14
|
+
project currently lives, so sessions are discoverable ACROSS folders — the
|
|
15
|
+
resume picker can list "this folder" or "all folders", and `--resume <id>`
|
|
16
|
+
can land back in a project's CURRENT folder even after a rename.
|
|
17
|
+
|
|
18
|
+
So: sessions survive folder renames AND are searchable across the machine —
|
|
19
|
+
which neither Claude Code nor Codex manage.
|
|
20
|
+
|
|
21
|
+
Public session ids are `rk_<hash>` — the uuid tail of the trajectory stem
|
|
22
|
+
(filenames keep their sortable timestamp form on disk; humans get the short
|
|
23
|
+
hash, like opencode's `ses_…`).
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import os
|
|
29
|
+
import time
|
|
30
|
+
import uuid
|
|
31
|
+
from dataclasses import dataclass
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Optional
|
|
34
|
+
|
|
35
|
+
_HOME_ENV = os.environ.get("ROCKYCODE_HOME")
|
|
36
|
+
HOME_ROOT = Path(_HOME_ENV).expanduser() if _HOME_ENV else Path.home() / ".rockycode"
|
|
37
|
+
REGISTRY = HOME_ROOT / "projects.json"
|
|
38
|
+
PROJECT_REL = Path(".rockycode") / "project.json"
|
|
39
|
+
TRAJ_REL = Path(".rockycode") / "trajectories"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Project:
|
|
44
|
+
id: str
|
|
45
|
+
name: str
|
|
46
|
+
root: Path
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class SessionInfo:
|
|
51
|
+
session_id: str
|
|
52
|
+
path: Path
|
|
53
|
+
project_id: str
|
|
54
|
+
project_name: str
|
|
55
|
+
project_path: str
|
|
56
|
+
model: str
|
|
57
|
+
started_at: float
|
|
58
|
+
n_messages: int
|
|
59
|
+
summary: str
|
|
60
|
+
title: str = "" # flash-generated; empty on old/offline sessions
|
|
61
|
+
|
|
62
|
+
@property
|
|
63
|
+
def display_title(self) -> str:
|
|
64
|
+
return self.title or self.summary
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# ---- project identity + registry -------------------------------------------
|
|
68
|
+
|
|
69
|
+
def get_project(workdir: Path) -> Project:
|
|
70
|
+
"""Stable identity for a folder. Creates `.rockycode/project.json` on first
|
|
71
|
+
use; that file travels with the folder, so the id survives renames."""
|
|
72
|
+
workdir = Path(workdir).resolve()
|
|
73
|
+
pf = workdir / PROJECT_REL
|
|
74
|
+
if pf.exists():
|
|
75
|
+
try:
|
|
76
|
+
data = json.loads(pf.read_text(encoding="utf-8"))
|
|
77
|
+
proj = Project(id=data["project_id"], name=data.get("name", workdir.name), root=workdir)
|
|
78
|
+
except (json.JSONDecodeError, OSError, KeyError):
|
|
79
|
+
proj = _new_project(workdir, pf)
|
|
80
|
+
else:
|
|
81
|
+
proj = _new_project(workdir, pf)
|
|
82
|
+
_register(proj)
|
|
83
|
+
return proj
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _new_project(workdir: Path, pf: Path) -> Project:
|
|
87
|
+
proj = Project(id=uuid.uuid4().hex, name=workdir.name, root=workdir)
|
|
88
|
+
try:
|
|
89
|
+
pf.parent.mkdir(parents=True, exist_ok=True)
|
|
90
|
+
pf.write_text(
|
|
91
|
+
json.dumps({"project_id": proj.id, "name": proj.name, "created": time.time()}, indent=2),
|
|
92
|
+
encoding="utf-8",
|
|
93
|
+
)
|
|
94
|
+
except OSError:
|
|
95
|
+
# Read-only / unwritable cwd: use an ephemeral id for this run instead of
|
|
96
|
+
# crashing chat at startup. It just won't persist across launches (so
|
|
97
|
+
# resume-by-project won't find it), which is the right degradation.
|
|
98
|
+
pass
|
|
99
|
+
return proj
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _load_registry() -> dict:
|
|
103
|
+
if REGISTRY.exists():
|
|
104
|
+
try:
|
|
105
|
+
return json.loads(REGISTRY.read_text(encoding="utf-8"))
|
|
106
|
+
except (json.JSONDecodeError, OSError):
|
|
107
|
+
return {}
|
|
108
|
+
return {}
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _register(proj: Project) -> None:
|
|
112
|
+
"""Record (and self-heal) where this project lives. `current` is the
|
|
113
|
+
authoritative path; old paths linger but are harmless. Best-effort: the
|
|
114
|
+
registry only powers cross-folder resume, so an unwritable home must never
|
|
115
|
+
block startup."""
|
|
116
|
+
try:
|
|
117
|
+
HOME_ROOT.mkdir(parents=True, exist_ok=True)
|
|
118
|
+
reg = _load_registry()
|
|
119
|
+
entry = reg.get(proj.id, {})
|
|
120
|
+
paths = set(entry.get("paths", []))
|
|
121
|
+
paths.add(str(proj.root))
|
|
122
|
+
reg[proj.id] = {
|
|
123
|
+
"name": proj.name,
|
|
124
|
+
"current": str(proj.root),
|
|
125
|
+
"paths": sorted(paths),
|
|
126
|
+
"last_seen": time.time(),
|
|
127
|
+
}
|
|
128
|
+
REGISTRY.write_text(json.dumps(reg, indent=2), encoding="utf-8")
|
|
129
|
+
except OSError:
|
|
130
|
+
pass
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
# ---- session discovery -----------------------------------------------------
|
|
134
|
+
|
|
135
|
+
def _is_chat_session(meta: dict) -> bool:
|
|
136
|
+
# bench, goal, and routine runs land in the same trajectories dir but
|
|
137
|
+
# carry a runner / instance_id; the resume picker only wants interactive
|
|
138
|
+
# chat sessions. (Goal/routine runs DO carry project_id now — that's for
|
|
139
|
+
# the dream, which grades them alongside chats, not for the picker.)
|
|
140
|
+
return "instance_id" not in meta and meta.get("runner") not in ("rockycode", "goal", "routine")
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def global_traj_dir() -> Path:
|
|
144
|
+
"""The single global trajectory store. Read at call time so a test that
|
|
145
|
+
redirects HOME_ROOT takes effect."""
|
|
146
|
+
return HOME_ROOT / "trajectories"
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _read_info(traj: Path) -> Optional[SessionInfo]:
|
|
150
|
+
"""Build a SessionInfo from a trajectory; project identity comes from its
|
|
151
|
+
own meta (project_id/project_name/workdir), since all sessions share the
|
|
152
|
+
one global dir now."""
|
|
153
|
+
try:
|
|
154
|
+
lines = traj.read_text(encoding="utf-8", errors="replace").splitlines()
|
|
155
|
+
except OSError:
|
|
156
|
+
return None
|
|
157
|
+
if not lines:
|
|
158
|
+
return None
|
|
159
|
+
meta: dict = {}
|
|
160
|
+
summary = ""
|
|
161
|
+
title = ""
|
|
162
|
+
n_messages = 0
|
|
163
|
+
started_at = traj.stat().st_mtime
|
|
164
|
+
for ln in lines:
|
|
165
|
+
try:
|
|
166
|
+
rec = json.loads(ln)
|
|
167
|
+
except json.JSONDecodeError:
|
|
168
|
+
continue
|
|
169
|
+
kind, data = rec.get("kind"), rec.get("data", {})
|
|
170
|
+
if kind == "meta":
|
|
171
|
+
meta = data
|
|
172
|
+
started_at = rec.get("t", started_at)
|
|
173
|
+
elif kind == "title":
|
|
174
|
+
# last one wins — the trajectory is append-only, so regenerating a
|
|
175
|
+
# title later just appends a fresher record
|
|
176
|
+
t = data.get("title")
|
|
177
|
+
if isinstance(t, str) and t.strip():
|
|
178
|
+
title = t.strip()
|
|
179
|
+
elif kind == "message":
|
|
180
|
+
n_messages += 1
|
|
181
|
+
if not summary and data.get("role") == "user":
|
|
182
|
+
c = data.get("content")
|
|
183
|
+
if isinstance(c, str) and c.strip():
|
|
184
|
+
summary = c.strip().splitlines()[0][:80]
|
|
185
|
+
if not _is_chat_session(meta):
|
|
186
|
+
return None
|
|
187
|
+
workdir = meta.get("workdir", "")
|
|
188
|
+
return SessionInfo(
|
|
189
|
+
session_id=traj.stem,
|
|
190
|
+
path=traj,
|
|
191
|
+
project_id=meta.get("project_id", ""),
|
|
192
|
+
project_name=meta.get("project_name") or (Path(workdir).name if workdir else "?"),
|
|
193
|
+
project_path=workdir,
|
|
194
|
+
model=meta.get("model", "?"),
|
|
195
|
+
started_at=started_at,
|
|
196
|
+
n_messages=n_messages,
|
|
197
|
+
summary=summary or "(no message)",
|
|
198
|
+
title=title,
|
|
199
|
+
)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def list_sessions(
|
|
203
|
+
scope: str = "project",
|
|
204
|
+
*,
|
|
205
|
+
workdir: Optional[Path] = None,
|
|
206
|
+
query: Optional[str] = None,
|
|
207
|
+
limit: int = 50,
|
|
208
|
+
) -> list[SessionInfo]:
|
|
209
|
+
"""Chat sessions from the global store, newest first. scope='project'
|
|
210
|
+
keeps only this workdir's project (by project_id); scope='all' keeps every
|
|
211
|
+
project. query filters by summary/project-name substring."""
|
|
212
|
+
traj_dir = global_traj_dir()
|
|
213
|
+
if not traj_dir.is_dir():
|
|
214
|
+
return []
|
|
215
|
+
target_pid: Optional[str] = None
|
|
216
|
+
if scope != "all" and workdir is not None:
|
|
217
|
+
target_pid = get_project(workdir).id
|
|
218
|
+
infos: list[SessionInfo] = []
|
|
219
|
+
for f in traj_dir.glob("*.jsonl"):
|
|
220
|
+
info = _read_info(f)
|
|
221
|
+
if info is None:
|
|
222
|
+
continue
|
|
223
|
+
if target_pid is not None and info.project_id != target_pid:
|
|
224
|
+
continue
|
|
225
|
+
infos.append(info)
|
|
226
|
+
if query:
|
|
227
|
+
q = query.lower()
|
|
228
|
+
infos = [i for i in infos
|
|
229
|
+
if q in i.title.lower() or q in i.summary.lower() or q in i.project_name.lower()]
|
|
230
|
+
infos.sort(key=lambda i: i.started_at, reverse=True)
|
|
231
|
+
return infos[:limit]
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
# ---- public session ids ------------------------------------------------------
|
|
235
|
+
|
|
236
|
+
ID_PREFIX = "rk_"
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def public_id(session_id: str) -> str:
|
|
240
|
+
"""`rk_ab12cd34` — the uuid tail of the trajectory stem, rocky-prefixed.
|
|
241
|
+
The filename keeps its sortable stamp form on disk; humans get the hash."""
|
|
242
|
+
return ID_PREFIX + session_id.rsplit("-", 1)[-1]
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def resolve_session(token: str) -> tuple[Optional[SessionInfo], str]:
|
|
246
|
+
"""One session from a user-typed id. Accepted forms: `rk_ab12cd34`, bare
|
|
247
|
+
`ab12cd34`, a unique hash prefix (≥4 chars), or a full legacy stem.
|
|
248
|
+
Returns (info, error) — exactly one is set."""
|
|
249
|
+
t = token.strip().lower()
|
|
250
|
+
if t.startswith(ID_PREFIX):
|
|
251
|
+
t = t[len(ID_PREFIX):]
|
|
252
|
+
if not t:
|
|
253
|
+
return None, "empty session id"
|
|
254
|
+
sessions = list_sessions(scope="all", limit=100_000)
|
|
255
|
+
exact = [s for s in sessions if s.session_id.lower() == t]
|
|
256
|
+
if exact:
|
|
257
|
+
return exact[0], ""
|
|
258
|
+
if len(t) >= 4:
|
|
259
|
+
hits = [s for s in sessions
|
|
260
|
+
if s.session_id.rsplit("-", 1)[-1].lower().startswith(t)]
|
|
261
|
+
if len(hits) == 1:
|
|
262
|
+
return hits[0], ""
|
|
263
|
+
if len(hits) > 1:
|
|
264
|
+
opts = ", ".join(f"{public_id(s.session_id)} ({s.display_title[:30]})" for s in hits[:5])
|
|
265
|
+
return None, f"'{token}' is ambiguous — matches {opts}"
|
|
266
|
+
return None, f"no session matches '{token}' — run `rockycode --resume` to browse"
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def project_current_path(project_id: str) -> Optional[Path]:
|
|
270
|
+
"""Where a project lives NOW, per the registry — survives folder renames.
|
|
271
|
+
None when the registry has no entry (or the recorded path is gone)."""
|
|
272
|
+
entry = _load_registry().get(project_id) or {}
|
|
273
|
+
cur = entry.get("current")
|
|
274
|
+
if cur and Path(cur).is_dir():
|
|
275
|
+
return Path(cur)
|
|
276
|
+
for p in reversed(entry.get("paths", [])):
|
|
277
|
+
if Path(p).is_dir():
|
|
278
|
+
return Path(p)
|
|
279
|
+
return None
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def load_history(traj: Path) -> list[dict]:
|
|
283
|
+
"""Reconstruct an engine message history from a trajectory file — exactly
|
|
284
|
+
the {role, content, tool_calls, tool_call_id} dicts that were sent to the
|
|
285
|
+
API (reasoning_content was never stored, so this is clean to replay)."""
|
|
286
|
+
history: list[dict] = []
|
|
287
|
+
try:
|
|
288
|
+
lines = Path(traj).read_text(encoding="utf-8", errors="replace").splitlines()
|
|
289
|
+
except OSError:
|
|
290
|
+
return history
|
|
291
|
+
for ln in lines:
|
|
292
|
+
try:
|
|
293
|
+
rec = json.loads(ln)
|
|
294
|
+
except json.JSONDecodeError:
|
|
295
|
+
continue
|
|
296
|
+
if rec.get("kind") == "message":
|
|
297
|
+
history.append(rec["data"])
|
|
298
|
+
return history
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: architecture-viz
|
|
3
|
+
description: Visualize a neural-net block (attention, transformer, MoE, diffusion-LLM) as an artifact three linked ways — the paper diagram, the tensor-shape ribbon, and the user's PyTorch — cross-highlighted so hovering one lights the other two. Grounded in the real paper/code, never drawn from memory.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# architecture-viz — one block, three linked views
|
|
7
|
+
|
|
8
|
+
The register (LOCKED — the user chose it after rejecting two others):
|
|
9
|
+
**a recognizable paper diagram + a shape ribbon + the user's PyTorch code, all
|
|
10
|
+
cross-highlighted.** Hover any box, shape-chip, or code line and the matching
|
|
11
|
+
two light up. That link is the whole point — it's what turns an abstract figure
|
|
12
|
+
into something the user can tie to the code they actually write.
|
|
13
|
+
|
|
14
|
+
## Why exactly these three (do not drop any)
|
|
15
|
+
|
|
16
|
+
- **Paper diagram** — the figure people recognize at a glance (e.g. Attention
|
|
17
|
+
Is All You Need's scaled-dot-product / multi-head). The anchor.
|
|
18
|
+
- **Shape ribbon** — how the tensor shape changes step by step
|
|
19
|
+
(`(n,d) → (n,dₖ)×3 → (n,n) → (n,n) → (n,dₖ)`). This is the real value.
|
|
20
|
+
- **PyTorch** — the language the user writes. NEVER show a formal language
|
|
21
|
+
(Lean/TorchLean syntax) as the teaching vehicle: a language they never
|
|
22
|
+
learned teaches nothing and frustrates. Proving stays a plain-English
|
|
23
|
+
footnote only: "want a shape proven for every n? `/research prove`".
|
|
24
|
+
|
|
25
|
+
The three share a `data-step` id per stage; that shared id IS the cross-link.
|
|
26
|
+
|
|
27
|
+
## Ground it — never draw from memory
|
|
28
|
+
|
|
29
|
+
The diagram, shapes, and code must be DERIVED, not remembered — a subtly wrong
|
|
30
|
+
architecture is worse than none (the user acts on it). Risk scales with how
|
|
31
|
+
novel the block is:
|
|
32
|
+
|
|
33
|
+
- **Standard transformer / attention** — well-known; still label real-vs-example.
|
|
34
|
+
- **DeepSeek V3.2 MoE, diffusion-LLM, anything recent** — HIGH risk from memory.
|
|
35
|
+
First get ground truth: `read_file` the user's model code if they have it, or
|
|
36
|
+
`web_fetch` the paper / model card, and build the diagram + shapes + torch
|
|
37
|
+
from THAT. Diffusion-LLM is not autoregressive — its "forward" is a denoising
|
|
38
|
+
loop; don't force it into a transformer figure.
|
|
39
|
+
- Put a provenance line at the bottom: `◆ derived` (what came from code/paper)
|
|
40
|
+
vs `◇ illustrative` (example sizes). Be honest about which.
|
|
41
|
+
|
|
42
|
+
## Build it — rocky artifact rules (important)
|
|
43
|
+
|
|
44
|
+
`create_artifact` takes BODY content only and STRIPS every `<style>` block, then
|
|
45
|
+
applies rocky's light theme. So:
|
|
46
|
+
|
|
47
|
+
- **No `<style>` block** — it will be deleted. Style with **inline `style=`**
|
|
48
|
+
attributes only (those survive), using rocky's light palette:
|
|
49
|
+
bg is themed for you; use `#efeafa` panels, `#9d7cd8`/`#7c5cba` strokes/accent,
|
|
50
|
+
`#6a4ca3` headings, `#2a2a38` text, `#ede6fb` for the lit/highlight state.
|
|
51
|
+
- The cross-highlight is done in a **`<script>`** (scripts survive): on
|
|
52
|
+
`mouseenter`/`focus` of any `[data-step]`, set inline styles on every element
|
|
53
|
+
with the same `data-step`; revert on `mouseleave`/`blur`. No CSS `:hover`.
|
|
54
|
+
- Self-contained: no CDN, no external fonts/images. Inline SVG for the diagram.
|
|
55
|
+
- Use rocky's classes where they fit: `card`, `tag`/`tag-purple`/`tag-amber`.
|
|
56
|
+
- Reuse the SAME artifact title to update in place on a re-run.
|
|
57
|
+
|
|
58
|
+
## The scaffold
|
|
59
|
+
|
|
60
|
+
`template.html` in this skill's directory is a complete, working example (scaled
|
|
61
|
+
dot-product attention) in exactly this shape. **Copy it and adapt** the three
|
|
62
|
+
panels for the block at hand — swap the SVG figure, the ribbon chips, and the
|
|
63
|
+
torch lines, keeping the `data-step` ids matched across all three. Don't
|
|
64
|
+
re-derive the highlight script; reuse it.
|
|
65
|
+
|
|
66
|
+
## Multi-panel architectures
|
|
67
|
+
|
|
68
|
+
For a whole model (MoE: router → experts → combine; a transformer layer:
|
|
69
|
+
attention + FFN + residual/norm; diffusion: the denoising steps), give each
|
|
70
|
+
sub-block its own diagram+ribbon+torch row, each internally linked, stacked top
|
|
71
|
+
to bottom — one story, every stage labeled with the shape it carries.
|