rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/dream/core.py
ADDED
|
@@ -0,0 +1,523 @@
|
|
|
1
|
+
"""The dream pass (M2): offline memory consolidation on a local Ollama model.
|
|
2
|
+
|
|
3
|
+
Design: docs/memory-dream.md §3. Jobs in this milestone:
|
|
4
|
+
|
|
5
|
+
1. episode digestion — each un-dreamed trajectory becomes an episode note
|
|
6
|
+
(task / outcome / what worked / what failed), failures recorded while
|
|
7
|
+
fresh; durable facts extracted as candidates
|
|
8
|
+
2. project digest — the dream owns ONE marked section of MEMORY.md
|
|
9
|
+
(between dream:state markers); the rest stays hand-curated
|
|
10
|
+
3. reconciliation — each candidate fact vs its nearest existing memories:
|
|
11
|
+
NOOP / ADD / UPDATE / ARCHIVE (archive never deletes — M0 rule)
|
|
12
|
+
6. re-embed — sync index.db when something changed
|
|
13
|
+
|
|
14
|
+
Idle trigger, decay, and skill promotion are M3. Everything auto-applies
|
|
15
|
+
(user decision 2026-06-12) — `--dry-run` previews instead.
|
|
16
|
+
|
|
17
|
+
The model runs through Ollama's NATIVE /api/chat with `think: false`: the
|
|
18
|
+
OpenAI-compat endpoint ignores every thinking switch for qwen3.5 (measured
|
|
19
|
+
2026-06-13: 2 completion tokens vs ~100+ for a bare "ok").
|
|
20
|
+
|
|
21
|
+
"Already dreamed" is derived from the files, not a state db: a session id
|
|
22
|
+
listed in any episode's `evidence:` has been digested. Delete the episode
|
|
23
|
+
file and the session becomes dreamable again — files stay the truth.
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import os
|
|
29
|
+
import re
|
|
30
|
+
import time
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
from typing import Callable, Optional
|
|
34
|
+
|
|
35
|
+
import httpx
|
|
36
|
+
|
|
37
|
+
from rockycode.memory.store import Memory, MemoryStore, _slugify
|
|
38
|
+
|
|
39
|
+
OLLAMA_URL = os.getenv("ROCKYCODE_OLLAMA_URL", "http://localhost:11434")
|
|
40
|
+
# 2b on purpose (bake-off 2026-07-17, n=5): every qwen3.5 size follows the
|
|
41
|
+
# JSON contracts fine (think:false verified working on native /api/chat,
|
|
42
|
+
# ollama 0.24.0) — the sizes differ in JUDGMENT, not format. 4b/9b decline
|
|
43
|
+
# ("[]"/null) on evidence 2b happily mines. The pipeline wants the eager
|
|
44
|
+
# miner: precision comes later (hot needs reinforcement, installs need a
|
|
45
|
+
# human), while a declined pattern is lost forever. Env-overridable.
|
|
46
|
+
DREAM_MODEL = os.getenv("ROCKYCODE_DREAM_MODEL", "qwen3.5:2b")
|
|
47
|
+
|
|
48
|
+
TRAJECTORY_DIRNAME = Path(".rockycode") / "trajectories"
|
|
49
|
+
LOCK_STALE_S = 3600
|
|
50
|
+
MIN_SESSION_MESSAGES = 4 # meta+system+user only → nothing to learn
|
|
51
|
+
MAX_TRANSCRIPT_CHARS = 6_000
|
|
52
|
+
MAX_STATE_LINES = 25
|
|
53
|
+
|
|
54
|
+
DREAM_MARK_START = "<!-- dream:state -->"
|
|
55
|
+
DREAM_MARK_END = "<!-- /dream:state -->"
|
|
56
|
+
|
|
57
|
+
DIGEST_PROMPT = """\
|
|
58
|
+
You are consolidating a coding agent's work session into a memory note.
|
|
59
|
+
|
|
60
|
+
SESSION TRANSCRIPT (condensed):
|
|
61
|
+
{transcript}
|
|
62
|
+
|
|
63
|
+
Write exactly these markdown sections, nothing outside them:
|
|
64
|
+
## task
|
|
65
|
+
One or two sentences: what the session tried to accomplish.
|
|
66
|
+
## outcome
|
|
67
|
+
success / partial / failed — plus one sentence of evidence.
|
|
68
|
+
## worked
|
|
69
|
+
Bullets: approaches or commands that worked. Write "- none" if none.
|
|
70
|
+
## failed
|
|
71
|
+
Bullets: approaches that failed or wasted time, so they are not retried. "- none" if none.
|
|
72
|
+
## facts
|
|
73
|
+
Bullets: durable project facts worth remembering across sessions (paths,
|
|
74
|
+
commands, configuration gotchas). Each bullet ONE standalone line. "- none" if none.
|
|
75
|
+
## importance
|
|
76
|
+
One integer 1-10: how much future sessions benefit from this note.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
RECONCILE_PROMPT = """\
|
|
80
|
+
A coding agent wants to save a new memory. Decide how it relates to the
|
|
81
|
+
existing memories below.
|
|
82
|
+
|
|
83
|
+
NEW FACT: {fact}
|
|
84
|
+
|
|
85
|
+
EXISTING MEMORIES:
|
|
86
|
+
{existing}
|
|
87
|
+
|
|
88
|
+
Reply with EXACTLY one decision on the first line:
|
|
89
|
+
NOOP — already covered by an existing memory
|
|
90
|
+
ADD — genuinely new information, save it alongside
|
|
91
|
+
UPDATE <name> — improves/extends that memory. Then a line with only ---
|
|
92
|
+
followed by the full merged memory text.
|
|
93
|
+
ARCHIVE <name> — that memory is now wrong or obsolete; the new fact replaces it
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
STATE_PROMPT = """\
|
|
97
|
+
You maintain the "current state" section of a coding project's memory file.
|
|
98
|
+
|
|
99
|
+
CURRENT STATE SECTION (may be empty):
|
|
100
|
+
{current}
|
|
101
|
+
|
|
102
|
+
NEW EPISODE NOTES SINCE LAST UPDATE:
|
|
103
|
+
{episodes}
|
|
104
|
+
|
|
105
|
+
Rewrite the state section: what the project is, current focus, recent work,
|
|
106
|
+
known gotchas. Markdown bullets only, at most {max_lines} lines, dense,
|
|
107
|
+
no preamble, no heading.
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class OllamaChat:
|
|
112
|
+
"""Minimal native /api/chat client — think:false actually works here."""
|
|
113
|
+
|
|
114
|
+
def __init__(self, model: str = DREAM_MODEL, base_url: str = OLLAMA_URL) -> None:
|
|
115
|
+
self.model = model
|
|
116
|
+
self.base_url = base_url
|
|
117
|
+
|
|
118
|
+
async def chat(self, prompt: str, max_tokens: int = 2048) -> str:
|
|
119
|
+
async with httpx.AsyncClient(timeout=300.0) as http:
|
|
120
|
+
resp = await http.post(
|
|
121
|
+
f"{self.base_url}/api/chat",
|
|
122
|
+
json={
|
|
123
|
+
"model": self.model,
|
|
124
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
125
|
+
"think": False,
|
|
126
|
+
"stream": False,
|
|
127
|
+
"options": {"num_predict": max_tokens},
|
|
128
|
+
},
|
|
129
|
+
)
|
|
130
|
+
resp.raise_for_status()
|
|
131
|
+
return (resp.json().get("message") or {}).get("content", "").strip()
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@dataclass
|
|
135
|
+
class DreamReport:
|
|
136
|
+
sessions_digested: int = 0
|
|
137
|
+
sessions_judged: int = 0
|
|
138
|
+
sessions_skipped: int = 0
|
|
139
|
+
facts_added: int = 0
|
|
140
|
+
facts_updated: int = 0
|
|
141
|
+
facts_archived: int = 0
|
|
142
|
+
facts_noop: int = 0
|
|
143
|
+
weaknesses_added: int = 0
|
|
144
|
+
weaknesses_reinforced: int = 0
|
|
145
|
+
proposals_drafted: int = 0
|
|
146
|
+
state_updated: bool = False
|
|
147
|
+
reindexed: Optional[tuple[int, int, int]] = None
|
|
148
|
+
decisions: list[str] = field(default_factory=list)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
# ---- trajectory condensing ----------------------------------------------------
|
|
152
|
+
|
|
153
|
+
def load_session(path: Path) -> Optional[dict]:
|
|
154
|
+
meta, messages, outcome, heuristic, feedback = {}, [], None, None, None
|
|
155
|
+
try:
|
|
156
|
+
for line in path.read_text(encoding="utf-8", errors="replace").splitlines():
|
|
157
|
+
try:
|
|
158
|
+
rec = json.loads(line)
|
|
159
|
+
except json.JSONDecodeError:
|
|
160
|
+
continue
|
|
161
|
+
if rec.get("kind") == "meta":
|
|
162
|
+
meta = rec.get("data", {})
|
|
163
|
+
elif rec.get("kind") == "message":
|
|
164
|
+
messages.append(rec.get("data", {}))
|
|
165
|
+
elif rec.get("kind") == "outcome":
|
|
166
|
+
outcome = rec.get("data", {}) # last wins: judge > heuristic
|
|
167
|
+
if outcome.get("source") == "heuristic":
|
|
168
|
+
heuristic = outcome # kept separately — mining needs the counters
|
|
169
|
+
elif rec.get("kind") == "feedback":
|
|
170
|
+
feedback = rec.get("data", {}) # the exit sheet — LOCAL ONLY
|
|
171
|
+
except OSError:
|
|
172
|
+
return None
|
|
173
|
+
if len(messages) < MIN_SESSION_MESSAGES - 1:
|
|
174
|
+
return None
|
|
175
|
+
if not any(m.get("role") == "assistant" for m in messages):
|
|
176
|
+
return None
|
|
177
|
+
return {
|
|
178
|
+
"session_id": path.stem, "path": str(path), "meta": meta,
|
|
179
|
+
"messages": messages, "outcome": outcome, "heuristic": heuristic,
|
|
180
|
+
"feedback": feedback,
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def condense(session: dict, *, feedback: bool = False) -> str:
|
|
185
|
+
parts: list[str] = []
|
|
186
|
+
for msg in session["messages"]:
|
|
187
|
+
role, content = msg.get("role"), msg.get("content") or ""
|
|
188
|
+
if role == "user":
|
|
189
|
+
parts.append(f"[user] {content[:1200]}")
|
|
190
|
+
elif role == "assistant":
|
|
191
|
+
for tc in msg.get("tool_calls") or []:
|
|
192
|
+
fn = tc.get("function", {})
|
|
193
|
+
args = (fn.get("arguments") or "").replace("\n", " ")[:120]
|
|
194
|
+
parts.append(f"[{fn.get('name', 'tool')}] {args}")
|
|
195
|
+
if content:
|
|
196
|
+
parts.append(f"[rocky] {content[:400]}")
|
|
197
|
+
elif role == "tool":
|
|
198
|
+
first = content.strip().splitlines()[0][:160] if content.strip() else ""
|
|
199
|
+
if first.startswith(("[error]", "[timeout]", "[exit")) and not first.startswith("[exit 0]"):
|
|
200
|
+
parts.append(f" ↳ {first}")
|
|
201
|
+
if session["outcome"]:
|
|
202
|
+
parts.append(f"[outcome] {json.dumps(session['outcome'], ensure_ascii=False)[:400]}")
|
|
203
|
+
# The exit sheet is opt-in per CALLER, not per session: its trajectory
|
|
204
|
+
# record promises "never sent to the model provider", so only a condense
|
|
205
|
+
# destined for the LOCAL Ollama dream may pass feedback=True. A future
|
|
206
|
+
# cloud judge must keep the default.
|
|
207
|
+
if feedback and session.get("feedback"):
|
|
208
|
+
fb = session["feedback"]
|
|
209
|
+
note = f" — {fb.get('text', '')}" if fb.get("text") else ""
|
|
210
|
+
parts.append(f"[user exit-feedback] mood={fb.get('mood')}{note}"[:300])
|
|
211
|
+
text = "\n".join(parts)
|
|
212
|
+
if len(text) > MAX_TRANSCRIPT_CHARS:
|
|
213
|
+
half = MAX_TRANSCRIPT_CHARS // 2
|
|
214
|
+
text = f"{text[:half]}\n… [middle of session omitted] …\n{text[-half:]}"
|
|
215
|
+
return text
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def parse_sections(text: str) -> dict[str, str]:
|
|
219
|
+
sections: dict[str, str] = {}
|
|
220
|
+
current = None
|
|
221
|
+
for line in text.splitlines():
|
|
222
|
+
m = re.match(r"^##+\s*(\w+)", line.strip())
|
|
223
|
+
if m:
|
|
224
|
+
current = m.group(1).lower()
|
|
225
|
+
sections[current] = ""
|
|
226
|
+
elif current:
|
|
227
|
+
sections[current] += line + "\n"
|
|
228
|
+
return {k: v.strip() for k, v in sections.items()}
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def keyword_neighbors(store: MemoryStore, fact: str, k: int = 3) -> list[Memory]:
|
|
232
|
+
"""No-embeddings neighbor lookup: rank by content-word overlap. A whole-
|
|
233
|
+
string substring match (store.search) never hits paraphrased facts."""
|
|
234
|
+
words = {w for w in re.findall(r"\w+", fact.lower()) if len(w) > 2}
|
|
235
|
+
words |= set(re.findall(r"[-ヿ㐀-䶿一-鿿豈-]", fact))
|
|
236
|
+
scored: list[tuple[int, Memory]] = []
|
|
237
|
+
for mem in store.load_all():
|
|
238
|
+
if mem.type == "episode":
|
|
239
|
+
continue
|
|
240
|
+
text = f"{mem.name} {mem.description} {mem.body}".lower()
|
|
241
|
+
hits = sum(1 for w in words if w in text)
|
|
242
|
+
if hits >= 2:
|
|
243
|
+
scored.append((hits, mem))
|
|
244
|
+
scored.sort(key=lambda t: -t[0])
|
|
245
|
+
return [m for _, m in scored[:k]]
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def parse_bullets(text: str) -> list[str]:
|
|
249
|
+
out = []
|
|
250
|
+
for line in text.splitlines():
|
|
251
|
+
line = line.strip().lstrip("-*").strip()
|
|
252
|
+
if line and line.lower() not in ("none", "none.", "无"):
|
|
253
|
+
out.append(line)
|
|
254
|
+
return out
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
# ---- the runner ----------------------------------------------------------------
|
|
258
|
+
|
|
259
|
+
class DreamRunner:
|
|
260
|
+
def __init__(
|
|
261
|
+
self,
|
|
262
|
+
workdir: Path,
|
|
263
|
+
*,
|
|
264
|
+
model: str = DREAM_MODEL,
|
|
265
|
+
chat: Optional[OllamaChat] = None,
|
|
266
|
+
dry_run: bool = False,
|
|
267
|
+
log: Callable[[str], None] = lambda s: None,
|
|
268
|
+
exclude: Optional[set[str]] = None,
|
|
269
|
+
judge=None,
|
|
270
|
+
) -> None:
|
|
271
|
+
self.workdir = workdir
|
|
272
|
+
self.store = MemoryStore.for_workdir(workdir)
|
|
273
|
+
self.chat = chat or OllamaChat(model=model)
|
|
274
|
+
self.dry_run = dry_run
|
|
275
|
+
self.log = log
|
|
276
|
+
# Session ids to leave alone this pass — the TUI's launch catch-up
|
|
277
|
+
# excludes the LIVE session (its outcome record doesn't exist yet).
|
|
278
|
+
self.exclude = exclude or set()
|
|
279
|
+
# Optional TranscriptJudge (dream/judge.py). Runs BEFORE digestion so
|
|
280
|
+
# the episode note sees the judged outcome; skipped entirely on
|
|
281
|
+
# --dry-run (a preview must be free — the judge bills a cloud call).
|
|
282
|
+
self.judge = judge
|
|
283
|
+
self.report = DreamReport()
|
|
284
|
+
|
|
285
|
+
# -- lockfile (single writer across TUI-idle and manual runs) --------------
|
|
286
|
+
|
|
287
|
+
@property
|
|
288
|
+
def _lock(self) -> Path:
|
|
289
|
+
return self.store.root / ".dream.lock"
|
|
290
|
+
|
|
291
|
+
def _acquire_lock(self) -> bool:
|
|
292
|
+
if self._lock.exists() and time.time() - self._lock.stat().st_mtime < LOCK_STALE_S:
|
|
293
|
+
return False
|
|
294
|
+
self._lock.parent.mkdir(parents=True, exist_ok=True)
|
|
295
|
+
self._lock.write_text(str(os.getpid()), encoding="utf-8")
|
|
296
|
+
return True
|
|
297
|
+
|
|
298
|
+
def _release_lock(self) -> None:
|
|
299
|
+
try:
|
|
300
|
+
self._lock.unlink(missing_ok=True)
|
|
301
|
+
except OSError:
|
|
302
|
+
pass
|
|
303
|
+
|
|
304
|
+
# -- helpers ----------------------------------------------------------------
|
|
305
|
+
|
|
306
|
+
def _digested_ids(self) -> set[str]:
|
|
307
|
+
ids: set[str] = set()
|
|
308
|
+
for mem in self.store.load_all(include_archived=True):
|
|
309
|
+
if mem.type == "episode":
|
|
310
|
+
ids.update(mem.evidence)
|
|
311
|
+
return ids
|
|
312
|
+
|
|
313
|
+
def _pending_sessions(self, limit: int) -> list[dict]:
|
|
314
|
+
from rockycode import session as _session
|
|
315
|
+
traj_dir = _session.global_traj_dir()
|
|
316
|
+
if not traj_dir.is_dir():
|
|
317
|
+
return []
|
|
318
|
+
pid = _session.get_project(self.workdir).id
|
|
319
|
+
digested = self._digested_ids()
|
|
320
|
+
sessions = []
|
|
321
|
+
for path in sorted(traj_dir.glob("*.jsonl")):
|
|
322
|
+
if path.stem in digested or path.stem in self.exclude:
|
|
323
|
+
continue
|
|
324
|
+
session = load_session(path)
|
|
325
|
+
if session is None:
|
|
326
|
+
self.report.sessions_skipped += 1
|
|
327
|
+
continue
|
|
328
|
+
# global store → only digest THIS project's sessions (no cross-
|
|
329
|
+
# project memory contamination).
|
|
330
|
+
if session["meta"].get("project_id") != pid:
|
|
331
|
+
continue
|
|
332
|
+
sessions.append(session)
|
|
333
|
+
if len(sessions) >= limit:
|
|
334
|
+
break
|
|
335
|
+
return sessions
|
|
336
|
+
|
|
337
|
+
# -- job 1: episode digestion ------------------------------------------------
|
|
338
|
+
|
|
339
|
+
async def digest(self, session: dict) -> tuple[list[str], str, Optional[str]]:
|
|
340
|
+
"""Returns (candidate facts, note text for the state prompt, and a
|
|
341
|
+
failure note for weakness mining — None when the session shows no
|
|
342
|
+
failure signals)."""
|
|
343
|
+
from rockycode.dream.mining import failure_note
|
|
344
|
+
|
|
345
|
+
sid = session["session_id"]
|
|
346
|
+
answer = await self.chat.chat(
|
|
347
|
+
DIGEST_PROMPT.format(transcript=condense(session, feedback=True))
|
|
348
|
+
)
|
|
349
|
+
sections = parse_sections(answer)
|
|
350
|
+
task = (sections.get("task") or "(unknown task)").strip()
|
|
351
|
+
try:
|
|
352
|
+
importance = max(1, min(10, int(re.search(r"\d+", sections.get("importance", "5")).group())))
|
|
353
|
+
except (AttributeError, ValueError):
|
|
354
|
+
importance = 5
|
|
355
|
+
|
|
356
|
+
body = "\n\n".join(
|
|
357
|
+
f"## {key}\n{sections.get(key, '- none')}" for key in ("task", "outcome", "worked", "failed")
|
|
358
|
+
)
|
|
359
|
+
episode = Memory(
|
|
360
|
+
name=f"ep-{sid[:15]}",
|
|
361
|
+
type="episode",
|
|
362
|
+
description=task.splitlines()[0][:150],
|
|
363
|
+
importance=importance,
|
|
364
|
+
origin="dream",
|
|
365
|
+
evidence=[sid],
|
|
366
|
+
body=body,
|
|
367
|
+
)
|
|
368
|
+
self.report.decisions.append(f"episode ep-{sid[:15]}: {episode.description}")
|
|
369
|
+
if not self.dry_run:
|
|
370
|
+
self.store.save(episode)
|
|
371
|
+
self.report.sessions_digested += 1
|
|
372
|
+
return (
|
|
373
|
+
parse_bullets(sections.get("facts", "")),
|
|
374
|
+
f"### {episode.description}\n{body[:600]}",
|
|
375
|
+
failure_note(session, sections),
|
|
376
|
+
)
|
|
377
|
+
|
|
378
|
+
# -- job 3: reconciliation ----------------------------------------------------
|
|
379
|
+
|
|
380
|
+
async def reconcile(self, fact: str, sid: str, index=None) -> None:
|
|
381
|
+
neighbors: list[Memory] = []
|
|
382
|
+
if index is not None:
|
|
383
|
+
try:
|
|
384
|
+
neighbors = [m for m, _ in await index.search(fact, k=3)]
|
|
385
|
+
except Exception: # noqa: BLE001 — fall back to substring neighbors
|
|
386
|
+
neighbors = []
|
|
387
|
+
if not neighbors:
|
|
388
|
+
neighbors = keyword_neighbors(self.store, fact)
|
|
389
|
+
neighbors = [m for m in neighbors if m.type != "episode"]
|
|
390
|
+
|
|
391
|
+
def add() -> None:
|
|
392
|
+
self.report.facts_added += 1
|
|
393
|
+
self.report.decisions.append(f"ADD: {fact[:90]}")
|
|
394
|
+
if not self.dry_run:
|
|
395
|
+
self.store.save(Memory(
|
|
396
|
+
name=_slugify(fact), type="fact", description=fact[:150],
|
|
397
|
+
origin="dream", evidence=[sid], body=fact,
|
|
398
|
+
))
|
|
399
|
+
|
|
400
|
+
if not neighbors:
|
|
401
|
+
add()
|
|
402
|
+
return
|
|
403
|
+
|
|
404
|
+
existing = "\n".join(f"[{m.name}] ({m.type}) {m.description}: {m.body[:300]}" for m in neighbors)
|
|
405
|
+
answer = await self.chat.chat(RECONCILE_PROMPT.format(fact=fact, existing=existing), max_tokens=1024)
|
|
406
|
+
first = answer.splitlines()[0].strip() if answer else "NOOP"
|
|
407
|
+
verb = first.split()[0].upper() if first.split() else "NOOP"
|
|
408
|
+
target = first.split()[1].strip("`'\"") if len(first.split()) > 1 else ""
|
|
409
|
+
known = {m.name for m in neighbors}
|
|
410
|
+
|
|
411
|
+
if verb == "ADD":
|
|
412
|
+
add()
|
|
413
|
+
elif verb == "UPDATE" and target in known:
|
|
414
|
+
mem = self.store.get(target)
|
|
415
|
+
merged = answer.split("---", 1)[1].strip() if "---" in answer else f"{mem.body}\n\n{fact}"
|
|
416
|
+
self.report.facts_updated += 1
|
|
417
|
+
self.report.decisions.append(f"UPDATE {target}: {fact[:80]}")
|
|
418
|
+
if not self.dry_run:
|
|
419
|
+
mem.body = merged
|
|
420
|
+
if sid not in mem.evidence:
|
|
421
|
+
mem.evidence.append(sid)
|
|
422
|
+
self.store.save(mem)
|
|
423
|
+
elif verb == "ARCHIVE" and target in known:
|
|
424
|
+
self.report.facts_archived += 1
|
|
425
|
+
self.report.decisions.append(f"ARCHIVE {target} (superseded): {fact[:80]}")
|
|
426
|
+
if not self.dry_run:
|
|
427
|
+
self.store.archive(target)
|
|
428
|
+
add()
|
|
429
|
+
else:
|
|
430
|
+
self.report.facts_noop += 1
|
|
431
|
+
self.report.decisions.append(f"NOOP: {fact[:90]}")
|
|
432
|
+
|
|
433
|
+
# -- job 2: project digest ------------------------------------------------------
|
|
434
|
+
|
|
435
|
+
async def update_state(self, new_episodes: list[str]) -> None:
|
|
436
|
+
index_path = self.store.root / "MEMORY.md"
|
|
437
|
+
text = self.store.index_text()
|
|
438
|
+
m = re.search(f"{re.escape(DREAM_MARK_START)}(.*?){re.escape(DREAM_MARK_END)}", text, re.DOTALL)
|
|
439
|
+
current = m.group(1).strip() if m else ""
|
|
440
|
+
|
|
441
|
+
answer = await self.chat.chat(STATE_PROMPT.format(
|
|
442
|
+
current=current or "(empty)",
|
|
443
|
+
episodes="\n\n".join(new_episodes) or "(none)",
|
|
444
|
+
max_lines=MAX_STATE_LINES,
|
|
445
|
+
), max_tokens=1024)
|
|
446
|
+
state = "\n".join(answer.splitlines()[:MAX_STATE_LINES]).strip()
|
|
447
|
+
if not state:
|
|
448
|
+
return
|
|
449
|
+
block = f"{DREAM_MARK_START}\n## current state (dream-maintained)\n{state}\n{DREAM_MARK_END}"
|
|
450
|
+
if m:
|
|
451
|
+
text = text[: m.start()] + block + text[m.end():]
|
|
452
|
+
else:
|
|
453
|
+
text = (text.rstrip() + "\n\n" if text.strip() else "# MEMORY\n\n") + block + "\n"
|
|
454
|
+
self.report.state_updated = True
|
|
455
|
+
self.report.decisions.append("MEMORY.md state section rewritten")
|
|
456
|
+
if not self.dry_run:
|
|
457
|
+
index_path.parent.mkdir(parents=True, exist_ok=True)
|
|
458
|
+
# Atomic + utf-8: MEMORY.md is the user's hand-curated file. Write a
|
|
459
|
+
# temp then os.replace, so a crash mid-write can't truncate/corrupt
|
|
460
|
+
# it; utf-8 so CJK/emoji don't crash on a non-UTF-8 (Windows) locale.
|
|
461
|
+
tmp = index_path.with_name(index_path.name + ".tmp")
|
|
462
|
+
tmp.write_text(text, encoding="utf-8")
|
|
463
|
+
os.replace(tmp, index_path)
|
|
464
|
+
|
|
465
|
+
# -- the pass --------------------------------------------------------------------
|
|
466
|
+
|
|
467
|
+
async def run(self, limit: int = 10, index=None) -> DreamReport:
|
|
468
|
+
if not self._acquire_lock():
|
|
469
|
+
raise RuntimeError("another dream is in progress (.dream.lock is fresh)")
|
|
470
|
+
try:
|
|
471
|
+
sessions = self._pending_sessions(limit)
|
|
472
|
+
if self.judge is not None and not self.dry_run and sessions:
|
|
473
|
+
self.log(f"judge pass · grading {len(sessions)} transcript(s) (cloud)")
|
|
474
|
+
from rockycode.engine.trajectory import append_record
|
|
475
|
+
for session in sessions:
|
|
476
|
+
graded = await self.judge.grade(session)
|
|
477
|
+
if graded is None:
|
|
478
|
+
continue # gated out or call failed — heuristic stands
|
|
479
|
+
self.report.sessions_judged += 1
|
|
480
|
+
self.report.decisions.append(
|
|
481
|
+
f"JUDGE {session['session_id'][:15]}: score {graded['score']}"
|
|
482
|
+
)
|
|
483
|
+
append_record(Path(session["path"]), "outcome", graded)
|
|
484
|
+
session["outcome"] = graded # the digest below sees the verdict
|
|
485
|
+
self.log(f"job 1/4 · digest {len(sessions)} session(s)")
|
|
486
|
+
episode_summaries: list[str] = []
|
|
487
|
+
all_facts: list[tuple[str, str]] = []
|
|
488
|
+
failure_notes: list[tuple[str, str]] = []
|
|
489
|
+
for session in sessions:
|
|
490
|
+
facts, note, fnote = await self.digest(session)
|
|
491
|
+
episode_summaries.append(note)
|
|
492
|
+
all_facts.extend((f, session["session_id"]) for f in facts)
|
|
493
|
+
if fnote:
|
|
494
|
+
failure_notes.append((session["session_id"], fnote))
|
|
495
|
+
|
|
496
|
+
if failure_notes:
|
|
497
|
+
self.log(f"weakness mining · {len(failure_notes)} failure note(s)")
|
|
498
|
+
from rockycode.dream.mining import mine_weaknesses
|
|
499
|
+
await mine_weaknesses(self, failure_notes)
|
|
500
|
+
|
|
501
|
+
if sessions:
|
|
502
|
+
from rockycode.dream.proposals import draft_proposals, draft_routine_proposals
|
|
503
|
+
sids = [s["session_id"] for s in sessions]
|
|
504
|
+
await draft_proposals(self, episode_summaries, sids)
|
|
505
|
+
await draft_routine_proposals(self, episode_summaries, sids)
|
|
506
|
+
|
|
507
|
+
self.log(f"job 2/4 · reconcile {len(all_facts)} fact(s)")
|
|
508
|
+
for fact, sid in all_facts:
|
|
509
|
+
await self.reconcile(fact, sid, index=index)
|
|
510
|
+
|
|
511
|
+
if sessions:
|
|
512
|
+
self.log("job 3/4 · rewrite project state")
|
|
513
|
+
await self.update_state(episode_summaries)
|
|
514
|
+
|
|
515
|
+
if index is not None and not self.dry_run:
|
|
516
|
+
self.log("job 4/4 · re-embed changed memories")
|
|
517
|
+
try:
|
|
518
|
+
self.report.reindexed = await index.reindex()
|
|
519
|
+
except Exception: # noqa: BLE001 — embedding refresh is best-effort
|
|
520
|
+
pass
|
|
521
|
+
return self.report
|
|
522
|
+
finally:
|
|
523
|
+
self._release_lock()
|
rockycode/dream/judge.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""The layered multi-angle judge (self-evolve phase 1, slice 2).
|
|
2
|
+
|
|
3
|
+
Grades a finished session's transcript into an `outcome` record
|
|
4
|
+
(source="judge") appended to that session's trajectory — the dream-grade
|
|
5
|
+
reward that sits after phase 0's heuristic one. Readers take the LAST
|
|
6
|
+
outcome record, so judge > heuristic without rewriting anything.
|
|
7
|
+
|
|
8
|
+
Layered, so the cloud is the last resort, not the first:
|
|
9
|
+
1. gate — free: the heuristic counters decide whether there is anything
|
|
10
|
+
to grade. Single-turn tool-less sessions, sessions that predate
|
|
11
|
+
outcome capture, and already-judged sessions are skipped,
|
|
12
|
+
not billed.
|
|
13
|
+
2. angles — ONE call on an OpenAI-compatible client (the session's own —
|
|
14
|
+
chat and `rockycode dream` reuse existing auth, no new keys):
|
|
15
|
+
completion, adherence, efficiency, user_feeling. The last is
|
|
16
|
+
inferred from BEHAVIOR in the transcript — corrections,
|
|
17
|
+
repeated asks, interrupts, thanks, abandonment.
|
|
18
|
+
3. score — code, not model: a fixed-weight aggregate, auditable and
|
|
19
|
+
stable across judge-model upgrades.
|
|
20
|
+
|
|
21
|
+
PRIVACY: the transcript comes from condense() with its default
|
|
22
|
+
feedback=False — the exit sheet is never in a judge prompt (the DeepSeek-
|
|
23
|
+
for-transcript / Ollama-for-sheet split, see the trajectory feedback
|
|
24
|
+
contract). Only the local Ollama dream reads the sheet.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import json
|
|
29
|
+
import re
|
|
30
|
+
import time
|
|
31
|
+
from typing import Optional
|
|
32
|
+
|
|
33
|
+
from rockycode.dream.core import condense
|
|
34
|
+
|
|
35
|
+
ANGLES = ("completion", "adherence", "efficiency", "user_feeling")
|
|
36
|
+
|
|
37
|
+
# The aggregate lives in code so a score of 0.78 means the same thing next
|
|
38
|
+
# month. Completion dominates (the harness exists to finish tasks); feeling
|
|
39
|
+
# outweighs efficiency because a happy slow session beats a fast wrong one.
|
|
40
|
+
WEIGHTS = {"completion": 0.40, "adherence": 0.25, "efficiency": 0.15, "user_feeling": 0.20}
|
|
41
|
+
|
|
42
|
+
JUDGE_PROMPT = """\
|
|
43
|
+
You are grading one finished coding-agent session from its condensed
|
|
44
|
+
transcript. No user rating is available to you — infer the user's
|
|
45
|
+
experience from behavior alone (corrections, repeated asks, interrupts,
|
|
46
|
+
gratitude, abandonment). A trailing [outcome] line holds mechanical
|
|
47
|
+
counters (tool errors, denials, interrupts, tests) — use it as evidence.
|
|
48
|
+
|
|
49
|
+
TRANSCRIPT:
|
|
50
|
+
{transcript}
|
|
51
|
+
|
|
52
|
+
Score each angle from 0.0 (bad) to 1.0 (great):
|
|
53
|
+
- completion: did the session accomplish what the user asked for?
|
|
54
|
+
- adherence: did the agent follow instructions and stay in scope?
|
|
55
|
+
- efficiency: was the path direct — few wasted steps, errors, retries?
|
|
56
|
+
- user_feeling: how satisfied does the user's BEHAVIOR suggest they were?
|
|
57
|
+
|
|
58
|
+
Reply with ONLY a JSON object, no prose around it:
|
|
59
|
+
{{"completion": 0.0, "adherence": 0.0, "efficiency": 0.0, "user_feeling": 0.0,
|
|
60
|
+
"rationale": "one or two sentences citing transcript evidence"}}
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def gate(session: dict) -> bool:
|
|
65
|
+
"""Layer 1 — anything to grade? Free, and fails closed: no heuristic
|
|
66
|
+
outcome record (pre-phase-0 session, or one that never really ran) means
|
|
67
|
+
no cloud call. An existing judge record means the same (a --dry-run pass
|
|
68
|
+
or a crash between append and digest must not double-bill)."""
|
|
69
|
+
out = session.get("outcome") or {}
|
|
70
|
+
if out.get("source") != "heuristic":
|
|
71
|
+
return False
|
|
72
|
+
return out.get("tool_calls", 0) >= 1 or out.get("turns", 0) >= 2
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _parse(answer: str) -> Optional[dict]:
|
|
76
|
+
"""Lenient JSON extraction: the first {...} block, clamped angles.
|
|
77
|
+
Any missing/non-numeric angle voids the whole grade — a partial score
|
|
78
|
+
would silently skew the aggregate."""
|
|
79
|
+
m = re.search(r"\{.*\}", answer, re.DOTALL)
|
|
80
|
+
if m is None:
|
|
81
|
+
return None
|
|
82
|
+
try:
|
|
83
|
+
obj = json.loads(m.group())
|
|
84
|
+
except json.JSONDecodeError:
|
|
85
|
+
return None
|
|
86
|
+
if not isinstance(obj, dict):
|
|
87
|
+
return None
|
|
88
|
+
angles: dict[str, float] = {}
|
|
89
|
+
for a in ANGLES:
|
|
90
|
+
try:
|
|
91
|
+
angles[a] = max(0.0, min(1.0, float(obj[a])))
|
|
92
|
+
except (KeyError, TypeError, ValueError):
|
|
93
|
+
return None
|
|
94
|
+
return {"angles": angles, "rationale": str(obj.get("rationale", "")).strip()[:400]}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class TranscriptJudge:
|
|
98
|
+
"""Wraps an OpenAI-compatible client the caller already owns — the live
|
|
99
|
+
engine's in the TUI, a require_key() one in `rockycode dream`. No key
|
|
100
|
+
available → the caller simply passes no judge and dream stays local."""
|
|
101
|
+
|
|
102
|
+
def __init__(self, client, model: str) -> None:
|
|
103
|
+
self.client = client
|
|
104
|
+
self.model = model
|
|
105
|
+
|
|
106
|
+
async def grade(self, session: dict) -> Optional[dict]:
|
|
107
|
+
"""One judge outcome dict, or None (gated out / call failed / bad
|
|
108
|
+
JSON). Never raises — a failed grade must not stop the dream pass."""
|
|
109
|
+
if not gate(session):
|
|
110
|
+
return None
|
|
111
|
+
try:
|
|
112
|
+
resp = await self.client.chat.completions.create(
|
|
113
|
+
model=self.model,
|
|
114
|
+
# condense() default: feedback stays OUT of this cloud prompt.
|
|
115
|
+
messages=[{"role": "user", "content": JUDGE_PROMPT.format(
|
|
116
|
+
transcript=condense(session))}],
|
|
117
|
+
max_tokens=512,
|
|
118
|
+
extra_body={"thinking": {"type": "disabled"}}, # cheap + fast, like loop.py's off-mode
|
|
119
|
+
)
|
|
120
|
+
answer = (resp.choices[0].message.content or "") if resp.choices else ""
|
|
121
|
+
except Exception: # noqa: BLE001 — API trouble = no grade, dream goes on
|
|
122
|
+
return None
|
|
123
|
+
parsed = _parse(answer)
|
|
124
|
+
if parsed is None:
|
|
125
|
+
return None
|
|
126
|
+
score = round(sum(WEIGHTS[a] * parsed["angles"][a] for a in ANGLES), 4)
|
|
127
|
+
return {
|
|
128
|
+
"source": "judge",
|
|
129
|
+
"graded_at": time.time(),
|
|
130
|
+
"judge_model": self.model,
|
|
131
|
+
"angles": parsed["angles"],
|
|
132
|
+
"score": score,
|
|
133
|
+
"rationale": parsed["rationale"],
|
|
134
|
+
}
|