rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""Weakness mining (self-evolve phase 1, slice 3) — Self-Harness's first stage.
|
|
2
|
+
|
|
3
|
+
After the dream digests a pass's sessions, the failures among them are mined
|
|
4
|
+
for RECURRING patterns with causal information, stored as `weakness` memories
|
|
5
|
+
(<memory>/weaknesses/*.md, archive-not-delete like everything else). They
|
|
6
|
+
surface as index lines in the system prompt and via recall_memory — the
|
|
7
|
+
groundwork for the proposals inbox, which will turn hot weaknesses into
|
|
8
|
+
prompt/skill proposals.
|
|
9
|
+
|
|
10
|
+
Two layers, mirroring the judge:
|
|
11
|
+
1. failure_note — free, code-side gate: a session contributes only when it
|
|
12
|
+
shows real failure signals (low judge score, tool/engine errors,
|
|
13
|
+
interrupts, failed tests, or the digest's own "## failed" bullets).
|
|
14
|
+
No signals in the whole pass → no mining call at all.
|
|
15
|
+
2. mine_weaknesses — ONE local Ollama call over the pass's failure notes
|
|
16
|
+
plus the known weaknesses, so recurrence REINFORCES (importance up,
|
|
17
|
+
evidence appended) instead of duplicating. Local on purpose: episode
|
|
18
|
+
digests may carry exit-sheet-derived text, which must never reach a
|
|
19
|
+
cloud model — mining stays on the Ollama side of the split.
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import json
|
|
24
|
+
import re
|
|
25
|
+
from typing import Optional
|
|
26
|
+
|
|
27
|
+
from rockycode.dream.core import parse_bullets
|
|
28
|
+
from rockycode.memory.store import Memory, _slugify
|
|
29
|
+
|
|
30
|
+
# Below this judge score a session counts as a failure signal even when the
|
|
31
|
+
# mechanical counters look clean (e.g. the agent confidently did the wrong thing).
|
|
32
|
+
JUDGE_BAR = 0.7
|
|
33
|
+
|
|
34
|
+
MAX_PATTERNS = 3 # per pass — weaknesses should be rare and load-bearing
|
|
35
|
+
|
|
36
|
+
MINING_PROMPT = """\
|
|
37
|
+
You maintain a coding agent's short list of its own recurring weaknesses.
|
|
38
|
+
|
|
39
|
+
FAILURE NOTES from recent sessions:
|
|
40
|
+
{notes}
|
|
41
|
+
|
|
42
|
+
KNOWN WEAKNESSES:
|
|
43
|
+
{existing}
|
|
44
|
+
|
|
45
|
+
Identify at most {max_patterns} RECURRING failure patterns with a likely
|
|
46
|
+
cause. Only patterns the notes actually support — no speculation, and an
|
|
47
|
+
empty list is a fine answer. If a note is another instance of a KNOWN
|
|
48
|
+
weakness, reinforce that one instead of writing a new one.
|
|
49
|
+
|
|
50
|
+
Reply with ONLY a JSON array (possibly empty):
|
|
51
|
+
[{{"pattern": "one-line name of the failure pattern",
|
|
52
|
+
"cause": "likely root cause, one or two sentences",
|
|
53
|
+
"advice": "one actionable instruction that would avoid it",
|
|
54
|
+
"reinforces": "name-of-known-weakness or null"}}]
|
|
55
|
+
"""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def failure_note(session: dict, sections: dict) -> Optional[str]:
|
|
59
|
+
"""Layer 1 — compact failure evidence for one digested session, or None
|
|
60
|
+
when it shows no failure signals (free; decides whether mining runs)."""
|
|
61
|
+
heur = session.get("heuristic") or {}
|
|
62
|
+
out = session.get("outcome") or {}
|
|
63
|
+
signals: list[str] = []
|
|
64
|
+
if out.get("source") == "judge":
|
|
65
|
+
try:
|
|
66
|
+
score = float(out.get("score", 1.0))
|
|
67
|
+
except (TypeError, ValueError):
|
|
68
|
+
score = 1.0
|
|
69
|
+
if score < JUDGE_BAR:
|
|
70
|
+
signals.append(f"judge score {score} — {str(out.get('rationale', ''))[:200]}")
|
|
71
|
+
if heur.get("tool_errors"):
|
|
72
|
+
signals.append(f"{heur['tool_errors']} tool error(s)")
|
|
73
|
+
if heur.get("engine_errors"):
|
|
74
|
+
signals.append(f"{heur['engine_errors']} engine error(s)")
|
|
75
|
+
if heur.get("interrupts"):
|
|
76
|
+
signals.append(f"{heur['interrupts']} user interrupt(s) mid-work")
|
|
77
|
+
tests = heur.get("tests") or {}
|
|
78
|
+
if tests.get("run", 0) > tests.get("passed", 0):
|
|
79
|
+
signals.append(f"tests: only {tests.get('passed', 0)}/{tests['run']} passed")
|
|
80
|
+
failed = parse_bullets(sections.get("failed", ""))
|
|
81
|
+
if failed:
|
|
82
|
+
signals.append("failed approaches: " + "; ".join(failed[:4]))
|
|
83
|
+
if not signals:
|
|
84
|
+
return None
|
|
85
|
+
task = (sections.get("task") or "(unknown task)").strip().splitlines()[0][:150]
|
|
86
|
+
return f"### {task}\n" + "\n".join(f"- {s}" for s in signals)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _parse_items(answer: str) -> list[dict]:
|
|
90
|
+
"""Lenient array extraction; items missing any required field are dropped
|
|
91
|
+
(a pattern without cause/advice is a vibe, not a weakness)."""
|
|
92
|
+
m = re.search(r"\[.*\]", answer, re.DOTALL)
|
|
93
|
+
if m is None:
|
|
94
|
+
return []
|
|
95
|
+
try:
|
|
96
|
+
arr = json.loads(m.group())
|
|
97
|
+
except json.JSONDecodeError:
|
|
98
|
+
return []
|
|
99
|
+
if not isinstance(arr, list):
|
|
100
|
+
return []
|
|
101
|
+
items = []
|
|
102
|
+
for it in arr:
|
|
103
|
+
if not isinstance(it, dict):
|
|
104
|
+
continue
|
|
105
|
+
if all(isinstance(it.get(k), str) and it[k].strip() for k in ("pattern", "cause", "advice")):
|
|
106
|
+
items.append(it)
|
|
107
|
+
return items[:MAX_PATTERNS]
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
async def mine_weaknesses(runner, notes: list[tuple[str, str]]) -> None:
|
|
111
|
+
"""Layer 2 — one local model call over this pass's failure notes.
|
|
112
|
+
Mutates runner.report; respects runner.dry_run (decisions only)."""
|
|
113
|
+
if not notes:
|
|
114
|
+
return
|
|
115
|
+
existing = [m for m in runner.store.load_all() if m.type == "weakness"]
|
|
116
|
+
existing_txt = "\n".join(f"[{m.name}] {m.description}" for m in existing) or "(none)"
|
|
117
|
+
answer = await runner.chat.chat(
|
|
118
|
+
MINING_PROMPT.format(
|
|
119
|
+
notes="\n\n".join(n for _, n in notes),
|
|
120
|
+
existing=existing_txt,
|
|
121
|
+
max_patterns=MAX_PATTERNS,
|
|
122
|
+
),
|
|
123
|
+
max_tokens=1024,
|
|
124
|
+
)
|
|
125
|
+
sids = [sid for sid, _ in notes]
|
|
126
|
+
known = {m.name for m in existing}
|
|
127
|
+
for it in _parse_items(answer):
|
|
128
|
+
target = (it.get("reinforces") or "").strip("`'\" ")
|
|
129
|
+
name = target if target in known else _slugify(it["pattern"])
|
|
130
|
+
if name in known:
|
|
131
|
+
# Recurrence: same pattern seen again → more important, more evidence.
|
|
132
|
+
mem = runner.store.get(name)
|
|
133
|
+
runner.report.weaknesses_reinforced += 1
|
|
134
|
+
runner.report.decisions.append(f"WEAKNESS ~{name} (reinforced): {it['pattern'][:80]}")
|
|
135
|
+
if not runner.dry_run and mem is not None:
|
|
136
|
+
mem.importance = min(10, mem.importance + 1)
|
|
137
|
+
mem.evidence.extend(s for s in sids if s not in mem.evidence)
|
|
138
|
+
runner.store.save(mem)
|
|
139
|
+
else:
|
|
140
|
+
runner.report.weaknesses_added += 1
|
|
141
|
+
runner.report.decisions.append(f"WEAKNESS +{name}: {it['pattern'][:80]}")
|
|
142
|
+
if not runner.dry_run:
|
|
143
|
+
runner.store.save(Memory(
|
|
144
|
+
name=name,
|
|
145
|
+
type="weakness",
|
|
146
|
+
description=it["pattern"][:150],
|
|
147
|
+
importance=6,
|
|
148
|
+
origin="dream",
|
|
149
|
+
evidence=list(sids),
|
|
150
|
+
body=(f"{it['pattern']}\n\n**cause:** {it['cause']}\n"
|
|
151
|
+
f"**advice:** {it['advice']}"),
|
|
152
|
+
))
|
|
@@ -0,0 +1,440 @@
|
|
|
1
|
+
"""The proposals inbox (self-evolve phase 1, slice 4) — dream's only actuator.
|
|
2
|
+
|
|
3
|
+
Memory writes are free; anything EXECUTABLE is proposal-only. The dream may
|
|
4
|
+
draft a skill it thinks the project needs — from hot (reinforced) weaknesses
|
|
5
|
+
and this pass's episodes — but nothing self-installs: drafts land in a global
|
|
6
|
+
pending inbox and the user approves or archives each from the /proposals card
|
|
7
|
+
in the TUI. Approve installs a SKILL.md into the global ~/.rockycode/skills
|
|
8
|
+
(project skills still win discovery); archive keeps the file for provenance —
|
|
9
|
+
nothing is deleted, same rule as memory.
|
|
10
|
+
|
|
11
|
+
Drafting runs on LOCAL Ollama (episode text may carry exit-sheet-derived
|
|
12
|
+
content, which never goes to a cloud model) and is deliberately scarce:
|
|
13
|
+
at most ONE draft per dream pass, and none while the project already has
|
|
14
|
+
MAX_PENDING_PER_PROJECT waiting — an inbox that piles up is one that gets
|
|
15
|
+
ignored.
|
|
16
|
+
|
|
17
|
+
Proposal files carry `evidence:` as rk_ PUBLIC session ids (the stable
|
|
18
|
+
human-facing handle; resolve_session maps them back).
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
import os
|
|
24
|
+
import re
|
|
25
|
+
import time
|
|
26
|
+
from dataclasses import dataclass, field
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
from typing import Optional
|
|
29
|
+
|
|
30
|
+
from rockycode.engine.skills import _parse_frontmatter
|
|
31
|
+
|
|
32
|
+
PENDING, APPROVED, ARCHIVED = "pending", "approved", "archived"
|
|
33
|
+
MAX_PENDING_PER_PROJECT = 3
|
|
34
|
+
HOT_IMPORTANCE = 7 # a weakness reinforced at least once
|
|
35
|
+
|
|
36
|
+
PROPOSE_PROMPT = """\
|
|
37
|
+
You draft SKILL playbooks for a coding agent — but ONLY when the evidence
|
|
38
|
+
truly supports one. A skill is a reusable, step-by-step procedure the agent
|
|
39
|
+
should follow for a recurring kind of task in this project.
|
|
40
|
+
|
|
41
|
+
RECURRING WEAKNESSES (reinforced across sessions):
|
|
42
|
+
{weaknesses}
|
|
43
|
+
|
|
44
|
+
THIS PASS'S SESSION NOTES:
|
|
45
|
+
{episodes}
|
|
46
|
+
|
|
47
|
+
SKILLS THAT ALREADY EXIST (never re-draft these):
|
|
48
|
+
{existing}
|
|
49
|
+
|
|
50
|
+
The weaknesses above are already RECURRING — each was observed across
|
|
51
|
+
multiple sessions. If a step-by-step checklist would prevent one of them,
|
|
52
|
+
draft that skill now. With no weaknesses listed, draft only when the session
|
|
53
|
+
notes themselves show a recurring procedure worth packaging. If neither
|
|
54
|
+
holds, reply with exactly: null
|
|
55
|
+
|
|
56
|
+
Reply with ONLY a JSON object (or the word null):
|
|
57
|
+
{{"name": "short-kebab-case-name",
|
|
58
|
+
"description": "one line: when the agent should reach for this",
|
|
59
|
+
"when_to_use": "one or two sentences",
|
|
60
|
+
"steps": "markdown bullet list of concrete steps"}}
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _home_root() -> Path:
|
|
65
|
+
base = os.environ.get("ROCKYCODE_HOME")
|
|
66
|
+
return Path(base).expanduser() if base else Path.home() / ".rockycode"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def proposals_dir() -> Path:
|
|
70
|
+
"""Global inbox, $ROCKYCODE_HOME-aware like trajectories. Proposal meta
|
|
71
|
+
carries project_id, so the TUI filters per project."""
|
|
72
|
+
return _home_root() / "proposals"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def skills_home() -> Path:
|
|
76
|
+
"""Global install target for approved skills (in skill discovery order
|
|
77
|
+
after the project dirs — a project skill always wins)."""
|
|
78
|
+
return _home_root() / "skills"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass
|
|
82
|
+
class Proposal:
|
|
83
|
+
name: str
|
|
84
|
+
kind: str = "skill" # skill | (later: routine, prompt-edit)
|
|
85
|
+
description: str = ""
|
|
86
|
+
status: str = PENDING
|
|
87
|
+
origin: str = "dream"
|
|
88
|
+
created: str = ""
|
|
89
|
+
project_id: str = ""
|
|
90
|
+
project_name: str = ""
|
|
91
|
+
reason: str = "" # why dream drafted it (weakness/pattern)
|
|
92
|
+
evidence: list[str] = field(default_factory=list) # rk_ public session ids
|
|
93
|
+
body: str = "" # the drafted SKILL.md body
|
|
94
|
+
path: Optional[Path] = None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _slugify(text: str, max_len: int = 48) -> str:
|
|
98
|
+
slug = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
|
|
99
|
+
return slug[:max_len].rstrip("-") or "proposal"
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def to_markdown(p: Proposal) -> str:
|
|
103
|
+
return "\n".join([
|
|
104
|
+
"---",
|
|
105
|
+
f"name: {p.name}",
|
|
106
|
+
f"kind: {p.kind}",
|
|
107
|
+
f"description: {p.description}",
|
|
108
|
+
f"status: {p.status}",
|
|
109
|
+
f"origin: {p.origin}",
|
|
110
|
+
f"created: {p.created}",
|
|
111
|
+
f"project_id: {p.project_id}",
|
|
112
|
+
f"project_name: {p.project_name}",
|
|
113
|
+
f"reason: {p.reason}",
|
|
114
|
+
f"evidence: [{', '.join(p.evidence)}]",
|
|
115
|
+
"---",
|
|
116
|
+
"",
|
|
117
|
+
p.body.strip(),
|
|
118
|
+
"",
|
|
119
|
+
])
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def parse_proposal(text: str, path: Optional[Path] = None) -> Proposal:
|
|
123
|
+
fields, body = _parse_frontmatter(text)
|
|
124
|
+
ev = [v.strip().strip("'\"") for v in fields.get("evidence", "").strip("[]").split(",")
|
|
125
|
+
if v.strip().strip("'\"")]
|
|
126
|
+
return Proposal(
|
|
127
|
+
name=fields.get("name") or (path.stem if path else "proposal"),
|
|
128
|
+
kind=fields.get("kind", "skill"),
|
|
129
|
+
description=fields.get("description", ""),
|
|
130
|
+
status=fields.get("status", PENDING),
|
|
131
|
+
origin=fields.get("origin", "dream"),
|
|
132
|
+
created=fields.get("created", ""),
|
|
133
|
+
project_id=fields.get("project_id", ""),
|
|
134
|
+
project_name=fields.get("project_name", ""),
|
|
135
|
+
reason=fields.get("reason", ""),
|
|
136
|
+
evidence=ev,
|
|
137
|
+
body=body.strip(),
|
|
138
|
+
path=path,
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
class ProposalStore:
|
|
143
|
+
"""pending/ approved/ archived/ under one global root — status IS the
|
|
144
|
+
directory, so `ls` tells the truth and nothing needs a database."""
|
|
145
|
+
|
|
146
|
+
def __init__(self, root: Optional[Path] = None) -> None:
|
|
147
|
+
self.root = root or proposals_dir()
|
|
148
|
+
|
|
149
|
+
def list(self, status: str = PENDING, project_id: Optional[str] = None) -> list[Proposal]:
|
|
150
|
+
folder = self.root / status
|
|
151
|
+
if not folder.is_dir():
|
|
152
|
+
return []
|
|
153
|
+
out = []
|
|
154
|
+
for f in sorted(folder.glob("*.md")):
|
|
155
|
+
try:
|
|
156
|
+
p = parse_proposal(f.read_text(encoding="utf-8", errors="replace"), path=f)
|
|
157
|
+
except OSError:
|
|
158
|
+
continue
|
|
159
|
+
if project_id is None or p.project_id == project_id:
|
|
160
|
+
out.append(p)
|
|
161
|
+
return out
|
|
162
|
+
|
|
163
|
+
def all_names(self) -> set[str]:
|
|
164
|
+
"""Every proposal name in any status — the dedup set for drafting."""
|
|
165
|
+
names = set()
|
|
166
|
+
for status in (PENDING, APPROVED, ARCHIVED):
|
|
167
|
+
folder = self.root / status
|
|
168
|
+
if folder.is_dir():
|
|
169
|
+
names.update(f.stem for f in folder.glob("*.md"))
|
|
170
|
+
return names
|
|
171
|
+
|
|
172
|
+
def save(self, p: Proposal) -> Path:
|
|
173
|
+
if not p.created:
|
|
174
|
+
p.created = time.strftime("%Y-%m-%d")
|
|
175
|
+
folder = self.root / p.status
|
|
176
|
+
folder.mkdir(parents=True, exist_ok=True)
|
|
177
|
+
path = folder / f"{_slugify(p.name)}.md"
|
|
178
|
+
path.write_text(to_markdown(p), encoding="utf-8")
|
|
179
|
+
p.path = path
|
|
180
|
+
return path
|
|
181
|
+
|
|
182
|
+
def _move(self, p: Proposal, status: str) -> Path:
|
|
183
|
+
old = p.path
|
|
184
|
+
p.status = status
|
|
185
|
+
new_path = self.save(p)
|
|
186
|
+
if old is not None and old != new_path:
|
|
187
|
+
try:
|
|
188
|
+
old.unlink()
|
|
189
|
+
except OSError:
|
|
190
|
+
pass
|
|
191
|
+
return new_path
|
|
192
|
+
|
|
193
|
+
def approve(self, p: Proposal) -> Path:
|
|
194
|
+
"""Install the drafted proposal by KIND and file it under approved/ for
|
|
195
|
+
provenance. A skill installs a SKILL.md; a routine installs a
|
|
196
|
+
routine.toml (disabled-of-auto, no lease — the user grants the run
|
|
197
|
+
envelope on the enable card, so nothing dream-drafted ever self-runs).
|
|
198
|
+
Returns the installed path."""
|
|
199
|
+
if p.kind == "routine":
|
|
200
|
+
return self._approve_routine(p)
|
|
201
|
+
name = _slugify(p.name)
|
|
202
|
+
target = skills_home() / name
|
|
203
|
+
while (target / "SKILL.md").exists():
|
|
204
|
+
name += "-dream"
|
|
205
|
+
target = skills_home() / name
|
|
206
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
207
|
+
skill_md = "\n".join([
|
|
208
|
+
"---",
|
|
209
|
+
f"name: {name}",
|
|
210
|
+
f"description: {p.description}",
|
|
211
|
+
"origin: dream",
|
|
212
|
+
f"evidence: [{', '.join(p.evidence)}]",
|
|
213
|
+
f"project: {p.project_name}",
|
|
214
|
+
"---",
|
|
215
|
+
"",
|
|
216
|
+
p.body.strip(),
|
|
217
|
+
"",
|
|
218
|
+
])
|
|
219
|
+
installed = target / "SKILL.md"
|
|
220
|
+
installed.write_text(skill_md, encoding="utf-8")
|
|
221
|
+
self._move(p, APPROVED)
|
|
222
|
+
return installed
|
|
223
|
+
|
|
224
|
+
def _approve_routine(self, p: Proposal) -> Path:
|
|
225
|
+
"""Install a drafted routine.toml (p.body IS the toml). Installed
|
|
226
|
+
enabled-but-click-to-run: auto=false and no lease, so it shows as a due
|
|
227
|
+
card the user runs by hand until they explicitly grant the auto lease —
|
|
228
|
+
a dream draft never earns unattended execution on its own."""
|
|
229
|
+
from rockycode.routines import routines_dir
|
|
230
|
+
name = _slugify(p.name)
|
|
231
|
+
target = routines_dir() / name
|
|
232
|
+
while (target / "routine.toml").exists():
|
|
233
|
+
name += "-dream"
|
|
234
|
+
target = routines_dir() / name
|
|
235
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
236
|
+
(target / "routine.toml").write_text(p.body.strip() + "\n", encoding="utf-8")
|
|
237
|
+
self._move(p, APPROVED)
|
|
238
|
+
return target / "routine.toml"
|
|
239
|
+
|
|
240
|
+
def archive(self, p: Proposal) -> Path:
|
|
241
|
+
return self._move(p, ARCHIVED)
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def failing_routines(project_id: str, *, window: int = 4, min_fails: int = 2) -> list[dict]:
|
|
245
|
+
"""Routines whose recent runs keep failing — the signal for an amendment
|
|
246
|
+
proposal. A run status is 'done' on success; anything else (blocked/budget/
|
|
247
|
+
error) counts as a failure. Returns the routine + its recent failure count."""
|
|
248
|
+
from rockycode.routines import RoutineStore
|
|
249
|
+
store = RoutineStore()
|
|
250
|
+
out = []
|
|
251
|
+
for r in store.list(project_id=project_id):
|
|
252
|
+
runs = store.state(r).runs[-window:]
|
|
253
|
+
fails = [x for x in runs if x.get("status") != "done"]
|
|
254
|
+
if len(runs) >= min_fails and len(fails) >= min_fails:
|
|
255
|
+
out.append({"routine": r, "fails": len(fails), "of": len(runs),
|
|
256
|
+
"last_status": (runs[-1].get("status") if runs else "error")})
|
|
257
|
+
return out
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
PROPOSE_ROUTINE_PROMPT = """\
|
|
261
|
+
You draft ROUTINES for a coding agent — a routine is a task the agent runs by
|
|
262
|
+
itself on a schedule (daily or weekly). Draft one ONLY when the evidence
|
|
263
|
+
genuinely supports it.
|
|
264
|
+
|
|
265
|
+
A FAILING ROUTINE that needs amending (fix its prompt/cadence/budget):
|
|
266
|
+
{failing}
|
|
267
|
+
|
|
268
|
+
THIS PASS'S SESSION NOTES (look for a recurring MANUAL task worth automating):
|
|
269
|
+
{episodes}
|
|
270
|
+
|
|
271
|
+
ROUTINES THAT ALREADY EXIST (never re-draft these):
|
|
272
|
+
{existing}
|
|
273
|
+
|
|
274
|
+
If a routine above is failing, draft an amendment (same name, a clearer prompt
|
|
275
|
+
or a smaller/larger cadence). Otherwise, if the notes show the user repeatedly
|
|
276
|
+
doing the same task by hand, draft a routine to do it for them. If neither
|
|
277
|
+
holds, reply with exactly: null
|
|
278
|
+
|
|
279
|
+
Reply with ONLY a JSON object (or the word null):
|
|
280
|
+
{{"name": "short-kebab-case-name",
|
|
281
|
+
"description": "one line: what this routine does",
|
|
282
|
+
"cadence": "daily" or "weekly",
|
|
283
|
+
"prompt": "the exact task line the agent should run each time"}}
|
|
284
|
+
"""
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _parse_routine_draft(answer: str) -> Optional[dict]:
|
|
288
|
+
if answer.strip().lower().startswith("null"):
|
|
289
|
+
return None
|
|
290
|
+
m = re.search(r"\{.*\}", answer, re.DOTALL)
|
|
291
|
+
if m is None:
|
|
292
|
+
return None
|
|
293
|
+
try:
|
|
294
|
+
obj = json.loads(m.group())
|
|
295
|
+
except json.JSONDecodeError:
|
|
296
|
+
return None
|
|
297
|
+
if not isinstance(obj, dict):
|
|
298
|
+
return None
|
|
299
|
+
if obj.get("cadence") not in ("daily", "weekly"):
|
|
300
|
+
obj["cadence"] = "daily"
|
|
301
|
+
if not all(isinstance(obj.get(k), str) and obj[k].strip()
|
|
302
|
+
for k in ("name", "description", "prompt")):
|
|
303
|
+
return None
|
|
304
|
+
return obj
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
async def draft_routine_proposals(runner, episode_summaries: list[str], session_ids: list[str]) -> None:
|
|
308
|
+
"""The dream's ROUTINE-proposal job (phase 2, slice 4): amend a failing
|
|
309
|
+
routine, or propose automating a recurring manual task. Local Ollama only,
|
|
310
|
+
at most one draft, inbox-capped. Mutates runner.report; respects dry_run."""
|
|
311
|
+
from rockycode.routines import Routine, RoutineStore, _emit_toml
|
|
312
|
+
from rockycode.session import get_project, public_id
|
|
313
|
+
|
|
314
|
+
store = ProposalStore()
|
|
315
|
+
project = get_project(runner.workdir)
|
|
316
|
+
if len(store.list(PENDING, project_id=project.id)) >= MAX_PENDING_PER_PROJECT:
|
|
317
|
+
return
|
|
318
|
+
|
|
319
|
+
failing = failing_routines(project.id)
|
|
320
|
+
if not failing and len(episode_summaries) < 2:
|
|
321
|
+
return # no failing routine and not enough recurrence to automate
|
|
322
|
+
|
|
323
|
+
rstore = RoutineStore()
|
|
324
|
+
existing = "\n".join(f"- {r.name}: {r.description}" for r in rstore.list(project.id)) or "(none)"
|
|
325
|
+
taken = {r.name for r in rstore.list(project.id)} | store.all_names()
|
|
326
|
+
fail_txt = "\n".join(
|
|
327
|
+
f"- {f['routine'].name}: failed {f['fails']}/{f['of']} recent runs "
|
|
328
|
+
f"(last: {f['last_status']}) — prompt was: {f['routine'].prompt[:160]}"
|
|
329
|
+
for f in failing) or "(none)"
|
|
330
|
+
|
|
331
|
+
runner.log("routine-proposal pass · drafting (local)")
|
|
332
|
+
answer = await runner.chat.chat(PROPOSE_ROUTINE_PROMPT.format(
|
|
333
|
+
failing=fail_txt,
|
|
334
|
+
episodes="\n\n".join(episode_summaries)[:4000],
|
|
335
|
+
existing=existing,
|
|
336
|
+
), max_tokens=1024)
|
|
337
|
+
draft = _parse_routine_draft(answer)
|
|
338
|
+
if draft is None:
|
|
339
|
+
return
|
|
340
|
+
name = _slugify(draft["name"])
|
|
341
|
+
is_amendment = any(f["routine"].name == name for f in failing)
|
|
342
|
+
if name in taken and not is_amendment:
|
|
343
|
+
return # a NEW routine can't reuse a name; an amendment reuses on purpose
|
|
344
|
+
|
|
345
|
+
# Conservative envelope — the user grants the real one on the enable card.
|
|
346
|
+
routine = Routine(
|
|
347
|
+
name=name, description=draft["description"][:200], cadence=draft["cadence"],
|
|
348
|
+
prompt=draft["prompt"], workdir=str(runner.workdir), project_id=project.id,
|
|
349
|
+
network=False, isolation=False, budget_run=0.10, max_steps=30,
|
|
350
|
+
auto=False, lease_deadline=0.0, enabled=True,
|
|
351
|
+
)
|
|
352
|
+
reason = (f"amend failing routine '{name}'" if is_amendment
|
|
353
|
+
else "recurring manual task in episodes")
|
|
354
|
+
proposal = Proposal(
|
|
355
|
+
name=name, kind="routine", description=draft["description"][:200],
|
|
356
|
+
project_id=project.id, project_name=project.name, reason=reason,
|
|
357
|
+
evidence=[public_id(sid) for sid in session_ids],
|
|
358
|
+
body=_emit_toml(routine),
|
|
359
|
+
)
|
|
360
|
+
if runner.dry_run:
|
|
361
|
+
runner.log(f"routine proposal (dry-run, not saved): {name} · {reason}")
|
|
362
|
+
return
|
|
363
|
+
store.save(proposal)
|
|
364
|
+
runner.report.proposals_drafted += 1
|
|
365
|
+
runner.log(f"drafted routine proposal: {name} — /proposals to review · {reason}")
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def _parse_draft(answer: str) -> Optional[dict]:
|
|
369
|
+
if answer.strip().lower().startswith("null"):
|
|
370
|
+
return None
|
|
371
|
+
m = re.search(r"\{.*\}", answer, re.DOTALL)
|
|
372
|
+
if m is None:
|
|
373
|
+
return None
|
|
374
|
+
try:
|
|
375
|
+
obj = json.loads(m.group())
|
|
376
|
+
except json.JSONDecodeError:
|
|
377
|
+
return None
|
|
378
|
+
if not isinstance(obj, dict):
|
|
379
|
+
return None
|
|
380
|
+
# The prompt asks for a markdown-string `steps`, but stronger models
|
|
381
|
+
# reasonably emit a JSON array (observed live with qwen3.5:9b) —
|
|
382
|
+
# normalize a list of strings into bullets instead of voiding the draft.
|
|
383
|
+
steps = obj.get("steps")
|
|
384
|
+
if isinstance(steps, list) and steps and all(isinstance(s, str) for s in steps):
|
|
385
|
+
obj["steps"] = "\n".join(f"- {s.strip()}" for s in steps)
|
|
386
|
+
if not all(isinstance(obj.get(k), str) and obj[k].strip()
|
|
387
|
+
for k in ("name", "description", "when_to_use", "steps")):
|
|
388
|
+
return None
|
|
389
|
+
return obj
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
async def draft_proposals(runner, episode_summaries: list[str], session_ids: list[str]) -> None:
|
|
393
|
+
"""The dream's proposal job — gate first, one local call, at most one
|
|
394
|
+
draft. Mutates runner.report; respects runner.dry_run."""
|
|
395
|
+
from rockycode.engine.skills import discover_skills
|
|
396
|
+
from rockycode.session import get_project, public_id
|
|
397
|
+
|
|
398
|
+
store = ProposalStore()
|
|
399
|
+
project = get_project(runner.workdir)
|
|
400
|
+
if len(store.list(PENDING, project_id=project.id)) >= MAX_PENDING_PER_PROJECT:
|
|
401
|
+
return # the inbox is full — earn attention before asking for more
|
|
402
|
+
hot = [m for m in runner.store.load_all()
|
|
403
|
+
if m.type == "weakness" and m.importance >= HOT_IMPORTANCE]
|
|
404
|
+
if not hot and len(episode_summaries) < 2:
|
|
405
|
+
return # not enough recurrence to package anything
|
|
406
|
+
|
|
407
|
+
skills = discover_skills(runner.workdir, home=Path.home())
|
|
408
|
+
existing = "\n".join(f"- {s.name}: {s.description}" for s in skills) or "(none)"
|
|
409
|
+
taken = {s.name for s in skills} | store.all_names()
|
|
410
|
+
|
|
411
|
+
runner.log("proposal pass · drafting (local)")
|
|
412
|
+
answer = await runner.chat.chat(PROPOSE_PROMPT.format(
|
|
413
|
+
weaknesses="\n".join(f"- [{m.name}] {m.description} — {m.body[:200]}" for m in hot) or "(none)",
|
|
414
|
+
episodes="\n\n".join(episode_summaries)[:4000],
|
|
415
|
+
existing=existing,
|
|
416
|
+
), max_tokens=1536)
|
|
417
|
+
draft = _parse_draft(answer)
|
|
418
|
+
if draft is None:
|
|
419
|
+
return
|
|
420
|
+
name = _slugify(draft["name"])
|
|
421
|
+
if name in taken:
|
|
422
|
+
return # already exists or already proposed — recurrence isn't novelty
|
|
423
|
+
|
|
424
|
+
body = (f"# {draft['name']}\n\n{draft['description']}\n\n"
|
|
425
|
+
f"## when to use\n{draft['when_to_use']}\n\n"
|
|
426
|
+
f"## steps\n{draft['steps']}")
|
|
427
|
+
reason = f"hot weakness: {hot[0].name}" if hot else "recurring pattern in episodes"
|
|
428
|
+
proposal = Proposal(
|
|
429
|
+
name=name,
|
|
430
|
+
description=draft["description"][:200],
|
|
431
|
+
project_id=project.id,
|
|
432
|
+
project_name=project.name,
|
|
433
|
+
reason=reason,
|
|
434
|
+
evidence=[public_id(sid) for sid in session_ids],
|
|
435
|
+
body=body,
|
|
436
|
+
)
|
|
437
|
+
runner.report.proposals_drafted += 1
|
|
438
|
+
runner.report.decisions.append(f"PROPOSAL +{name}: {proposal.description[:80]}")
|
|
439
|
+
if not runner.dry_run:
|
|
440
|
+
store.save(proposal)
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""rockycode engine: the model-and-UI-agnostic agent core.
|
|
2
|
+
|
|
3
|
+
The engine runs the ReAct loop and emits a stream of events. UIs (TUI now,
|
|
4
|
+
desktop later), the trajectory logger, and the future game layer are all
|
|
5
|
+
just subscribers to that stream — none of them are imported here.
|
|
6
|
+
"""
|
|
7
|
+
from rockycode.engine.events import AgentState, Event
|
|
8
|
+
from rockycode.engine.loop import Engine
|
|
9
|
+
|
|
10
|
+
__all__ = ["Engine", "Event", "AgentState"]
|