rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/engine/web.py
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
1
|
+
"""Web tools — chat only (bench stays offline + uncontaminated).
|
|
2
|
+
|
|
3
|
+
Verified live against the DeepSeek API (2026-06-22):
|
|
4
|
+
- OpenAI /v1 endpoint has NO native search (server tools rejected).
|
|
5
|
+
- Anthropic /anthropic endpoint runs native server-side search via the
|
|
6
|
+
`web_search_20260209` / `web_search_20250305` server tools. DeepSeek does
|
|
7
|
+
the searching AND page-reading on its servers → works behind restrictive
|
|
8
|
+
networks (client only needs api.deepseek.com). This is the primary backend.
|
|
9
|
+
- No server-side fetch tool. `web_fetch(url)` runs client-side and may fail
|
|
10
|
+
on blocked sites — a known limitation; native search covers most needs.
|
|
11
|
+
|
|
12
|
+
Search backends are tiered and polite:
|
|
13
|
+
native (DeepSeek) → brave (opt-in) → bing (keyless) → duckduckgo (keyless)
|
|
14
|
+
Native is the one that matters — and the only one that reliably works behind
|
|
15
|
+
restrictive networks: the client only needs api.deepseek.com, and the search
|
|
16
|
+
+ page-reading happen on DeepSeek's servers. Among the scrapers, Bing is
|
|
17
|
+
ordered before DuckDuckGo because bing.com reaches more restricted regions
|
|
18
|
+
(DuckDuckGo is blocked in some). Brave activates ONLY when the user sets
|
|
19
|
+
their own BRAVE_API_KEY — we never proxy a shared key (wrong fit for an MIT
|
|
20
|
+
project: cost, abuse, liability). Override with
|
|
21
|
+
ROCKYCODE_SEARCH_ORDER="native,bing,duckduckgo,brave".
|
|
22
|
+
|
|
23
|
+
Politeness: an honest identifying User-Agent (no browser impersonation), a
|
|
24
|
+
single attempt per backend (no retry storms), and a concurrency cap so
|
|
25
|
+
parallel research never floods a service.
|
|
26
|
+
|
|
27
|
+
`web_research(queries)` fans web_search out across queries in parallel — each
|
|
28
|
+
native call is itself a server-side search agent, so this is the "parallel
|
|
29
|
+
search" pattern with no local sub-agent loop to run.
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import asyncio
|
|
34
|
+
import ipaddress
|
|
35
|
+
import os
|
|
36
|
+
import socket
|
|
37
|
+
from urllib.parse import parse_qs, urlparse
|
|
38
|
+
|
|
39
|
+
from rockycode.engine.tools import Tool, _fn_schema, _truncate
|
|
40
|
+
from rockycode.onboarding import KEY_ENV, current_key, require_base_url
|
|
41
|
+
|
|
42
|
+
SEARCH_MAX_CHARS = 8_000
|
|
43
|
+
ANTHROPIC_VERSION = "2023-06-01"
|
|
44
|
+
SEARCH_TOOL_VARIANTS = ("web_search_20260209", "web_search_20250305")
|
|
45
|
+
RESEARCH_CONCURRENCY = 4 # polite cap on parallel lookups
|
|
46
|
+
|
|
47
|
+
# Honest, transparent UA — we say who we are rather than impersonate a browser.
|
|
48
|
+
POLITE_UA = "rockycode/0.1 (+https://github.com/cicialgo/rockycode; coding agent web tools)"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
# ---- config -----------------------------------------------------------------
|
|
52
|
+
|
|
53
|
+
def _anthropic_messages_url() -> str:
|
|
54
|
+
explicit = os.getenv("ROCKYCODE_ANTHROPIC_BASE")
|
|
55
|
+
if explicit:
|
|
56
|
+
base = explicit.rstrip("/")
|
|
57
|
+
else:
|
|
58
|
+
v1 = require_base_url().rstrip("/")
|
|
59
|
+
root = v1[:-3].rstrip("/") if v1.endswith("/v1") else v1
|
|
60
|
+
base = root + "/anthropic"
|
|
61
|
+
return base + "/v1/messages"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _search_model() -> str:
|
|
65
|
+
return os.getenv("ROCKYCODE_SEARCH_MODEL", "deepseek-v4-flash")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def default_search_order() -> tuple[str, ...]:
|
|
69
|
+
env = os.getenv("ROCKYCODE_SEARCH_ORDER")
|
|
70
|
+
if env:
|
|
71
|
+
return tuple(x.strip() for x in env.split(",") if x.strip())
|
|
72
|
+
order = ["native"]
|
|
73
|
+
# Brave only if the user brought their own key — never a proxied/shared one.
|
|
74
|
+
if os.getenv("BRAVE_API_KEY"):
|
|
75
|
+
order.append("brave")
|
|
76
|
+
# Bing before DuckDuckGo: bing.com is reachable from more restricted
|
|
77
|
+
# networks (incl. cn.bing.com); DuckDuckGo is blocked in some regions.
|
|
78
|
+
order += ["bing", "duckduckgo"]
|
|
79
|
+
return tuple(order)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# ---- SSRF guard -------------------------------------------------------------
|
|
83
|
+
|
|
84
|
+
def _resolve_ips(host: str) -> list[str]:
|
|
85
|
+
return [info[4][0] for info in socket.getaddrinfo(host, None)]
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
async def _assert_safe_url(url: str) -> list[str]:
|
|
89
|
+
"""Block non-http(s) and hosts resolving to private/loopback/link-local/
|
|
90
|
+
reserved addresses (incl. 169.254.169.254 cloud metadata). Returns the
|
|
91
|
+
VALIDATED IPs so the caller can connect to one of them directly — closing
|
|
92
|
+
the DNS-rebind TOCTOU where httpx would otherwise re-resolve the hostname to
|
|
93
|
+
a fresh (internal) address between this check and the socket connect."""
|
|
94
|
+
p = urlparse(url)
|
|
95
|
+
if p.scheme not in ("http", "https"):
|
|
96
|
+
raise ValueError(f"refusing non-http(s) url: {p.scheme or '?'}")
|
|
97
|
+
if not p.hostname:
|
|
98
|
+
raise ValueError("url has no host")
|
|
99
|
+
try:
|
|
100
|
+
ips = await asyncio.wait_for(asyncio.to_thread(_resolve_ips, p.hostname), timeout=5)
|
|
101
|
+
except asyncio.TimeoutError:
|
|
102
|
+
raise ValueError(f"dns timeout for host: {p.hostname}")
|
|
103
|
+
except socket.gaierror as e:
|
|
104
|
+
raise ValueError(f"cannot resolve host: {e}")
|
|
105
|
+
if not ips:
|
|
106
|
+
raise ValueError(f"no addresses for host: {p.hostname}")
|
|
107
|
+
for raw in ips:
|
|
108
|
+
ip = ipaddress.ip_address(raw)
|
|
109
|
+
if ip.is_private or ip.is_loopback or ip.is_link_local or ip.is_reserved or ip.is_multicast:
|
|
110
|
+
raise ValueError(f"refusing to fetch internal address {ip} ({p.hostname})")
|
|
111
|
+
return ips
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# ---- formatting -------------------------------------------------------------
|
|
115
|
+
|
|
116
|
+
def _format_native(text: str, sources: list[dict]) -> str:
|
|
117
|
+
if sources:
|
|
118
|
+
lines = "\n".join(f"- {s.get('title','')} — {s.get('url','')}" for s in sources[:8])
|
|
119
|
+
text = f"{text}\n\nsources:\n{lines}"
|
|
120
|
+
return _truncate(text, SEARCH_MAX_CHARS)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _format_items(items: list[dict]) -> str:
|
|
124
|
+
if not items:
|
|
125
|
+
return ""
|
|
126
|
+
blocks = [
|
|
127
|
+
f"{i+1}. {it.get('title','')}\n {it.get('url','')}\n {it.get('snippet','')}"
|
|
128
|
+
for i, it in enumerate(items)
|
|
129
|
+
]
|
|
130
|
+
return _truncate("\n".join(blocks), SEARCH_MAX_CHARS)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
# ---- backends (each: async query -> formatted str, or raises) ---------------
|
|
134
|
+
|
|
135
|
+
def _normalize_anthropic_usage(u: dict) -> dict:
|
|
136
|
+
"""Anthropic usage → the {prompt_tokens, prompt_cache_hit_tokens,
|
|
137
|
+
completion_tokens} shape the ledger/pricing expects."""
|
|
138
|
+
hit = u.get("cache_read_input_tokens", 0) or 0
|
|
139
|
+
create = u.get("cache_creation_input_tokens", 0) or 0
|
|
140
|
+
inp = u.get("input_tokens", 0) or 0 # anthropic excludes cached from this
|
|
141
|
+
return {
|
|
142
|
+
"prompt_tokens": inp + hit + create,
|
|
143
|
+
"prompt_cache_hit_tokens": hit,
|
|
144
|
+
"completion_tokens": u.get("output_tokens", 0) or 0,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
async def _native(query: str, ledger=None) -> str:
|
|
149
|
+
import httpx
|
|
150
|
+
|
|
151
|
+
key = current_key()
|
|
152
|
+
if not key:
|
|
153
|
+
raise RuntimeError(f"no API key for native search — set {KEY_ENV}")
|
|
154
|
+
headers = {"x-api-key": key, "anthropic-version": ANTHROPIC_VERSION, "content-type": "application/json"}
|
|
155
|
+
prompt = f"Search the web and answer concisely with key facts: {query}\nCite the sources you used."
|
|
156
|
+
async with httpx.AsyncClient(timeout=120) as client:
|
|
157
|
+
last = None
|
|
158
|
+
for variant in SEARCH_TOOL_VARIANTS:
|
|
159
|
+
r = await client.post(_anthropic_messages_url(), headers=headers, json={
|
|
160
|
+
"model": _search_model(), "max_tokens": 2048,
|
|
161
|
+
"tools": [{"type": variant, "name": "web_search", "max_uses": 5}],
|
|
162
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
163
|
+
})
|
|
164
|
+
if r.status_code == 400 and "unknown variant" in r.text:
|
|
165
|
+
last = r.text
|
|
166
|
+
continue
|
|
167
|
+
r.raise_for_status()
|
|
168
|
+
data = r.json()
|
|
169
|
+
# flash search tokens count toward the session cost too
|
|
170
|
+
if ledger is not None and isinstance(data.get("usage"), dict):
|
|
171
|
+
ledger.add(_search_model(), _normalize_anthropic_usage(data["usage"]))
|
|
172
|
+
text = "".join(b.get("text", "") for b in data.get("content", []) if b.get("type") == "text")
|
|
173
|
+
sources = [
|
|
174
|
+
{"title": it.get("title", ""), "url": it["url"]}
|
|
175
|
+
for b in data.get("content", []) if b.get("type") == "web_search_tool_result"
|
|
176
|
+
for it in (b.get("content") or []) if isinstance(it, dict) and it.get("url")
|
|
177
|
+
]
|
|
178
|
+
return _format_native(text.strip(), sources)
|
|
179
|
+
raise RuntimeError(f"no supported web_search variant: {last}")
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
async def _brave(query: str) -> str:
|
|
183
|
+
import httpx
|
|
184
|
+
|
|
185
|
+
key = os.getenv("BRAVE_API_KEY")
|
|
186
|
+
if not key:
|
|
187
|
+
raise RuntimeError("no BRAVE_API_KEY")
|
|
188
|
+
async with httpx.AsyncClient(timeout=20) as client:
|
|
189
|
+
r = await client.get(
|
|
190
|
+
"https://api.search.brave.com/res/v1/web/search",
|
|
191
|
+
params={"q": query, "count": 8},
|
|
192
|
+
headers={"X-Subscription-Token": key, "Accept": "application/json"},
|
|
193
|
+
)
|
|
194
|
+
r.raise_for_status()
|
|
195
|
+
results = (r.json().get("web") or {}).get("results", []) or []
|
|
196
|
+
return _format_items([
|
|
197
|
+
{"title": x.get("title", ""), "url": x.get("url", ""), "snippet": x.get("description", "")}
|
|
198
|
+
for x in results
|
|
199
|
+
])
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _ddg_real_url(href: str) -> str:
|
|
203
|
+
# DDG html wraps links as /l/?uddg=<encoded real url>
|
|
204
|
+
if "duckduckgo.com/l/" in href or href.startswith("//duckduckgo.com/l/"):
|
|
205
|
+
q = parse_qs(urlparse(href).query).get("uddg")
|
|
206
|
+
if q:
|
|
207
|
+
return q[0]
|
|
208
|
+
return href
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
async def _duckduckgo(query: str) -> str:
|
|
212
|
+
import httpx
|
|
213
|
+
from bs4 import BeautifulSoup
|
|
214
|
+
|
|
215
|
+
async with httpx.AsyncClient(timeout=20, follow_redirects=True) as client:
|
|
216
|
+
r = await client.post(
|
|
217
|
+
"https://html.duckduckgo.com/html/",
|
|
218
|
+
data={"q": query}, headers={"User-Agent": POLITE_UA},
|
|
219
|
+
)
|
|
220
|
+
soup = BeautifulSoup(r.text, "html.parser")
|
|
221
|
+
items = []
|
|
222
|
+
for res in soup.select(".result")[:8]:
|
|
223
|
+
a = res.select_one("a.result__a")
|
|
224
|
+
if not a or not a.get("href"):
|
|
225
|
+
continue
|
|
226
|
+
sn = res.select_one(".result__snippet")
|
|
227
|
+
items.append({"title": a.get_text(strip=True), "url": _ddg_real_url(a["href"]),
|
|
228
|
+
"snippet": sn.get_text(strip=True) if sn else ""})
|
|
229
|
+
return _format_items(items)
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
async def _bing(query: str) -> str:
|
|
233
|
+
import httpx
|
|
234
|
+
from bs4 import BeautifulSoup
|
|
235
|
+
|
|
236
|
+
async with httpx.AsyncClient(timeout=20, follow_redirects=True) as client:
|
|
237
|
+
r = await client.get("https://www.bing.com/search", params={"q": query},
|
|
238
|
+
headers={"User-Agent": POLITE_UA})
|
|
239
|
+
soup = BeautifulSoup(r.text, "html.parser")
|
|
240
|
+
items = []
|
|
241
|
+
for li in soup.select("#b_results > li.b_algo")[:8]:
|
|
242
|
+
a = li.select_one("h2 > a") or li.select_one("a.tilk")
|
|
243
|
+
if not a or not a.get("href"):
|
|
244
|
+
continue
|
|
245
|
+
title = a.get("aria-label") or a.get_text(strip=True)
|
|
246
|
+
p = li.select_one('p[class^="b_lineclamp"]') or li.select_one(".b_caption p")
|
|
247
|
+
items.append({"title": title, "url": a["href"], "snippet": p.get_text(strip=True) if p else ""})
|
|
248
|
+
return _format_items(items)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
DEFAULT_BACKENDS = {"native": _native, "bing": _bing, "duckduckgo": _duckduckgo, "brave": _brave}
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
FETCH_MAX_CHARS = 20_000 # cap fetched page text so one page can't flood context
|
|
255
|
+
FETCH_MAX_BYTES = 8_000_000 # ~8 MB hard cap on the raw body — OOM guard
|
|
256
|
+
MAX_REDIRECTS = 5
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _pinned_request(client, url: str, ip: str):
|
|
260
|
+
"""Build a GET that connects to the already-validated *ip* while presenting
|
|
261
|
+
the real hostname (Host header + TLS SNI/cert via the sni_hostname
|
|
262
|
+
extension). This defeats DNS rebinding: httpx never re-resolves the name, so
|
|
263
|
+
it can't be pointed at an internal address after our check."""
|
|
264
|
+
import httpx
|
|
265
|
+
|
|
266
|
+
u = httpx.URL(url)
|
|
267
|
+
host_header = u.host if u.port is None else f"{u.host}:{u.port}"
|
|
268
|
+
return client.build_request(
|
|
269
|
+
"GET",
|
|
270
|
+
u.copy_with(host=ip),
|
|
271
|
+
headers={"User-Agent": POLITE_UA, "Host": host_header},
|
|
272
|
+
extensions={"sni_hostname": u.host},
|
|
273
|
+
)
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def _too_large(content_length: str) -> bool:
|
|
277
|
+
"""True if a declared Content-Length already exceeds the byte cap — lets us
|
|
278
|
+
reject an oversized body up front, before reading a single chunk."""
|
|
279
|
+
return content_length.isdigit() and int(content_length) > FETCH_MAX_BYTES
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
async def _read_capped(response, cap: int) -> str:
|
|
283
|
+
"""Stream the body, stopping once *cap* bytes are read (so a hostile or huge
|
|
284
|
+
response can't exhaust memory), then decode."""
|
|
285
|
+
chunks: list[bytes] = []
|
|
286
|
+
total = 0
|
|
287
|
+
async for chunk in response.aiter_bytes():
|
|
288
|
+
if total + len(chunk) > cap:
|
|
289
|
+
chunks.append(chunk[: cap - total])
|
|
290
|
+
break
|
|
291
|
+
chunks.append(chunk)
|
|
292
|
+
total += len(chunk)
|
|
293
|
+
return b"".join(chunks).decode(response.encoding or "utf-8", errors="replace")
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
async def _web_fetch(url: str, *, _transport=None) -> str:
|
|
297
|
+
import httpx
|
|
298
|
+
from bs4 import BeautifulSoup
|
|
299
|
+
|
|
300
|
+
ips = await _assert_safe_url(url) # resolve+validate once; pin the result
|
|
301
|
+
# follow_redirects=False + a manual loop so EVERY hop is re-validated AND
|
|
302
|
+
# re-pinned: the SSRF guard on the first url is worthless if a 302 can then
|
|
303
|
+
# point at 169.254.169.254 / 127.0.0.1 / a LAN host. httpx would follow blindly.
|
|
304
|
+
client_kwargs = dict(timeout=20, follow_redirects=False)
|
|
305
|
+
if _transport is not None: # tests inject httpx.MockTransport
|
|
306
|
+
client_kwargs["transport"] = _transport
|
|
307
|
+
async with httpx.AsyncClient(**client_kwargs) as client:
|
|
308
|
+
r = None
|
|
309
|
+
for _ in range(MAX_REDIRECTS):
|
|
310
|
+
r = await client.send(_pinned_request(client, url, ips[0]), stream=True)
|
|
311
|
+
if not r.is_redirect:
|
|
312
|
+
break
|
|
313
|
+
loc = r.headers.get("location")
|
|
314
|
+
if not loc:
|
|
315
|
+
break
|
|
316
|
+
await r.aclose() # discard redirect body before following
|
|
317
|
+
url = str(httpx.URL(url).join(loc)) # resolve relative Location
|
|
318
|
+
ips = await _assert_safe_url(url) # re-validate + re-pin BEFORE following
|
|
319
|
+
else:
|
|
320
|
+
if r is not None:
|
|
321
|
+
await r.aclose()
|
|
322
|
+
return "[error] too many redirects"
|
|
323
|
+
|
|
324
|
+
try:
|
|
325
|
+
clen = r.headers.get("content-length", "")
|
|
326
|
+
if _too_large(clen):
|
|
327
|
+
return f"[error] response too large ({int(clen):,} bytes; cap {FETCH_MAX_BYTES:,})"
|
|
328
|
+
ctype = r.headers.get("content-type", "").lower()
|
|
329
|
+
body = await _read_capped(r, FETCH_MAX_BYTES)
|
|
330
|
+
finally:
|
|
331
|
+
await r.aclose()
|
|
332
|
+
|
|
333
|
+
if "html" not in ctype:
|
|
334
|
+
return _truncate(body, FETCH_MAX_CHARS)
|
|
335
|
+
soup = BeautifulSoup(body, "html.parser")
|
|
336
|
+
for tag in soup(["script", "style", "noscript", "header", "footer", "nav", "form"]):
|
|
337
|
+
tag.decompose()
|
|
338
|
+
return _truncate(soup.get_text("\n", strip=True), FETCH_MAX_CHARS)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
# ---- tool factory -----------------------------------------------------------
|
|
342
|
+
|
|
343
|
+
def build_web_tools(
|
|
344
|
+
*,
|
|
345
|
+
ledger=None,
|
|
346
|
+
search_order: tuple[str, ...] | None = None,
|
|
347
|
+
backends: dict | None = None,
|
|
348
|
+
fetch_fn=_web_fetch,
|
|
349
|
+
) -> dict[str, Tool]:
|
|
350
|
+
"""The three web tools. backends/fetch are injectable so tests run offline.
|
|
351
|
+
`ledger` (a pricing.UsageLedger) captures flash search-token usage so it
|
|
352
|
+
counts toward the session cost."""
|
|
353
|
+
order = search_order or default_search_order()
|
|
354
|
+
# native search reports its flash usage into the ledger
|
|
355
|
+
bk = backends or {
|
|
356
|
+
"native": lambda q: _native(q, ledger=ledger),
|
|
357
|
+
"bing": _bing, "duckduckgo": _duckduckgo, "brave": _brave,
|
|
358
|
+
}
|
|
359
|
+
sem = asyncio.Semaphore(RESEARCH_CONCURRENCY)
|
|
360
|
+
|
|
361
|
+
def _today() -> str:
|
|
362
|
+
from datetime import date
|
|
363
|
+
return date.today().isoformat()
|
|
364
|
+
|
|
365
|
+
async def web_search(query: str) -> str:
|
|
366
|
+
errors = []
|
|
367
|
+
for name in order:
|
|
368
|
+
fn = bk.get(name)
|
|
369
|
+
if fn is None:
|
|
370
|
+
continue
|
|
371
|
+
try:
|
|
372
|
+
result = await fn(query)
|
|
373
|
+
if result and result.strip():
|
|
374
|
+
# the date belongs at the point of recency judgment —
|
|
375
|
+
# and it stays true even when a session crosses midnight
|
|
376
|
+
return f"[searched {_today()}] {result}"
|
|
377
|
+
except Exception as e: # noqa: BLE001 — try next backend
|
|
378
|
+
errors.append(f"{name}: {type(e).__name__}: {e}")
|
|
379
|
+
return f"[error] all search backends failed — {' | '.join(errors) or 'no results'}"
|
|
380
|
+
|
|
381
|
+
async def web_research(queries: list[str]) -> str:
|
|
382
|
+
if not isinstance(queries, list) or not queries:
|
|
383
|
+
return "[error] web_research needs a non-empty list of queries"
|
|
384
|
+
|
|
385
|
+
async def one(q):
|
|
386
|
+
async with sem: # polite: cap concurrent lookups
|
|
387
|
+
return await web_search(q)
|
|
388
|
+
|
|
389
|
+
results = await asyncio.gather(*(one(q) for q in queries[:8]), return_exceptions=True)
|
|
390
|
+
out = [f"### {q}\n{('[error] ' + str(r)) if isinstance(r, Exception) else r}"
|
|
391
|
+
for q, r in zip(queries, results)]
|
|
392
|
+
return _truncate("\n\n".join(out), SEARCH_MAX_CHARS * 2)
|
|
393
|
+
|
|
394
|
+
async def web_fetch(url: str) -> str:
|
|
395
|
+
try:
|
|
396
|
+
return f"[fetched {_today()}] {await fetch_fn(url)}"
|
|
397
|
+
except Exception as e: # noqa: BLE001 — model-readable
|
|
398
|
+
return f"[error] fetch failed: {type(e).__name__}: {e}"
|
|
399
|
+
|
|
400
|
+
schemas = {
|
|
401
|
+
"web_search": _fn_schema(
|
|
402
|
+
"web_search",
|
|
403
|
+
"Search the web for current information; returns a concise answer with sources. "
|
|
404
|
+
"Runs server-side on DeepSeek (works behind restrictive networks).",
|
|
405
|
+
{"query": {"type": "string", "description": "What to search for."}},
|
|
406
|
+
["query"],
|
|
407
|
+
),
|
|
408
|
+
"web_research": _fn_schema(
|
|
409
|
+
"web_research",
|
|
410
|
+
"Search several independent questions at once, in parallel. Faster than "
|
|
411
|
+
"sequential web_search calls when a task needs multiple lookups.",
|
|
412
|
+
{"queries": {"type": "array", "items": {"type": "string"},
|
|
413
|
+
"description": "Independent search queries (max 8)."}},
|
|
414
|
+
["queries"],
|
|
415
|
+
),
|
|
416
|
+
"web_fetch": _fn_schema(
|
|
417
|
+
"web_fetch",
|
|
418
|
+
"Fetch one URL and return its readable text. Runs on this machine, so it may "
|
|
419
|
+
"fail on sites your network blocks; prefer web_search for general info.",
|
|
420
|
+
{"url": {"type": "string", "description": "The http(s) URL to fetch."}},
|
|
421
|
+
["url"],
|
|
422
|
+
),
|
|
423
|
+
}
|
|
424
|
+
fns = {"web_search": web_search, "web_research": web_research, "web_fetch": web_fetch}
|
|
425
|
+
# search/web_research run server-side on DeepSeek (read-only); web_fetch hits the
|
|
426
|
+
# local network from this machine (SSRF-guarded) — gate it like other risky I/O.
|
|
427
|
+
risk = {"web_search": "safe", "web_research": "safe", "web_fetch": "risky"}
|
|
428
|
+
return {
|
|
429
|
+
name: Tool(name=name, schema=schemas[name], fn=fns[name], risk=risk[name])
|
|
430
|
+
for name in fns
|
|
431
|
+
}
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Isolated workspace for a goal run: a git worktree on a fresh branch, so the
|
|
2
|
+
agent works on a COPY of the repo. Nothing it does — even a destructive command
|
|
3
|
+
that slips past the safety classifier — touches the user's working tree. The
|
|
4
|
+
morning review is `git diff` / merging the goal branch.
|
|
5
|
+
|
|
6
|
+
Falls back to a plain directory copy when the target isn't a git repo, so goal
|
|
7
|
+
mode still isolates the real files, just without the nice diff/merge.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import shutil
|
|
12
|
+
import subprocess
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Optional
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _git(repo: Path, *args: str) -> subprocess.CompletedProcess:
|
|
19
|
+
return subprocess.run(["git", "-C", str(repo), *args], capture_output=True, text=True)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
# Build/download detritus a goal milestone can drop into the workspace (a pip
|
|
23
|
+
# bootstrap in a slim sandbox pulls .deb files + a venv; npm pulls node_modules).
|
|
24
|
+
# Excluded from the commit so the review branch stays reviewable — the actual
|
|
25
|
+
# source changes aren't buried under megabytes of binaries.
|
|
26
|
+
_COMMIT_EXCLUDE = [
|
|
27
|
+
":(exclude)*.deb", ":(exclude)*.whl", ":(exclude)*.pyc", ":(exclude)*.egg-info",
|
|
28
|
+
":(exclude)venv/", ":(exclude).venv/", ":(exclude)env/",
|
|
29
|
+
":(exclude)node_modules/", ":(exclude)__pycache__/",
|
|
30
|
+
":(exclude)dist/", ":(exclude)build/", ":(exclude).mypy_cache/", ":(exclude).pytest_cache/",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _is_git_repo(path: Path) -> bool:
|
|
35
|
+
return _git(path, "rev-parse", "--is-inside-work-tree").returncode == 0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class GoalWorkspace:
|
|
40
|
+
path: Path # where the goal agent works (the isolated copy)
|
|
41
|
+
branch: Optional[str] # goal branch (git case) or None (plain-copy case)
|
|
42
|
+
origin: Path # the user's real repo
|
|
43
|
+
_worktree: bool
|
|
44
|
+
base: Optional[str] = None # fork-point commit SHA (git case), for diff/review
|
|
45
|
+
|
|
46
|
+
@classmethod
|
|
47
|
+
def create(cls, repo: Path, slug: str) -> "GoalWorkspace":
|
|
48
|
+
repo = Path(repo).resolve()
|
|
49
|
+
dest = repo.parent / f".rockycode-goal-{slug}"
|
|
50
|
+
if dest.exists():
|
|
51
|
+
raise FileExistsError(f"goal workspace already exists: {dest}")
|
|
52
|
+
if _is_git_repo(repo):
|
|
53
|
+
base = _git(repo, "rev-parse", "HEAD").stdout.strip() or "HEAD"
|
|
54
|
+
branch = f"goal/{slug}"
|
|
55
|
+
r = _git(repo, "worktree", "add", "-b", branch, str(dest), "HEAD")
|
|
56
|
+
if r.returncode != 0:
|
|
57
|
+
raise RuntimeError(f"git worktree add failed: {r.stderr.strip()}")
|
|
58
|
+
return cls(dest, branch, repo, True, base)
|
|
59
|
+
# not a git repo → plain copy (still isolates the real files)
|
|
60
|
+
shutil.copytree(
|
|
61
|
+
repo, dest,
|
|
62
|
+
ignore=shutil.ignore_patterns(".git", "node_modules", ".venv", "__pycache__"),
|
|
63
|
+
)
|
|
64
|
+
return cls(dest, None, repo, False, None)
|
|
65
|
+
|
|
66
|
+
def commit(self, message: str) -> bool:
|
|
67
|
+
"""Checkpoint the current worktree state onto the goal branch. Host-side
|
|
68
|
+
and scoped to THIS worktree; no-op for a plain-copy workspace or when
|
|
69
|
+
nothing changed. Returns True iff a commit was made.
|
|
70
|
+
|
|
71
|
+
Refuses to run unless the worktree's HEAD is the goal branch — a paranoia
|
|
72
|
+
guard so a goal run can NEVER advance the user's own branch (worktree
|
|
73
|
+
semantics already guarantee this; the check makes a surprise fail loud)."""
|
|
74
|
+
if not self._worktree:
|
|
75
|
+
return False
|
|
76
|
+
head = _git(self.path, "rev-parse", "--abbrev-ref", "HEAD").stdout.strip()
|
|
77
|
+
if head != self.branch:
|
|
78
|
+
raise RuntimeError(
|
|
79
|
+
f"refusing to commit: {self.path} is on '{head}', not the goal "
|
|
80
|
+
f"branch '{self.branch}'")
|
|
81
|
+
_git(self.path, "add", "-A", "--", ".", *_COMMIT_EXCLUDE)
|
|
82
|
+
if _git(self.path, "diff", "--cached", "--quiet").returncode == 0:
|
|
83
|
+
return False # nothing changed this milestone — no empty commit
|
|
84
|
+
# Inline identity: an unattended env may have no global git user set.
|
|
85
|
+
r = _git(self.path, "-c", "user.name=rockycode",
|
|
86
|
+
"-c", "user.email=goal@rockycode.local", "commit", "-m", message)
|
|
87
|
+
return r.returncode == 0
|
|
88
|
+
|
|
89
|
+
def diff(self) -> str:
|
|
90
|
+
"""The full change the goal made (committed + uncommitted), for review."""
|
|
91
|
+
if self._worktree:
|
|
92
|
+
return _git(self.path, "diff", self.base or "HEAD").stdout
|
|
93
|
+
return "[copy workspace — no git diff; compare the directory manually]"
|
|
94
|
+
|
|
95
|
+
def cleanup(self, *, keep: bool = True) -> None:
|
|
96
|
+
"""keep=True (default): leave the branch/worktree so you can review and
|
|
97
|
+
merge in the morning. keep=False: remove the worktree (the branch stays,
|
|
98
|
+
in the git case, so committed work survives)."""
|
|
99
|
+
if keep:
|
|
100
|
+
return
|
|
101
|
+
if self._worktree:
|
|
102
|
+
_git(self.origin, "worktree", "remove", "--force", str(self.path))
|
|
103
|
+
elif self.path.exists():
|
|
104
|
+
shutil.rmtree(self.path, ignore_errors=True)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def prune_goal_worktrees(repo: Path) -> list[str]:
|
|
108
|
+
"""Remove leftover goal worktrees (`.rockycode-goal-*`) registered on *repo*
|
|
109
|
+
— they accumulate one per run. Branches are KEPT (committed work survives on
|
|
110
|
+
`goal/<slug>`, reviewable/mergeable). Returns the paths removed. Also drops
|
|
111
|
+
plain-copy leftovers next to the repo. Best-effort; never raises."""
|
|
112
|
+
repo = Path(repo).resolve()
|
|
113
|
+
removed: list[str] = []
|
|
114
|
+
if _is_git_repo(repo):
|
|
115
|
+
for line in _git(repo, "worktree", "list", "--porcelain").stdout.splitlines():
|
|
116
|
+
if not line.startswith("worktree "):
|
|
117
|
+
continue
|
|
118
|
+
path = line[len("worktree "):].strip()
|
|
119
|
+
if Path(path).name.startswith(".rockycode-goal-"):
|
|
120
|
+
if _git(repo, "worktree", "remove", "--force", path).returncode == 0:
|
|
121
|
+
removed.append(path)
|
|
122
|
+
_git(repo, "worktree", "prune") # clear stale admin entries
|
|
123
|
+
# plain-copy fallback workspaces (non-git runs) live beside the repo too
|
|
124
|
+
for d in repo.parent.glob(".rockycode-goal-*"):
|
|
125
|
+
if d.is_dir() and str(d) not in removed:
|
|
126
|
+
shutil.rmtree(d, ignore_errors=True)
|
|
127
|
+
removed.append(str(d))
|
|
128
|
+
return removed
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""rockycode memory: multi-level, user-inspectable, files-as-truth.
|
|
2
|
+
|
|
3
|
+
M0 of docs/memory-dream.md — store, prompt injection, recall/remember tools.
|
|
4
|
+
"""
|
|
5
|
+
from rockycode.memory.store import Memory, MemoryStore, build_memory_tools, memory_prompt_section
|
|
6
|
+
|
|
7
|
+
__all__ = ["Memory", "MemoryStore", "build_memory_tools", "memory_prompt_section"]
|