rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,592 @@
|
|
|
1
|
+
"""Explore: buy verified findings from a bounded, read-only, fresh-context run.
|
|
2
|
+
|
|
3
|
+
The primitive underneath rocky's delegation features. A parent (the chat model,
|
|
4
|
+
or the goal verify/review path) purchases an investigation at a fixed price —
|
|
5
|
+
capped steps, capped wall-clock, its own context — and receives ONLY the final
|
|
6
|
+
report; the child's search noise (greps, file dumps, dead ends) never enters
|
|
7
|
+
the parent's history or its prompt-cache path.
|
|
8
|
+
|
|
9
|
+
What makes the report trustworthy is the FINDINGS CONTRACT: claims must cite
|
|
10
|
+
evidence as `path:line "anchor text"`, and check_citations() mechanically
|
|
11
|
+
re-verifies every citation against the tree (or a git ref for branch reviews)
|
|
12
|
+
before the parent sees the report. Hallucinated evidence is flagged inline —
|
|
13
|
+
a prompt rule in other harnesses, a checker here. The same footer doubles as
|
|
14
|
+
an automatic grounding signal on each logged episode (source: "explore").
|
|
15
|
+
|
|
16
|
+
Safety model (mechanisms, not prompt text):
|
|
17
|
+
- The child's registry is BUILT read-only: the safe-tier subset of the normal
|
|
18
|
+
registry (read_file / grep / glob / check_code) plus a hard-gated read-only
|
|
19
|
+
bash (allowlisted binaries, relative paths only, no shell metacharacters —
|
|
20
|
+
see _ro_bash_check). No write_file, no edit_file.
|
|
21
|
+
- No `explore` tool in the child registry → children are leaf workers. Depth
|
|
22
|
+
cap of 1 by construction; no recursion counter to get wrong.
|
|
23
|
+
- The parent-facing tool is risk="safe", which is a CONTRACT with loop.py's
|
|
24
|
+
batch rule: a batch is concurrent only if every call is safe-tier. Because
|
|
25
|
+
a child mutates nothing, several explore calls in one assistant turn fan
|
|
26
|
+
out in parallel through the existing gather path — and any future WRITER
|
|
27
|
+
role must register as "risky" so the same rule serializes it with all
|
|
28
|
+
other mutations. Parallel writers stay impossible by construction.
|
|
29
|
+
- Children inherit the parent's workdir/allowed_roots jail, and their tool
|
|
30
|
+
outputs (and the report returned to the parent) pass through the normal
|
|
31
|
+
execute() redaction.
|
|
32
|
+
|
|
33
|
+
DeepSeek economics: each role's system prompt is a byte-stable constant and
|
|
34
|
+
the child toolset is fixed, so every explore of a role shares the same prompt
|
|
35
|
+
prefix — repeated purchases in a session hit the prefix cache instead of
|
|
36
|
+
re-paying the parent's ever-growing history.
|
|
37
|
+
"""
|
|
38
|
+
from __future__ import annotations
|
|
39
|
+
|
|
40
|
+
import asyncio
|
|
41
|
+
import os
|
|
42
|
+
import re
|
|
43
|
+
import shlex
|
|
44
|
+
import time
|
|
45
|
+
from pathlib import Path
|
|
46
|
+
from typing import Optional
|
|
47
|
+
|
|
48
|
+
from rockycode.engine import tools as tools_mod
|
|
49
|
+
from rockycode.engine.events import TurnFinished
|
|
50
|
+
from rockycode.engine.tools import Tool
|
|
51
|
+
|
|
52
|
+
EXPLORE_MAX_STEPS = 20 # parent runs 50; a focused purchase should not need more
|
|
53
|
+
EXPLORE_FINALIZE_STEPS = 2
|
|
54
|
+
EXPLORE_EFFORT = "high" # parent default is "max"; delegated lookups don't need it
|
|
55
|
+
EXPLORE_TIMEOUT_S = int(os.getenv("ROCKYCODE_EXPLORE_TIMEOUT", "600"))
|
|
56
|
+
CITATION_CAP = 30 # citations checked per report — bounds checker cost
|
|
57
|
+
|
|
58
|
+
# ---------------------------------------------------------------------------
|
|
59
|
+
# Role prompts — BYTE-STABLE constants (task/context go in the user message,
|
|
60
|
+
# never in here) so every explore of a role reuses the same cached prefix.
|
|
61
|
+
# ---------------------------------------------------------------------------
|
|
62
|
+
|
|
63
|
+
_OUTPUT_CONTRACT = """
|
|
64
|
+
Report format — your final message is ALL the buyer will see, so it must stand
|
|
65
|
+
alone:
|
|
66
|
+
- FINDINGS: what you established. Ground every claim in a citation.
|
|
67
|
+
- EVIDENCE: one bullet per citation, EXACTLY this shape (straight quotes):
|
|
68
|
+
- path/to/file.py:123 "text copied verbatim from that line"
|
|
69
|
+
ONE line number (never a range); the path written from the REPO ROOT
|
|
70
|
+
(rockycode/engine/loop.py, never just loop.py); nothing between the path
|
|
71
|
+
and the colon — no backticks, no "(branch)" notes. When you reviewed a
|
|
72
|
+
branch or ref, cite the same plain way: the harness checks against the ref
|
|
73
|
+
you were given.
|
|
74
|
+
The harness re-reads every citation and flags any it cannot verify — an
|
|
75
|
+
unverifiable citation is worse than none, so quote real lines only.
|
|
76
|
+
- GAPS: what you could not determine, and which searches you ran that came up
|
|
77
|
+
empty (a negative claim without the searches behind it is worthless).
|
|
78
|
+
Be complete but not padded; the report replaces your whole transcript."""
|
|
79
|
+
|
|
80
|
+
EXPLORE_PROMPT = """You are rocky's explore agent: a read-only investigator \
|
|
81
|
+
answering one focused question about a codebase. You work in a fresh context; \
|
|
82
|
+
the agent that bought this investigation sees none of your tool calls — only \
|
|
83
|
+
your final report.
|
|
84
|
+
|
|
85
|
+
Investigate thoroughly: prefer grep/glob to locate, read_file to confirm. \
|
|
86
|
+
Follow the code, not your assumptions; check more than one naming convention \
|
|
87
|
+
before concluding something does not exist. You cannot write, edit, or \
|
|
88
|
+
delegate — if the task seems to need that, report it as a finding instead.
|
|
89
|
+
""" + _OUTPUT_CONTRACT
|
|
90
|
+
|
|
91
|
+
REVIEW_PROMPT = """You are rocky's review agent: an independent, read-only \
|
|
92
|
+
second opinion on a change (a diff, branch, or set of files). You work in a \
|
|
93
|
+
fresh context on purpose — judge only what the code says, not what the author \
|
|
94
|
+
intended. The buyer sees only your final report.
|
|
95
|
+
|
|
96
|
+
Read the change AND enough surrounding code to judge it in context (callers, \
|
|
97
|
+
tests, error paths). Rank what you find by severity; a missed failure mode \
|
|
98
|
+
outranks any style point, and style points are out of scope unless asked. \
|
|
99
|
+
Confirm the good as well: say what you checked and found sound, so silence \
|
|
100
|
+
is not ambiguous. Start your report with a one-line VERDICT.
|
|
101
|
+
""" + _OUTPUT_CONTRACT
|
|
102
|
+
|
|
103
|
+
VERIFY_PROMPT = """You are rocky's verify agent: a read-only inspector deciding \
|
|
104
|
+
whether a claimed milestone is ACTUALLY complete in the working tree. You work \
|
|
105
|
+
in a fresh context on purpose — judge the code's end state, never the worker's \
|
|
106
|
+
account of it.
|
|
107
|
+
|
|
108
|
+
Inspect what changed (git status, git diff, read the touched files; run \
|
|
109
|
+
check_code if useful) and test the claim against reality. The buyer supplies \
|
|
110
|
+
baseline-vs-now check output and decision rules in the task; apply them \
|
|
111
|
+
exactly. Your report's FIRST line must be exactly "PASS — <one-line reason>" \
|
|
112
|
+
or "FAIL — <one-line reason>" — the verdict first, no hedging before it.
|
|
113
|
+
""" + _OUTPUT_CONTRACT
|
|
114
|
+
|
|
115
|
+
# "verify" is goal-internal: reachable through make_goal_verifier, deliberately
|
|
116
|
+
# absent from the chat tool's role enum (EXPLORE_SCHEMA).
|
|
117
|
+
ROLE_PROMPTS: dict[str, str] = {
|
|
118
|
+
"explore": EXPLORE_PROMPT,
|
|
119
|
+
"review": REVIEW_PROMPT,
|
|
120
|
+
"verify": VERIFY_PROMPT,
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
# ---------------------------------------------------------------------------
|
|
124
|
+
# Read-only bash gate. safety.classify_command is a DANGER classifier (an
|
|
125
|
+
# innocuous `echo x > f` passes it), so the child gets its own gate that is
|
|
126
|
+
# read-only by construction. Layered, strictest first:
|
|
127
|
+
# 1. no shell metacharacters at all: chaining, redirects (both ways — `<`
|
|
128
|
+
# would still open arbitrary files), backgrounding, substitution, `$`
|
|
129
|
+
# expansion, multi-line. Single pipes are the one composition allowed.
|
|
130
|
+
# 2. every pipe segment's binary must be on a small read-only allowlist;
|
|
131
|
+
# git additionally needs an allowlisted read-only subcommand (this also
|
|
132
|
+
# kills `git -C /elsewhere`, since `-C` is not a subcommand).
|
|
133
|
+
# 3. relative paths only — cwd is the jailed workdir, so every operand
|
|
134
|
+
# stays in-tree without parsing which args are paths. `cat` is deliberately
|
|
135
|
+
# NOT allowlisted: in-tree reads are read_file's job (jailed, line-numbered,
|
|
136
|
+
# secret-refusing).
|
|
137
|
+
# 4. tokens matching the secret-file patterns (.env, id_rsa, *.pem …) are
|
|
138
|
+
# refused, mirroring read_file. Residual risk — secrets already committed
|
|
139
|
+
# to git history via `git show` — is accepted: execute() redaction still
|
|
140
|
+
# scrubs known token shapes from the output.
|
|
141
|
+
# ---------------------------------------------------------------------------
|
|
142
|
+
|
|
143
|
+
_RO_BINS = {"ls", "wc", "head", "tail", "stat", "file", "du", "diff", "git"}
|
|
144
|
+
_RO_GIT_SUBS = {
|
|
145
|
+
"status", "log", "diff", "show", "blame", "shortlog", "describe",
|
|
146
|
+
"rev-parse", "ls-files", "grep", "branch",
|
|
147
|
+
}
|
|
148
|
+
# `git branch` mutates with these; listing stays allowed.
|
|
149
|
+
_GIT_BRANCH_MUTATING = re.compile(r"(^|\s)(-d|-D|-m|-M|-c|-C|--delete|--move|--copy|--force)(\s|$)")
|
|
150
|
+
_FORBIDDEN = re.compile(r"[;&`><$\n]|--output\b")
|
|
151
|
+
|
|
152
|
+
_RO_REFUSAL = (
|
|
153
|
+
"[blocked] explore bash is read-only: allowlisted binaries only "
|
|
154
|
+
f"({', '.join(sorted(_RO_BINS))}; git subcommands: {', '.join(sorted(_RO_GIT_SUBS))}), "
|
|
155
|
+
"relative paths, single pipes; no redirects, chaining, `$`, or secret files. "
|
|
156
|
+
"Use read_file / grep / glob for file access."
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _ro_bash_check(command: str) -> Optional[str]:
|
|
161
|
+
"""Return a refusal string, or None if *command* is read-only-safe."""
|
|
162
|
+
if _FORBIDDEN.search(command):
|
|
163
|
+
return _RO_REFUSAL
|
|
164
|
+
for segment in command.split("|"):
|
|
165
|
+
try:
|
|
166
|
+
tokens = shlex.split(segment)
|
|
167
|
+
except ValueError:
|
|
168
|
+
return _RO_REFUSAL # unbalanced quotes — refuse rather than guess
|
|
169
|
+
if not tokens:
|
|
170
|
+
return _RO_REFUSAL # empty segment (also catches `a || b` remnants)
|
|
171
|
+
head, rest = tokens[0], tokens[1:]
|
|
172
|
+
if head not in _RO_BINS: # also refuses VAR=x prefixes: '=' names no bin
|
|
173
|
+
return _RO_REFUSAL
|
|
174
|
+
if head == "git":
|
|
175
|
+
if not rest or rest[0] not in _RO_GIT_SUBS:
|
|
176
|
+
return _RO_REFUSAL
|
|
177
|
+
if rest[0] == "branch" and _GIT_BRANCH_MUTATING.search(segment):
|
|
178
|
+
return _RO_REFUSAL
|
|
179
|
+
for tok in rest:
|
|
180
|
+
if tok.startswith("-"):
|
|
181
|
+
continue
|
|
182
|
+
if tok.startswith(("/", "~")) or ".." in tok.split("/"):
|
|
183
|
+
return _RO_REFUSAL # relative, in-tree paths only
|
|
184
|
+
if tools_mod._is_secret_file(Path(tok)):
|
|
185
|
+
return _RO_REFUSAL
|
|
186
|
+
return None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
_RO_BASH_SCHEMA = tools_mod._fn_schema(
|
|
190
|
+
"bash",
|
|
191
|
+
"Run a READ-ONLY shell command in the working directory (git status/log/"
|
|
192
|
+
"diff/show/blame, ls, wc, head, tail, stat, du, diff; single pipes allowed). "
|
|
193
|
+
"Anything that writes, chains, redirects, or leaves the tree is refused.",
|
|
194
|
+
{"command": {"type": "string", "description": "The read-only command to run."}},
|
|
195
|
+
["command"],
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def build_explore_registry(
|
|
200
|
+
workdir: Path, allowed_roots: tuple[Path, ...] = ()
|
|
201
|
+
) -> dict[str, Tool]:
|
|
202
|
+
"""The child's toolset: the safe tier of the normal registry plus gated
|
|
203
|
+
read-only bash. Built FRESH (not filtered from the parent's live registry)
|
|
204
|
+
so session extras — MCP, memory, web, skills — never leak into children,
|
|
205
|
+
and the toolset stays deterministic and byte-stable per role."""
|
|
206
|
+
base = tools_mod.build_registry(workdir, allowed_roots)
|
|
207
|
+
reg = {name: t for name, t in base.items() if t.risk == "safe"}
|
|
208
|
+
|
|
209
|
+
async def _ro_bash(command: str) -> str:
|
|
210
|
+
refusal = _ro_bash_check(command)
|
|
211
|
+
if refusal:
|
|
212
|
+
return refusal
|
|
213
|
+
return await tools_mod._bash(command, workdir=workdir)
|
|
214
|
+
|
|
215
|
+
# Read-only by construction, hence safe-tier: the child's own read batches
|
|
216
|
+
# (including bash) parallelize through the same loop.py rule.
|
|
217
|
+
reg["bash"] = Tool(name="bash", schema=_RO_BASH_SCHEMA, fn=_ro_bash, risk="safe")
|
|
218
|
+
return reg # note: no `explore` tool — children cannot delegate
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
# ---------------------------------------------------------------------------
|
|
222
|
+
# The citation checker — the original core of the findings contract. Parses
|
|
223
|
+
# `path:line "anchor"` citations out of the report and re-verifies each one
|
|
224
|
+
# against the working tree (or `git show <ref>:<path>` for branch reviews).
|
|
225
|
+
# Lenient-but-honest v1: the anchor may sit within ±3 lines of the stated
|
|
226
|
+
# line (models miscount); a miss marks the citation [unverified] in a footer
|
|
227
|
+
# rather than rejecting the report — a fuzzy child must not deadlock a review.
|
|
228
|
+
# ---------------------------------------------------------------------------
|
|
229
|
+
|
|
230
|
+
# Tolerates the deviations models actually produce (seen in E2E): backticked
|
|
231
|
+
# paths (stripped before scanning), a short parenthetical between path and
|
|
232
|
+
# colon ("(branch)"), and line ranges (first line taken).
|
|
233
|
+
_CITE_RX = re.compile(
|
|
234
|
+
r'(?P<path>[A-Za-z0-9_][\w./-]*\.[A-Za-z0-9_]{1,8})'
|
|
235
|
+
r'(?:\s*\([^)\n]{1,24}\))?'
|
|
236
|
+
r':(?P<line>\d{1,6})(?:-\d{1,6})?'
|
|
237
|
+
r'(?:\s*[—–-]*\s*"(?P<snip>[^"\n]{3,160})")?'
|
|
238
|
+
)
|
|
239
|
+
_CITE_WINDOW = 3 # anchor accepted within ± this many lines of the stated line
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def _norm(s: str) -> str:
|
|
243
|
+
return " ".join(s.split())
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
async def _git_show(workdir: Path, ref: str, path: str) -> Optional[str]:
|
|
247
|
+
"""Contents of *path* on *ref*, or None. Lets citations into a not-checked-
|
|
248
|
+
out goal branch verify against what the branch actually says."""
|
|
249
|
+
try:
|
|
250
|
+
proc = await asyncio.create_subprocess_exec(
|
|
251
|
+
"git", "-C", str(workdir), "show", f"{ref}:{path}",
|
|
252
|
+
stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.DEVNULL,
|
|
253
|
+
)
|
|
254
|
+
out, _ = await proc.communicate()
|
|
255
|
+
return out.decode("utf-8", "replace") if proc.returncode == 0 else None
|
|
256
|
+
except OSError:
|
|
257
|
+
return None
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _anchor_ok(text: str, line: int, snip: Optional[str]) -> tuple[bool, str]:
|
|
261
|
+
lines = text.splitlines()
|
|
262
|
+
if snip is None:
|
|
263
|
+
# No anchor to check — existence-only (weak) verification.
|
|
264
|
+
ok = 1 <= line <= len(lines)
|
|
265
|
+
return ok, "" if ok else "line beyond end of file"
|
|
266
|
+
lo, hi = max(0, line - 1 - _CITE_WINDOW), min(len(lines), line + _CITE_WINDOW)
|
|
267
|
+
target = _norm(snip)
|
|
268
|
+
if any(target in _norm(ln) for ln in lines[lo:hi]):
|
|
269
|
+
return True, ""
|
|
270
|
+
# The check exists to catch FABRICATED evidence: a verbatim quote with a
|
|
271
|
+
# stale line number (models miscount; seen in E2E) is sloppy, not fake.
|
|
272
|
+
if any(target in _norm(ln) for ln in lines):
|
|
273
|
+
return True, ""
|
|
274
|
+
return False, "anchor not found in file"
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
_SKIP_DIRS = {".git", "node_modules", ".venv", "venv", "env", "__pycache__", "dist", "build"}
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _resolve_short_path(workdir: Path, rel: str) -> Optional[Path]:
|
|
281
|
+
"""A citation like 'loop.py:51' with the repo-root prefix missing (the most
|
|
282
|
+
common model deviation seen in E2E): accept it iff exactly ONE tree file
|
|
283
|
+
ends with the cited path — honest under-qualification, not fabrication."""
|
|
284
|
+
parts = tuple(Path(rel).parts)
|
|
285
|
+
hits: list[Path] = []
|
|
286
|
+
try:
|
|
287
|
+
for p in workdir.rglob(parts[-1]):
|
|
288
|
+
if any(s in p.parts for s in _SKIP_DIRS) or not p.is_file():
|
|
289
|
+
continue
|
|
290
|
+
if tuple(p.parts[-len(parts):]) == parts:
|
|
291
|
+
hits.append(p)
|
|
292
|
+
if len(hits) > 1:
|
|
293
|
+
return None # ambiguous — refuse to guess
|
|
294
|
+
except OSError:
|
|
295
|
+
return None
|
|
296
|
+
return hits[0] if len(hits) == 1 else None
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
async def check_citations(
|
|
300
|
+
report: str, *, workdir: Path, git_ref: Optional[str] = None
|
|
301
|
+
) -> str:
|
|
302
|
+
"""Verify every `path:line "anchor"` citation in *report*; return the
|
|
303
|
+
one-line-per-problem footer the buyer sees. Never raises."""
|
|
304
|
+
seen: dict[tuple, Optional[str]] = {}
|
|
305
|
+
report = report.replace("`", "") # backticked paths still cite
|
|
306
|
+
for m in _CITE_RX.finditer(report):
|
|
307
|
+
key = (m["path"], int(m["line"]), m["snip"])
|
|
308
|
+
if key not in seen and len(seen) < CITATION_CAP:
|
|
309
|
+
seen[key] = None
|
|
310
|
+
if not seen:
|
|
311
|
+
return "[citations: none found — treat unevidenced claims with caution]"
|
|
312
|
+
|
|
313
|
+
for path, line, snip in list(seen):
|
|
314
|
+
p, err = tools_mod._jail(path, workdir)
|
|
315
|
+
text: Optional[str] = None
|
|
316
|
+
if err is None and p is not None and not p.is_file():
|
|
317
|
+
alt = await asyncio.to_thread(_resolve_short_path, workdir, path)
|
|
318
|
+
if alt is not None:
|
|
319
|
+
p = alt
|
|
320
|
+
if err is None and p is not None and p.is_file():
|
|
321
|
+
try:
|
|
322
|
+
text = p.read_text(errors="replace")
|
|
323
|
+
except OSError:
|
|
324
|
+
text = None
|
|
325
|
+
if text is None or (git_ref and not _anchor_ok(text, line, snip)[0]):
|
|
326
|
+
# Not verifiable against the tree — try the reviewed ref's version
|
|
327
|
+
# (a citation is good if it holds on EITHER side).
|
|
328
|
+
branch_text = await _git_show(workdir, git_ref, path) if git_ref else None
|
|
329
|
+
if branch_text is not None:
|
|
330
|
+
text = branch_text
|
|
331
|
+
if err is not None:
|
|
332
|
+
seen[(path, line, snip)] = "path outside the workdir"
|
|
333
|
+
continue
|
|
334
|
+
if text is None:
|
|
335
|
+
seen[(path, line, snip)] = "file not found"
|
|
336
|
+
continue
|
|
337
|
+
ok, why = _anchor_ok(text, line, snip)
|
|
338
|
+
seen[(path, line, snip)] = None if ok else why
|
|
339
|
+
|
|
340
|
+
bad = {k: why for k, why in seen.items() if why}
|
|
341
|
+
if not bad:
|
|
342
|
+
return f"[citations: {len(seen)}/{len(seen)} verified]"
|
|
343
|
+
detail = ", ".join(f"{p}:{ln} ({why})" for (p, ln, _s), why in bad.items())
|
|
344
|
+
return f"[citations: {len(seen) - len(bad)}/{len(seen)} verified · unverified: {detail}]"
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
# ---------------------------------------------------------------------------
|
|
348
|
+
# Running a child + the buyer-facing surfaces.
|
|
349
|
+
# ---------------------------------------------------------------------------
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
def _final_answer(history: list[dict]) -> str:
|
|
353
|
+
"""Last non-empty assistant text — the child's report."""
|
|
354
|
+
for msg in reversed(history):
|
|
355
|
+
if msg.get("role") == "assistant" and (msg.get("content") or "").strip():
|
|
356
|
+
return msg["content"].strip()
|
|
357
|
+
return ""
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
async def run_explore(
|
|
361
|
+
*,
|
|
362
|
+
task: str,
|
|
363
|
+
role: str,
|
|
364
|
+
context: str = "",
|
|
365
|
+
model: str,
|
|
366
|
+
client,
|
|
367
|
+
workdir: Path,
|
|
368
|
+
allowed_roots: tuple[Path, ...] = (),
|
|
369
|
+
thinking: bool = True,
|
|
370
|
+
effort: str = EXPLORE_EFFORT,
|
|
371
|
+
max_tokens: int = 384_000,
|
|
372
|
+
ledger=None,
|
|
373
|
+
parent_session: str = "",
|
|
374
|
+
git_ref: Optional[str] = None,
|
|
375
|
+
max_steps: int = EXPLORE_MAX_STEPS,
|
|
376
|
+
engine_cls=None,
|
|
377
|
+
) -> str:
|
|
378
|
+
"""Buy one investigation: spawn a fresh child Engine on *task*, verify the
|
|
379
|
+
report's citations, and return report + citation footer + stats line.
|
|
380
|
+
|
|
381
|
+
*git_ref* lets branch-review citations verify against `git show ref:path`
|
|
382
|
+
(a goal branch is not checked out). The child logs its own trajectory
|
|
383
|
+
(source=explore, linked to the parent session) — a clean single-task
|
|
384
|
+
episode whose citation footer doubles as a grounding signal. Usage folds
|
|
385
|
+
into *ledger* so /cost stays truthful. *engine_cls* is a test seam, same
|
|
386
|
+
spirit as goal.Driver.
|
|
387
|
+
"""
|
|
388
|
+
prompt = ROLE_PROMPTS.get(role)
|
|
389
|
+
if prompt is None:
|
|
390
|
+
roles = ", ".join(sorted(ROLE_PROMPTS))
|
|
391
|
+
return f"[error] unknown explore role {role!r} — expected one of: {roles}"
|
|
392
|
+
if engine_cls is None:
|
|
393
|
+
from rockycode.engine.loop import Engine as engine_cls # avoid import cycle
|
|
394
|
+
|
|
395
|
+
child = engine_cls(
|
|
396
|
+
model=model,
|
|
397
|
+
thinking=thinking,
|
|
398
|
+
reasoning_effort=effort,
|
|
399
|
+
max_tokens=max_tokens,
|
|
400
|
+
workdir=workdir,
|
|
401
|
+
allowed_roots=allowed_roots,
|
|
402
|
+
system_prompt=prompt,
|
|
403
|
+
client=client,
|
|
404
|
+
registry=build_explore_registry(workdir, allowed_roots),
|
|
405
|
+
max_steps=max_steps,
|
|
406
|
+
finalize_steps=EXPLORE_FINALIZE_STEPS,
|
|
407
|
+
trajectory_meta={
|
|
408
|
+
"source": "explore",
|
|
409
|
+
"role": role,
|
|
410
|
+
"parent_session": parent_session,
|
|
411
|
+
"task": task[:200],
|
|
412
|
+
},
|
|
413
|
+
)
|
|
414
|
+
message = task if not context else (
|
|
415
|
+
f"{task}\n\nContext from the buyer:\n{context}"
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
steps, usage, timed_out = 0, {}, False
|
|
419
|
+
t0 = time.monotonic()
|
|
420
|
+
|
|
421
|
+
async def _drain() -> None:
|
|
422
|
+
nonlocal steps, usage
|
|
423
|
+
async for ev in child.run_turn(message):
|
|
424
|
+
if isinstance(ev, TurnFinished):
|
|
425
|
+
steps, usage = ev.steps, ev.usage
|
|
426
|
+
|
|
427
|
+
try:
|
|
428
|
+
await asyncio.wait_for(_drain(), timeout=EXPLORE_TIMEOUT_S)
|
|
429
|
+
except asyncio.TimeoutError:
|
|
430
|
+
timed_out = True # salvage whatever partial answer exists below
|
|
431
|
+
|
|
432
|
+
if ledger is not None and usage:
|
|
433
|
+
ledger.add(model, usage)
|
|
434
|
+
|
|
435
|
+
answer = _final_answer(child.history)
|
|
436
|
+
footer = ""
|
|
437
|
+
if timed_out:
|
|
438
|
+
answer = (
|
|
439
|
+
f"[timeout] explore exceeded {EXPLORE_TIMEOUT_S}s and was stopped."
|
|
440
|
+
+ (f" Partial output before the cutoff:\n{answer}" if answer else "")
|
|
441
|
+
)
|
|
442
|
+
elif not answer:
|
|
443
|
+
answer = "[error] the explore run finished without producing a report"
|
|
444
|
+
else:
|
|
445
|
+
footer = await check_citations(answer, workdir=workdir, git_ref=git_ref)
|
|
446
|
+
|
|
447
|
+
stats = (
|
|
448
|
+
f"[explore:{role} — {steps} steps · "
|
|
449
|
+
f"{usage.get('prompt_tokens', 0):,}p + {usage.get('completion_tokens', 0):,}c tokens · "
|
|
450
|
+
f"{time.monotonic() - t0:.0f}s · session {child.trajectory.session_id}]"
|
|
451
|
+
)
|
|
452
|
+
parts = [answer] + ([footer] if footer else []) + [stats]
|
|
453
|
+
return "\n\n".join(parts)
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
def make_branch_reviewer(engine):
|
|
457
|
+
"""A reviewer callable for goal_review.build_goal_tools: buys a grounded
|
|
458
|
+
review of a goal branch instead of dumping its diff into the chat context.
|
|
459
|
+
Bound to the live Engine; reads its settings at CALL time."""
|
|
460
|
+
|
|
461
|
+
async def _review(branch: str) -> str:
|
|
462
|
+
task = (
|
|
463
|
+
f"Review the git branch `{branch}` — the committed work of an "
|
|
464
|
+
f"autonomous goal run. It is NOT checked out: read its diff with "
|
|
465
|
+
f"`git diff HEAD...{branch}`, its commits with "
|
|
466
|
+
f"`git log --oneline HEAD..{branch}`, and any changed file's full "
|
|
467
|
+
f"branch version with `git show {branch}:path/to/file`. Read enough "
|
|
468
|
+
f"of the CURRENT branch (read_file) to judge the change in context — "
|
|
469
|
+
f"callers, tests, error paths. Deliver: a one-line VERDICT "
|
|
470
|
+
f"(merge-ready or not, and why), issues ranked by severity, and "
|
|
471
|
+
f"what you checked that is sound."
|
|
472
|
+
)
|
|
473
|
+
return await run_explore(
|
|
474
|
+
task=task,
|
|
475
|
+
role="review",
|
|
476
|
+
model=engine.model,
|
|
477
|
+
client=engine.client,
|
|
478
|
+
workdir=engine.workdir,
|
|
479
|
+
allowed_roots=engine.allowed_roots,
|
|
480
|
+
thinking=engine.thinking,
|
|
481
|
+
max_tokens=engine.max_tokens,
|
|
482
|
+
ledger=getattr(engine, "ledger", None),
|
|
483
|
+
parent_session=engine.trajectory.session_id,
|
|
484
|
+
git_ref=branch,
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
return _review
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
_VERDICT_LATE = re.compile(r"(?i)judgment:\s*(pass|fail)")
|
|
491
|
+
# Models decorate the verdict line ("VERDICT: **PASS** — …") even when told not
|
|
492
|
+
# to (seen in the review E2E); strip the dressing so goal.parse_verdict's strict
|
|
493
|
+
# first-line check sees a bare PASS/FAIL instead of conservatively failing.
|
|
494
|
+
_VERDICT_DRESSING = re.compile(r"(?i)\A[\s*#]*(?:verdict\s*[:—-]\s*)?[\s*]*")
|
|
495
|
+
|
|
496
|
+
|
|
497
|
+
def make_goal_verifier(*, client, model, workdir: Path, ledger=None, engine_cls=None):
|
|
498
|
+
"""Grounded milestone verification for goal mode (EngineDriver.verify): a
|
|
499
|
+
read-only verify child inspects the ACTUAL tree instead of judging from the
|
|
500
|
+
worker's self-summary. Returns an async callable
|
|
501
|
+
(milestone, summary, baseline, checks) -> (passed, report).
|
|
502
|
+
|
|
503
|
+
Raises when the child yields no clean verdict (or errors/times out), so the
|
|
504
|
+
caller can fall back to the summary judge — same degrade-gracefully shape
|
|
505
|
+
as goal_review's reviewer. Verdict semantics mirror goal._VERIFY_SYS: a
|
|
506
|
+
baseline error now gone was FIXED; an honest 'nothing to do' on a clean
|
|
507
|
+
state passes; pre-existing errors fail only a milestone that owned them;
|
|
508
|
+
a NEW error vs baseline is a regression and fails."""
|
|
509
|
+
|
|
510
|
+
async def _verify(*, milestone: str, summary: str, baseline: str, checks: str):
|
|
511
|
+
task = (
|
|
512
|
+
f"Decide whether this milestone's OBJECTIVE is achieved in the "
|
|
513
|
+
f"working tree:\n {milestone}\n\n"
|
|
514
|
+
f"The worker's claim (do NOT trust it — verify in the code):\n"
|
|
515
|
+
f"{summary or '(no summary given)'}\n\n"
|
|
516
|
+
f"Checks BEFORE the run began:\n{baseline or '(clean — no issues)'}\n\n"
|
|
517
|
+
f"Checks NOW:\n{checks}\n\n"
|
|
518
|
+
f"Inspect the tree — git status / git diff for what changed, read "
|
|
519
|
+
f"the touched files — and judge the END STATE by these rules: a "
|
|
520
|
+
f"baseline error that is GONE now was fixed (success, never an "
|
|
521
|
+
f"inconsistency); an honest 'nothing to do' on an already-clean "
|
|
522
|
+
f"state PASSES; an error present in both baseline and now is "
|
|
523
|
+
f"pre-existing and fails only if fixing it was THIS milestone's "
|
|
524
|
+
f"objective; a NEW error absent from the baseline is a regression "
|
|
525
|
+
f"this work introduced and FAILS."
|
|
526
|
+
)
|
|
527
|
+
report = await run_explore(
|
|
528
|
+
task=task, role="verify", model=model, client=client, workdir=workdir,
|
|
529
|
+
ledger=ledger, parent_session="goal-verify", max_steps=12,
|
|
530
|
+
engine_cls=engine_cls,
|
|
531
|
+
)
|
|
532
|
+
if report.startswith(("[error]", "[timeout]")):
|
|
533
|
+
raise RuntimeError(f"grounded verify unavailable: {report.splitlines()[0]}")
|
|
534
|
+
body = _VERDICT_DRESSING.sub("", report.strip(), count=1)
|
|
535
|
+
first = (body.splitlines() or [""])[0].strip().upper()
|
|
536
|
+
if not (first.startswith(("PASS", "FAIL")) or _VERDICT_LATE.search(body)):
|
|
537
|
+
raise RuntimeError("grounded verify returned no clean PASS/FAIL verdict")
|
|
538
|
+
from rockycode.engine.goal import parse_verdict # lazy: goal never imports explore
|
|
539
|
+
return parse_verdict(body)
|
|
540
|
+
|
|
541
|
+
return _verify
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
EXPLORE_SCHEMA = tools_mod._fn_schema(
|
|
545
|
+
"explore",
|
|
546
|
+
"Buy a focused READ-ONLY investigation from an explore agent with a fresh "
|
|
547
|
+
"context; you receive only its final report with harness-verified "
|
|
548
|
+
"citations — its search noise never enters your history. Roles: 'explore' "
|
|
549
|
+
"investigates the codebase (use instead of long grep/read chains when the "
|
|
550
|
+
"question spans many files); 'review' gives an independent second opinion "
|
|
551
|
+
"on a diff or design. Explore agents read/grep/glob/check_code and run "
|
|
552
|
+
"read-only shell; they cannot write, edit, or delegate. Several explore "
|
|
553
|
+
"calls in ONE message run in parallel. The agent sees nothing of this "
|
|
554
|
+
"conversation — make the task self-contained.",
|
|
555
|
+
{
|
|
556
|
+
"task": {
|
|
557
|
+
"type": "string",
|
|
558
|
+
"description": "Self-contained instructions: the question to answer or "
|
|
559
|
+
"thing to review, plus what a good report must cover.",
|
|
560
|
+
},
|
|
561
|
+
"role": {"type": "string", "enum": ["explore", "review"]},
|
|
562
|
+
"context": {
|
|
563
|
+
"type": "string",
|
|
564
|
+
"description": "Optional grounding: relevant paths, prior findings, constraints.",
|
|
565
|
+
},
|
|
566
|
+
},
|
|
567
|
+
["task", "role"],
|
|
568
|
+
)
|
|
569
|
+
|
|
570
|
+
|
|
571
|
+
def build_explore_tool(engine) -> dict[str, Tool]:
|
|
572
|
+
"""The chat-facing `explore` tool, bound to a live Engine. Reads the
|
|
573
|
+
engine's settings at CALL time (so a ledger attached after construction is
|
|
574
|
+
still found). risk='safe' is load-bearing — see the module docstring.
|
|
575
|
+
Not registered yet — lands with the chat caller (step 3)."""
|
|
576
|
+
|
|
577
|
+
async def _explore(task: str, role: str, context: str = "") -> str:
|
|
578
|
+
return await run_explore(
|
|
579
|
+
task=task,
|
|
580
|
+
role=role,
|
|
581
|
+
context=context,
|
|
582
|
+
model=engine.model,
|
|
583
|
+
client=engine.client,
|
|
584
|
+
workdir=engine.workdir,
|
|
585
|
+
allowed_roots=engine.allowed_roots,
|
|
586
|
+
thinking=engine.thinking,
|
|
587
|
+
max_tokens=engine.max_tokens,
|
|
588
|
+
ledger=getattr(engine, "ledger", None),
|
|
589
|
+
parent_session=engine.trajectory.session_id,
|
|
590
|
+
)
|
|
591
|
+
|
|
592
|
+
return {"explore": Tool(name="explore", schema=EXPLORE_SCHEMA, fn=_explore, risk="safe")}
|