rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,592 @@
1
+ """Explore: buy verified findings from a bounded, read-only, fresh-context run.
2
+
3
+ The primitive underneath rocky's delegation features. A parent (the chat model,
4
+ or the goal verify/review path) purchases an investigation at a fixed price —
5
+ capped steps, capped wall-clock, its own context — and receives ONLY the final
6
+ report; the child's search noise (greps, file dumps, dead ends) never enters
7
+ the parent's history or its prompt-cache path.
8
+
9
+ What makes the report trustworthy is the FINDINGS CONTRACT: claims must cite
10
+ evidence as `path:line "anchor text"`, and check_citations() mechanically
11
+ re-verifies every citation against the tree (or a git ref for branch reviews)
12
+ before the parent sees the report. Hallucinated evidence is flagged inline —
13
+ a prompt rule in other harnesses, a checker here. The same footer doubles as
14
+ an automatic grounding signal on each logged episode (source: "explore").
15
+
16
+ Safety model (mechanisms, not prompt text):
17
+ - The child's registry is BUILT read-only: the safe-tier subset of the normal
18
+ registry (read_file / grep / glob / check_code) plus a hard-gated read-only
19
+ bash (allowlisted binaries, relative paths only, no shell metacharacters —
20
+ see _ro_bash_check). No write_file, no edit_file.
21
+ - No `explore` tool in the child registry → children are leaf workers. Depth
22
+ cap of 1 by construction; no recursion counter to get wrong.
23
+ - The parent-facing tool is risk="safe", which is a CONTRACT with loop.py's
24
+ batch rule: a batch is concurrent only if every call is safe-tier. Because
25
+ a child mutates nothing, several explore calls in one assistant turn fan
26
+ out in parallel through the existing gather path — and any future WRITER
27
+ role must register as "risky" so the same rule serializes it with all
28
+ other mutations. Parallel writers stay impossible by construction.
29
+ - Children inherit the parent's workdir/allowed_roots jail, and their tool
30
+ outputs (and the report returned to the parent) pass through the normal
31
+ execute() redaction.
32
+
33
+ DeepSeek economics: each role's system prompt is a byte-stable constant and
34
+ the child toolset is fixed, so every explore of a role shares the same prompt
35
+ prefix — repeated purchases in a session hit the prefix cache instead of
36
+ re-paying the parent's ever-growing history.
37
+ """
38
+ from __future__ import annotations
39
+
40
+ import asyncio
41
+ import os
42
+ import re
43
+ import shlex
44
+ import time
45
+ from pathlib import Path
46
+ from typing import Optional
47
+
48
+ from rockycode.engine import tools as tools_mod
49
+ from rockycode.engine.events import TurnFinished
50
+ from rockycode.engine.tools import Tool
51
+
52
+ EXPLORE_MAX_STEPS = 20 # parent runs 50; a focused purchase should not need more
53
+ EXPLORE_FINALIZE_STEPS = 2
54
+ EXPLORE_EFFORT = "high" # parent default is "max"; delegated lookups don't need it
55
+ EXPLORE_TIMEOUT_S = int(os.getenv("ROCKYCODE_EXPLORE_TIMEOUT", "600"))
56
+ CITATION_CAP = 30 # citations checked per report — bounds checker cost
57
+
58
+ # ---------------------------------------------------------------------------
59
+ # Role prompts — BYTE-STABLE constants (task/context go in the user message,
60
+ # never in here) so every explore of a role reuses the same cached prefix.
61
+ # ---------------------------------------------------------------------------
62
+
63
+ _OUTPUT_CONTRACT = """
64
+ Report format — your final message is ALL the buyer will see, so it must stand
65
+ alone:
66
+ - FINDINGS: what you established. Ground every claim in a citation.
67
+ - EVIDENCE: one bullet per citation, EXACTLY this shape (straight quotes):
68
+ - path/to/file.py:123 "text copied verbatim from that line"
69
+ ONE line number (never a range); the path written from the REPO ROOT
70
+ (rockycode/engine/loop.py, never just loop.py); nothing between the path
71
+ and the colon — no backticks, no "(branch)" notes. When you reviewed a
72
+ branch or ref, cite the same plain way: the harness checks against the ref
73
+ you were given.
74
+ The harness re-reads every citation and flags any it cannot verify — an
75
+ unverifiable citation is worse than none, so quote real lines only.
76
+ - GAPS: what you could not determine, and which searches you ran that came up
77
+ empty (a negative claim without the searches behind it is worthless).
78
+ Be complete but not padded; the report replaces your whole transcript."""
79
+
80
+ EXPLORE_PROMPT = """You are rocky's explore agent: a read-only investigator \
81
+ answering one focused question about a codebase. You work in a fresh context; \
82
+ the agent that bought this investigation sees none of your tool calls — only \
83
+ your final report.
84
+
85
+ Investigate thoroughly: prefer grep/glob to locate, read_file to confirm. \
86
+ Follow the code, not your assumptions; check more than one naming convention \
87
+ before concluding something does not exist. You cannot write, edit, or \
88
+ delegate — if the task seems to need that, report it as a finding instead.
89
+ """ + _OUTPUT_CONTRACT
90
+
91
+ REVIEW_PROMPT = """You are rocky's review agent: an independent, read-only \
92
+ second opinion on a change (a diff, branch, or set of files). You work in a \
93
+ fresh context on purpose — judge only what the code says, not what the author \
94
+ intended. The buyer sees only your final report.
95
+
96
+ Read the change AND enough surrounding code to judge it in context (callers, \
97
+ tests, error paths). Rank what you find by severity; a missed failure mode \
98
+ outranks any style point, and style points are out of scope unless asked. \
99
+ Confirm the good as well: say what you checked and found sound, so silence \
100
+ is not ambiguous. Start your report with a one-line VERDICT.
101
+ """ + _OUTPUT_CONTRACT
102
+
103
+ VERIFY_PROMPT = """You are rocky's verify agent: a read-only inspector deciding \
104
+ whether a claimed milestone is ACTUALLY complete in the working tree. You work \
105
+ in a fresh context on purpose — judge the code's end state, never the worker's \
106
+ account of it.
107
+
108
+ Inspect what changed (git status, git diff, read the touched files; run \
109
+ check_code if useful) and test the claim against reality. The buyer supplies \
110
+ baseline-vs-now check output and decision rules in the task; apply them \
111
+ exactly. Your report's FIRST line must be exactly "PASS — <one-line reason>" \
112
+ or "FAIL — <one-line reason>" — the verdict first, no hedging before it.
113
+ """ + _OUTPUT_CONTRACT
114
+
115
+ # "verify" is goal-internal: reachable through make_goal_verifier, deliberately
116
+ # absent from the chat tool's role enum (EXPLORE_SCHEMA).
117
+ ROLE_PROMPTS: dict[str, str] = {
118
+ "explore": EXPLORE_PROMPT,
119
+ "review": REVIEW_PROMPT,
120
+ "verify": VERIFY_PROMPT,
121
+ }
122
+
123
+ # ---------------------------------------------------------------------------
124
+ # Read-only bash gate. safety.classify_command is a DANGER classifier (an
125
+ # innocuous `echo x > f` passes it), so the child gets its own gate that is
126
+ # read-only by construction. Layered, strictest first:
127
+ # 1. no shell metacharacters at all: chaining, redirects (both ways — `<`
128
+ # would still open arbitrary files), backgrounding, substitution, `$`
129
+ # expansion, multi-line. Single pipes are the one composition allowed.
130
+ # 2. every pipe segment's binary must be on a small read-only allowlist;
131
+ # git additionally needs an allowlisted read-only subcommand (this also
132
+ # kills `git -C /elsewhere`, since `-C` is not a subcommand).
133
+ # 3. relative paths only — cwd is the jailed workdir, so every operand
134
+ # stays in-tree without parsing which args are paths. `cat` is deliberately
135
+ # NOT allowlisted: in-tree reads are read_file's job (jailed, line-numbered,
136
+ # secret-refusing).
137
+ # 4. tokens matching the secret-file patterns (.env, id_rsa, *.pem …) are
138
+ # refused, mirroring read_file. Residual risk — secrets already committed
139
+ # to git history via `git show` — is accepted: execute() redaction still
140
+ # scrubs known token shapes from the output.
141
+ # ---------------------------------------------------------------------------
142
+
143
+ _RO_BINS = {"ls", "wc", "head", "tail", "stat", "file", "du", "diff", "git"}
144
+ _RO_GIT_SUBS = {
145
+ "status", "log", "diff", "show", "blame", "shortlog", "describe",
146
+ "rev-parse", "ls-files", "grep", "branch",
147
+ }
148
+ # `git branch` mutates with these; listing stays allowed.
149
+ _GIT_BRANCH_MUTATING = re.compile(r"(^|\s)(-d|-D|-m|-M|-c|-C|--delete|--move|--copy|--force)(\s|$)")
150
+ _FORBIDDEN = re.compile(r"[;&`><$\n]|--output\b")
151
+
152
+ _RO_REFUSAL = (
153
+ "[blocked] explore bash is read-only: allowlisted binaries only "
154
+ f"({', '.join(sorted(_RO_BINS))}; git subcommands: {', '.join(sorted(_RO_GIT_SUBS))}), "
155
+ "relative paths, single pipes; no redirects, chaining, `$`, or secret files. "
156
+ "Use read_file / grep / glob for file access."
157
+ )
158
+
159
+
160
+ def _ro_bash_check(command: str) -> Optional[str]:
161
+ """Return a refusal string, or None if *command* is read-only-safe."""
162
+ if _FORBIDDEN.search(command):
163
+ return _RO_REFUSAL
164
+ for segment in command.split("|"):
165
+ try:
166
+ tokens = shlex.split(segment)
167
+ except ValueError:
168
+ return _RO_REFUSAL # unbalanced quotes — refuse rather than guess
169
+ if not tokens:
170
+ return _RO_REFUSAL # empty segment (also catches `a || b` remnants)
171
+ head, rest = tokens[0], tokens[1:]
172
+ if head not in _RO_BINS: # also refuses VAR=x prefixes: '=' names no bin
173
+ return _RO_REFUSAL
174
+ if head == "git":
175
+ if not rest or rest[0] not in _RO_GIT_SUBS:
176
+ return _RO_REFUSAL
177
+ if rest[0] == "branch" and _GIT_BRANCH_MUTATING.search(segment):
178
+ return _RO_REFUSAL
179
+ for tok in rest:
180
+ if tok.startswith("-"):
181
+ continue
182
+ if tok.startswith(("/", "~")) or ".." in tok.split("/"):
183
+ return _RO_REFUSAL # relative, in-tree paths only
184
+ if tools_mod._is_secret_file(Path(tok)):
185
+ return _RO_REFUSAL
186
+ return None
187
+
188
+
189
+ _RO_BASH_SCHEMA = tools_mod._fn_schema(
190
+ "bash",
191
+ "Run a READ-ONLY shell command in the working directory (git status/log/"
192
+ "diff/show/blame, ls, wc, head, tail, stat, du, diff; single pipes allowed). "
193
+ "Anything that writes, chains, redirects, or leaves the tree is refused.",
194
+ {"command": {"type": "string", "description": "The read-only command to run."}},
195
+ ["command"],
196
+ )
197
+
198
+
199
+ def build_explore_registry(
200
+ workdir: Path, allowed_roots: tuple[Path, ...] = ()
201
+ ) -> dict[str, Tool]:
202
+ """The child's toolset: the safe tier of the normal registry plus gated
203
+ read-only bash. Built FRESH (not filtered from the parent's live registry)
204
+ so session extras — MCP, memory, web, skills — never leak into children,
205
+ and the toolset stays deterministic and byte-stable per role."""
206
+ base = tools_mod.build_registry(workdir, allowed_roots)
207
+ reg = {name: t for name, t in base.items() if t.risk == "safe"}
208
+
209
+ async def _ro_bash(command: str) -> str:
210
+ refusal = _ro_bash_check(command)
211
+ if refusal:
212
+ return refusal
213
+ return await tools_mod._bash(command, workdir=workdir)
214
+
215
+ # Read-only by construction, hence safe-tier: the child's own read batches
216
+ # (including bash) parallelize through the same loop.py rule.
217
+ reg["bash"] = Tool(name="bash", schema=_RO_BASH_SCHEMA, fn=_ro_bash, risk="safe")
218
+ return reg # note: no `explore` tool — children cannot delegate
219
+
220
+
221
+ # ---------------------------------------------------------------------------
222
+ # The citation checker — the original core of the findings contract. Parses
223
+ # `path:line "anchor"` citations out of the report and re-verifies each one
224
+ # against the working tree (or `git show <ref>:<path>` for branch reviews).
225
+ # Lenient-but-honest v1: the anchor may sit within ±3 lines of the stated
226
+ # line (models miscount); a miss marks the citation [unverified] in a footer
227
+ # rather than rejecting the report — a fuzzy child must not deadlock a review.
228
+ # ---------------------------------------------------------------------------
229
+
230
+ # Tolerates the deviations models actually produce (seen in E2E): backticked
231
+ # paths (stripped before scanning), a short parenthetical between path and
232
+ # colon ("(branch)"), and line ranges (first line taken).
233
+ _CITE_RX = re.compile(
234
+ r'(?P<path>[A-Za-z0-9_][\w./-]*\.[A-Za-z0-9_]{1,8})'
235
+ r'(?:\s*\([^)\n]{1,24}\))?'
236
+ r':(?P<line>\d{1,6})(?:-\d{1,6})?'
237
+ r'(?:\s*[—–-]*\s*"(?P<snip>[^"\n]{3,160})")?'
238
+ )
239
+ _CITE_WINDOW = 3 # anchor accepted within ± this many lines of the stated line
240
+
241
+
242
+ def _norm(s: str) -> str:
243
+ return " ".join(s.split())
244
+
245
+
246
+ async def _git_show(workdir: Path, ref: str, path: str) -> Optional[str]:
247
+ """Contents of *path* on *ref*, or None. Lets citations into a not-checked-
248
+ out goal branch verify against what the branch actually says."""
249
+ try:
250
+ proc = await asyncio.create_subprocess_exec(
251
+ "git", "-C", str(workdir), "show", f"{ref}:{path}",
252
+ stdout=asyncio.subprocess.PIPE, stderr=asyncio.subprocess.DEVNULL,
253
+ )
254
+ out, _ = await proc.communicate()
255
+ return out.decode("utf-8", "replace") if proc.returncode == 0 else None
256
+ except OSError:
257
+ return None
258
+
259
+
260
+ def _anchor_ok(text: str, line: int, snip: Optional[str]) -> tuple[bool, str]:
261
+ lines = text.splitlines()
262
+ if snip is None:
263
+ # No anchor to check — existence-only (weak) verification.
264
+ ok = 1 <= line <= len(lines)
265
+ return ok, "" if ok else "line beyond end of file"
266
+ lo, hi = max(0, line - 1 - _CITE_WINDOW), min(len(lines), line + _CITE_WINDOW)
267
+ target = _norm(snip)
268
+ if any(target in _norm(ln) for ln in lines[lo:hi]):
269
+ return True, ""
270
+ # The check exists to catch FABRICATED evidence: a verbatim quote with a
271
+ # stale line number (models miscount; seen in E2E) is sloppy, not fake.
272
+ if any(target in _norm(ln) for ln in lines):
273
+ return True, ""
274
+ return False, "anchor not found in file"
275
+
276
+
277
+ _SKIP_DIRS = {".git", "node_modules", ".venv", "venv", "env", "__pycache__", "dist", "build"}
278
+
279
+
280
+ def _resolve_short_path(workdir: Path, rel: str) -> Optional[Path]:
281
+ """A citation like 'loop.py:51' with the repo-root prefix missing (the most
282
+ common model deviation seen in E2E): accept it iff exactly ONE tree file
283
+ ends with the cited path — honest under-qualification, not fabrication."""
284
+ parts = tuple(Path(rel).parts)
285
+ hits: list[Path] = []
286
+ try:
287
+ for p in workdir.rglob(parts[-1]):
288
+ if any(s in p.parts for s in _SKIP_DIRS) or not p.is_file():
289
+ continue
290
+ if tuple(p.parts[-len(parts):]) == parts:
291
+ hits.append(p)
292
+ if len(hits) > 1:
293
+ return None # ambiguous — refuse to guess
294
+ except OSError:
295
+ return None
296
+ return hits[0] if len(hits) == 1 else None
297
+
298
+
299
+ async def check_citations(
300
+ report: str, *, workdir: Path, git_ref: Optional[str] = None
301
+ ) -> str:
302
+ """Verify every `path:line "anchor"` citation in *report*; return the
303
+ one-line-per-problem footer the buyer sees. Never raises."""
304
+ seen: dict[tuple, Optional[str]] = {}
305
+ report = report.replace("`", "") # backticked paths still cite
306
+ for m in _CITE_RX.finditer(report):
307
+ key = (m["path"], int(m["line"]), m["snip"])
308
+ if key not in seen and len(seen) < CITATION_CAP:
309
+ seen[key] = None
310
+ if not seen:
311
+ return "[citations: none found — treat unevidenced claims with caution]"
312
+
313
+ for path, line, snip in list(seen):
314
+ p, err = tools_mod._jail(path, workdir)
315
+ text: Optional[str] = None
316
+ if err is None and p is not None and not p.is_file():
317
+ alt = await asyncio.to_thread(_resolve_short_path, workdir, path)
318
+ if alt is not None:
319
+ p = alt
320
+ if err is None and p is not None and p.is_file():
321
+ try:
322
+ text = p.read_text(errors="replace")
323
+ except OSError:
324
+ text = None
325
+ if text is None or (git_ref and not _anchor_ok(text, line, snip)[0]):
326
+ # Not verifiable against the tree — try the reviewed ref's version
327
+ # (a citation is good if it holds on EITHER side).
328
+ branch_text = await _git_show(workdir, git_ref, path) if git_ref else None
329
+ if branch_text is not None:
330
+ text = branch_text
331
+ if err is not None:
332
+ seen[(path, line, snip)] = "path outside the workdir"
333
+ continue
334
+ if text is None:
335
+ seen[(path, line, snip)] = "file not found"
336
+ continue
337
+ ok, why = _anchor_ok(text, line, snip)
338
+ seen[(path, line, snip)] = None if ok else why
339
+
340
+ bad = {k: why for k, why in seen.items() if why}
341
+ if not bad:
342
+ return f"[citations: {len(seen)}/{len(seen)} verified]"
343
+ detail = ", ".join(f"{p}:{ln} ({why})" for (p, ln, _s), why in bad.items())
344
+ return f"[citations: {len(seen) - len(bad)}/{len(seen)} verified · unverified: {detail}]"
345
+
346
+
347
+ # ---------------------------------------------------------------------------
348
+ # Running a child + the buyer-facing surfaces.
349
+ # ---------------------------------------------------------------------------
350
+
351
+
352
+ def _final_answer(history: list[dict]) -> str:
353
+ """Last non-empty assistant text — the child's report."""
354
+ for msg in reversed(history):
355
+ if msg.get("role") == "assistant" and (msg.get("content") or "").strip():
356
+ return msg["content"].strip()
357
+ return ""
358
+
359
+
360
+ async def run_explore(
361
+ *,
362
+ task: str,
363
+ role: str,
364
+ context: str = "",
365
+ model: str,
366
+ client,
367
+ workdir: Path,
368
+ allowed_roots: tuple[Path, ...] = (),
369
+ thinking: bool = True,
370
+ effort: str = EXPLORE_EFFORT,
371
+ max_tokens: int = 384_000,
372
+ ledger=None,
373
+ parent_session: str = "",
374
+ git_ref: Optional[str] = None,
375
+ max_steps: int = EXPLORE_MAX_STEPS,
376
+ engine_cls=None,
377
+ ) -> str:
378
+ """Buy one investigation: spawn a fresh child Engine on *task*, verify the
379
+ report's citations, and return report + citation footer + stats line.
380
+
381
+ *git_ref* lets branch-review citations verify against `git show ref:path`
382
+ (a goal branch is not checked out). The child logs its own trajectory
383
+ (source=explore, linked to the parent session) — a clean single-task
384
+ episode whose citation footer doubles as a grounding signal. Usage folds
385
+ into *ledger* so /cost stays truthful. *engine_cls* is a test seam, same
386
+ spirit as goal.Driver.
387
+ """
388
+ prompt = ROLE_PROMPTS.get(role)
389
+ if prompt is None:
390
+ roles = ", ".join(sorted(ROLE_PROMPTS))
391
+ return f"[error] unknown explore role {role!r} — expected one of: {roles}"
392
+ if engine_cls is None:
393
+ from rockycode.engine.loop import Engine as engine_cls # avoid import cycle
394
+
395
+ child = engine_cls(
396
+ model=model,
397
+ thinking=thinking,
398
+ reasoning_effort=effort,
399
+ max_tokens=max_tokens,
400
+ workdir=workdir,
401
+ allowed_roots=allowed_roots,
402
+ system_prompt=prompt,
403
+ client=client,
404
+ registry=build_explore_registry(workdir, allowed_roots),
405
+ max_steps=max_steps,
406
+ finalize_steps=EXPLORE_FINALIZE_STEPS,
407
+ trajectory_meta={
408
+ "source": "explore",
409
+ "role": role,
410
+ "parent_session": parent_session,
411
+ "task": task[:200],
412
+ },
413
+ )
414
+ message = task if not context else (
415
+ f"{task}\n\nContext from the buyer:\n{context}"
416
+ )
417
+
418
+ steps, usage, timed_out = 0, {}, False
419
+ t0 = time.monotonic()
420
+
421
+ async def _drain() -> None:
422
+ nonlocal steps, usage
423
+ async for ev in child.run_turn(message):
424
+ if isinstance(ev, TurnFinished):
425
+ steps, usage = ev.steps, ev.usage
426
+
427
+ try:
428
+ await asyncio.wait_for(_drain(), timeout=EXPLORE_TIMEOUT_S)
429
+ except asyncio.TimeoutError:
430
+ timed_out = True # salvage whatever partial answer exists below
431
+
432
+ if ledger is not None and usage:
433
+ ledger.add(model, usage)
434
+
435
+ answer = _final_answer(child.history)
436
+ footer = ""
437
+ if timed_out:
438
+ answer = (
439
+ f"[timeout] explore exceeded {EXPLORE_TIMEOUT_S}s and was stopped."
440
+ + (f" Partial output before the cutoff:\n{answer}" if answer else "")
441
+ )
442
+ elif not answer:
443
+ answer = "[error] the explore run finished without producing a report"
444
+ else:
445
+ footer = await check_citations(answer, workdir=workdir, git_ref=git_ref)
446
+
447
+ stats = (
448
+ f"[explore:{role} — {steps} steps · "
449
+ f"{usage.get('prompt_tokens', 0):,}p + {usage.get('completion_tokens', 0):,}c tokens · "
450
+ f"{time.monotonic() - t0:.0f}s · session {child.trajectory.session_id}]"
451
+ )
452
+ parts = [answer] + ([footer] if footer else []) + [stats]
453
+ return "\n\n".join(parts)
454
+
455
+
456
+ def make_branch_reviewer(engine):
457
+ """A reviewer callable for goal_review.build_goal_tools: buys a grounded
458
+ review of a goal branch instead of dumping its diff into the chat context.
459
+ Bound to the live Engine; reads its settings at CALL time."""
460
+
461
+ async def _review(branch: str) -> str:
462
+ task = (
463
+ f"Review the git branch `{branch}` — the committed work of an "
464
+ f"autonomous goal run. It is NOT checked out: read its diff with "
465
+ f"`git diff HEAD...{branch}`, its commits with "
466
+ f"`git log --oneline HEAD..{branch}`, and any changed file's full "
467
+ f"branch version with `git show {branch}:path/to/file`. Read enough "
468
+ f"of the CURRENT branch (read_file) to judge the change in context — "
469
+ f"callers, tests, error paths. Deliver: a one-line VERDICT "
470
+ f"(merge-ready or not, and why), issues ranked by severity, and "
471
+ f"what you checked that is sound."
472
+ )
473
+ return await run_explore(
474
+ task=task,
475
+ role="review",
476
+ model=engine.model,
477
+ client=engine.client,
478
+ workdir=engine.workdir,
479
+ allowed_roots=engine.allowed_roots,
480
+ thinking=engine.thinking,
481
+ max_tokens=engine.max_tokens,
482
+ ledger=getattr(engine, "ledger", None),
483
+ parent_session=engine.trajectory.session_id,
484
+ git_ref=branch,
485
+ )
486
+
487
+ return _review
488
+
489
+
490
+ _VERDICT_LATE = re.compile(r"(?i)judgment:\s*(pass|fail)")
491
+ # Models decorate the verdict line ("VERDICT: **PASS** — …") even when told not
492
+ # to (seen in the review E2E); strip the dressing so goal.parse_verdict's strict
493
+ # first-line check sees a bare PASS/FAIL instead of conservatively failing.
494
+ _VERDICT_DRESSING = re.compile(r"(?i)\A[\s*#]*(?:verdict\s*[:—-]\s*)?[\s*]*")
495
+
496
+
497
+ def make_goal_verifier(*, client, model, workdir: Path, ledger=None, engine_cls=None):
498
+ """Grounded milestone verification for goal mode (EngineDriver.verify): a
499
+ read-only verify child inspects the ACTUAL tree instead of judging from the
500
+ worker's self-summary. Returns an async callable
501
+ (milestone, summary, baseline, checks) -> (passed, report).
502
+
503
+ Raises when the child yields no clean verdict (or errors/times out), so the
504
+ caller can fall back to the summary judge — same degrade-gracefully shape
505
+ as goal_review's reviewer. Verdict semantics mirror goal._VERIFY_SYS: a
506
+ baseline error now gone was FIXED; an honest 'nothing to do' on a clean
507
+ state passes; pre-existing errors fail only a milestone that owned them;
508
+ a NEW error vs baseline is a regression and fails."""
509
+
510
+ async def _verify(*, milestone: str, summary: str, baseline: str, checks: str):
511
+ task = (
512
+ f"Decide whether this milestone's OBJECTIVE is achieved in the "
513
+ f"working tree:\n {milestone}\n\n"
514
+ f"The worker's claim (do NOT trust it — verify in the code):\n"
515
+ f"{summary or '(no summary given)'}\n\n"
516
+ f"Checks BEFORE the run began:\n{baseline or '(clean — no issues)'}\n\n"
517
+ f"Checks NOW:\n{checks}\n\n"
518
+ f"Inspect the tree — git status / git diff for what changed, read "
519
+ f"the touched files — and judge the END STATE by these rules: a "
520
+ f"baseline error that is GONE now was fixed (success, never an "
521
+ f"inconsistency); an honest 'nothing to do' on an already-clean "
522
+ f"state PASSES; an error present in both baseline and now is "
523
+ f"pre-existing and fails only if fixing it was THIS milestone's "
524
+ f"objective; a NEW error absent from the baseline is a regression "
525
+ f"this work introduced and FAILS."
526
+ )
527
+ report = await run_explore(
528
+ task=task, role="verify", model=model, client=client, workdir=workdir,
529
+ ledger=ledger, parent_session="goal-verify", max_steps=12,
530
+ engine_cls=engine_cls,
531
+ )
532
+ if report.startswith(("[error]", "[timeout]")):
533
+ raise RuntimeError(f"grounded verify unavailable: {report.splitlines()[0]}")
534
+ body = _VERDICT_DRESSING.sub("", report.strip(), count=1)
535
+ first = (body.splitlines() or [""])[0].strip().upper()
536
+ if not (first.startswith(("PASS", "FAIL")) or _VERDICT_LATE.search(body)):
537
+ raise RuntimeError("grounded verify returned no clean PASS/FAIL verdict")
538
+ from rockycode.engine.goal import parse_verdict # lazy: goal never imports explore
539
+ return parse_verdict(body)
540
+
541
+ return _verify
542
+
543
+
544
+ EXPLORE_SCHEMA = tools_mod._fn_schema(
545
+ "explore",
546
+ "Buy a focused READ-ONLY investigation from an explore agent with a fresh "
547
+ "context; you receive only its final report with harness-verified "
548
+ "citations — its search noise never enters your history. Roles: 'explore' "
549
+ "investigates the codebase (use instead of long grep/read chains when the "
550
+ "question spans many files); 'review' gives an independent second opinion "
551
+ "on a diff or design. Explore agents read/grep/glob/check_code and run "
552
+ "read-only shell; they cannot write, edit, or delegate. Several explore "
553
+ "calls in ONE message run in parallel. The agent sees nothing of this "
554
+ "conversation — make the task self-contained.",
555
+ {
556
+ "task": {
557
+ "type": "string",
558
+ "description": "Self-contained instructions: the question to answer or "
559
+ "thing to review, plus what a good report must cover.",
560
+ },
561
+ "role": {"type": "string", "enum": ["explore", "review"]},
562
+ "context": {
563
+ "type": "string",
564
+ "description": "Optional grounding: relevant paths, prior findings, constraints.",
565
+ },
566
+ },
567
+ ["task", "role"],
568
+ )
569
+
570
+
571
+ def build_explore_tool(engine) -> dict[str, Tool]:
572
+ """The chat-facing `explore` tool, bound to a live Engine. Reads the
573
+ engine's settings at CALL time (so a ledger attached after construction is
574
+ still found). risk='safe' is load-bearing — see the module docstring.
575
+ Not registered yet — lands with the chat caller (step 3)."""
576
+
577
+ async def _explore(task: str, role: str, context: str = "") -> str:
578
+ return await run_explore(
579
+ task=task,
580
+ role=role,
581
+ context=context,
582
+ model=engine.model,
583
+ client=engine.client,
584
+ workdir=engine.workdir,
585
+ allowed_roots=engine.allowed_roots,
586
+ thinking=engine.thinking,
587
+ max_tokens=engine.max_tokens,
588
+ ledger=getattr(engine, "ledger", None),
589
+ parent_session=engine.trajectory.session_id,
590
+ )
591
+
592
+ return {"explore": Tool(name="explore", schema=EXPLORE_SCHEMA, fn=_explore, risk="safe")}