rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Scrub secrets from tool output before it enters the conversation history.
|
|
2
|
+
|
|
3
|
+
History is the single thing that is BOTH sent to the model provider (the API
|
|
4
|
+
prompt) AND written to the trajectory log, so redacting tool output at the
|
|
5
|
+
tools.execute() chokepoint fixes both at once — and the model itself only ever
|
|
6
|
+
sees `[redacted]`, so it can't later echo a key it read. (The user's own API key
|
|
7
|
+
still travels in the request AUTH HEADER — that is how authentication works and
|
|
8
|
+
is untouched; what we scrub is secret text that shows up in message CONTENT,
|
|
9
|
+
e.g. from `env`, `cat .env`, or a printed token.)
|
|
10
|
+
|
|
11
|
+
Two passes:
|
|
12
|
+
1. Known values — the literal values of the user's own sensitive env vars
|
|
13
|
+
(OPENAI_API_KEY, ANTHROPIC_AUTH_TOKEN, *_TOKEN, …). Highest precision:
|
|
14
|
+
redacts the actual secret wherever it appears (catches `env`/`printenv`).
|
|
15
|
+
2. Shapes — regexes for well-known secret formats when we don't know the
|
|
16
|
+
value (sk-…, ghp_…, AKIA…, AIza… google keys, JWTs, PRIVATE KEY blocks,
|
|
17
|
+
Bearer …, KEY=value) plus a generic high-entropy 32–64-char token
|
|
18
|
+
heuristic (mixed lower+UPPER+digit, so git SHAs / md5s are spared).
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import os
|
|
23
|
+
import re
|
|
24
|
+
|
|
25
|
+
# Env var NAMES considered secret; their VALUES are scrubbed from output.
|
|
26
|
+
_SENSITIVE_ENV = re.compile(r"(API_KEY|_TOKEN|_SECRET|PASSWORD|PASSWD|ANTHROPIC_|OPENAI_)", re.I)
|
|
27
|
+
# Don't redact trivially short values — avoids nuking "1"/"true"/"dev" if such a
|
|
28
|
+
# value ever lands in a sensitive-named var.
|
|
29
|
+
_MIN_VALUE_LEN = 6
|
|
30
|
+
|
|
31
|
+
_SHAPES: list[tuple[re.Pattern[str], str]] = [
|
|
32
|
+
(re.compile(r"-----BEGIN [A-Z ]*PRIVATE KEY-----.*?-----END [A-Z ]*PRIVATE KEY-----", re.S),
|
|
33
|
+
"[redacted: private key]"),
|
|
34
|
+
(re.compile(r"\b(?:sk|rk)-[A-Za-z0-9_-]{16,}"), "[redacted: api key]"),
|
|
35
|
+
(re.compile(r"\bgh[posur]_[A-Za-z0-9]{20,}"), "[redacted: github token]"),
|
|
36
|
+
(re.compile(r"\bAKIA[0-9A-Z]{16}\b"), "[redacted: aws key id]"),
|
|
37
|
+
(re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}"), "[redacted: slack token]"),
|
|
38
|
+
# Google API key: literal "AIza" + 35 url-safe base64 chars (fixed length).
|
|
39
|
+
(re.compile(r"\bAIza[0-9A-Za-z_-]{35}\b"), "[redacted: google api key]"),
|
|
40
|
+
# JWT: three base64url segments separated by dots, first starts with the
|
|
41
|
+
# canonical `eyJ` ({"…} header). Matches access/id tokens people print.
|
|
42
|
+
(re.compile(r"\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b"),
|
|
43
|
+
"[redacted: jwt]"),
|
|
44
|
+
(re.compile(r"(?i)\bBearer\s+[A-Za-z0-9._~+/-]{16,}=*"), "Bearer [redacted]"),
|
|
45
|
+
# NAME=value / NAME: value where the name ends in a secret word. The word
|
|
46
|
+
# must sit immediately before the separator, so MAX_TOKENS / TOKEN_COUNT
|
|
47
|
+
# (no separator right after the word) are NOT matched; ACCESS_TOKEN= is.
|
|
48
|
+
(re.compile(r"""(?im)^(\s*[\w.-]*(?:_API_KEY|_TOKEN|_SECRET|SECRET_KEY|PASSWORD|PASSWD)\s*[=:]\s*)
|
|
49
|
+
(["']?[^\s"']{8,}["']?)""", re.VERBOSE),
|
|
50
|
+
r"\1[redacted]"),
|
|
51
|
+
# Generic high-entropy token: a 32–64 char base62-ish run that mixes
|
|
52
|
+
# lower + UPPER + digit. The three-class requirement is what spares
|
|
53
|
+
# low-diversity strings that are NOT secrets — a 40-hex git SHA / md5 (no
|
|
54
|
+
# uppercase), an ALL-CAPS constant, a decimal id — while still catching the
|
|
55
|
+
# opaque provider keys that don't carry a recognizable prefix. Runs LAST so
|
|
56
|
+
# the labelled shapes above win. Best-effort; may occasionally over-redact a
|
|
57
|
+
# mixed-case blob, which on tool output is the safe direction.
|
|
58
|
+
(re.compile(r"\b(?=[A-Za-z0-9_-]*[a-z])(?=[A-Za-z0-9_-]*[A-Z])(?=[A-Za-z0-9_-]*\d)"
|
|
59
|
+
r"[A-Za-z0-9_-]{32,64}\b"),
|
|
60
|
+
"[redacted: token]"),
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _known_values() -> list[str]:
|
|
65
|
+
vals = {
|
|
66
|
+
v for k, v in os.environ.items()
|
|
67
|
+
if v and len(v) >= _MIN_VALUE_LEN and _SENSITIVE_ENV.search(k)
|
|
68
|
+
}
|
|
69
|
+
# Longest first: if one secret value is a substring of another, redact the
|
|
70
|
+
# bigger match before the smaller can partially hit it.
|
|
71
|
+
return sorted(vals, key=len, reverse=True)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def redact(text: str) -> str:
|
|
75
|
+
"""Return *text* with known secret values and secret-shaped tokens masked."""
|
|
76
|
+
if not text:
|
|
77
|
+
return text
|
|
78
|
+
for value in _known_values():
|
|
79
|
+
if value in text:
|
|
80
|
+
text = text.replace(value, "[redacted]")
|
|
81
|
+
for rx, repl in _SHAPES:
|
|
82
|
+
text = rx.sub(repl, text)
|
|
83
|
+
return text
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""Command-safety classification for autonomous (goal) mode.
|
|
2
|
+
|
|
3
|
+
Goal mode runs unattended, and the sandbox mounts the project READ-WRITE, so a
|
|
4
|
+
destructive command reaches the user's real files. Every bash command is
|
|
5
|
+
classified BEFORE it runs, and the goal/plan is pre-scanned at start. Three
|
|
6
|
+
tiers (highest severity wins):
|
|
7
|
+
|
|
8
|
+
block — destructive / irreversible: rm -rf of a filesystem/home/mount root,
|
|
9
|
+
mkfs, dd to a device, a fork bomb, shutdown. NEVER run in goal mode —
|
|
10
|
+
the model gets an error and must find a reversible path. Isolation
|
|
11
|
+
(goal runs on a COPY / git worktree) is the backstop so even a missed
|
|
12
|
+
one can't touch the user's real repo.
|
|
13
|
+
ask — risky but sometimes legitimately needed: git push, sudo, curl|sh, a
|
|
14
|
+
system OR language package install (apt/brew/pip/npm/cargo/… — they
|
|
15
|
+
run code from a registry, so a poisoned/typosquat package is a real
|
|
16
|
+
supply-chain risk). Surfaced at goal START for ONE up-front approval;
|
|
17
|
+
if unapproved at runtime, treated as block.
|
|
18
|
+
allow — everything else, incl. normal dev (rm -rf build/, git commit, npm run,
|
|
19
|
+
go build, pytest …).
|
|
20
|
+
|
|
21
|
+
This guards the agent against its own long-horizon mistakes; it is NOT a
|
|
22
|
+
security boundary against a hostile model (the container + copy isolation are).
|
|
23
|
+
"""
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import re
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass(frozen=True)
|
|
31
|
+
class Verdict:
|
|
32
|
+
action: str # "allow" | "ask" | "block"
|
|
33
|
+
reason: str
|
|
34
|
+
pattern: str
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
# rm -rf aimed at a filesystem/home/mount ROOT — not a named subdir. `rm -rf
|
|
38
|
+
# build/` stays allow; `rm -rf /`, `~`, `$HOME`, `/workspace`, `.`, `*` are block.
|
|
39
|
+
_RM_ROOT = re.compile(
|
|
40
|
+
r"""\brm\b[^\n|;&]*\s
|
|
41
|
+
(?:-\w+\s)*
|
|
42
|
+
(?:
|
|
43
|
+
/(?:\s|$|\*) # bare / or /*
|
|
44
|
+
| /(?:etc|var|usr|s?bin|lib\w*|boot|sys|proc|dev|opt|root|home|srv|run)(?:/|\s|$)
|
|
45
|
+
| ~(?:/\s*\*?)?(?:\s|$) # ~ ~/ ~/*
|
|
46
|
+
| \$HOME\b
|
|
47
|
+
| /workspace(?:/\s*\*?)?(?:\s|$)
|
|
48
|
+
| \.\s*$ # trailing .
|
|
49
|
+
| \*\s*$ # trailing *
|
|
50
|
+
)
|
|
51
|
+
""",
|
|
52
|
+
re.VERBOSE,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
_BLOCK = [
|
|
56
|
+
("fork-bomb", re.compile(r":\s*\(\s*\)\s*\{\s*:\s*\|\s*:\s*&\s*\}\s*;\s*:"), "fork bomb"),
|
|
57
|
+
("disk-write", re.compile(r"\b(mkfs\w*|fdisk|parted)\b|\bdd\b[^\n]*\bof=/dev/"), "raw disk / filesystem write"),
|
|
58
|
+
("redirect-device", re.compile(r">\s*/dev/(sd|nvme|disk|mmcblk)\w+"), "redirect to a block device"),
|
|
59
|
+
("power", re.compile(r"\b(shutdown|reboot|halt|poweroff|init\s+0)\b"), "power / host-state change"),
|
|
60
|
+
("chmod-root", re.compile(r"\b(chmod|chown)\b[^\n]*\s-\w*R\w*\s[^\n]*\s(/|~|\$HOME|/workspace)(\s|$)"),
|
|
61
|
+
"recursive perms/owner change on a filesystem root"),
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
_ASK = [
|
|
65
|
+
("git-push", re.compile(r"(?:^|[\n;&|(])\s*git\b[^\n'\"]*\bpush\b"),
|
|
66
|
+
"git push — publishes to a remote you can't review overnight"),
|
|
67
|
+
("privilege", re.compile(r"(?:^|[\s|;&(])sudo\b|(?:^|[\s|;&(])su\s"),
|
|
68
|
+
"privilege escalation (sudo/su)"),
|
|
69
|
+
("remote-exec", re.compile(r"\b(curl|wget|fetch)\b[^\n|]*\|\s*(sudo\s+)?(sh|bash|zsh|python\d?|perl|ruby|node)\b"),
|
|
70
|
+
"pipes a network download into a shell/interpreter"),
|
|
71
|
+
("sys-install", re.compile(r"\b(apt|apt-get|yum|dnf|brew|pacman|snap)\b\s+(install|add|-S)\b"),
|
|
72
|
+
"system package install"),
|
|
73
|
+
# Language package installs fetch (and typically EXECUTE — setup.py, wheels,
|
|
74
|
+
# npm lifecycle scripts) arbitrary code from a public registry: typosquat /
|
|
75
|
+
# poisoned-package supply-chain risk. Gate them like a system install.
|
|
76
|
+
("pkg-install", re.compile(
|
|
77
|
+
r"(?:^|[\s;&|(])(?:"
|
|
78
|
+
r"(?:python[23]?\s+-m\s+)?pip[23]?\s+install" # pip / pip3 / python -m pip install
|
|
79
|
+
r"|uv\s+(?:pip\s+install|add)"
|
|
80
|
+
r"|npm\s+(?:install|i|ci|add)(?:\s|$)"
|
|
81
|
+
r"|(?:yarn|pnpm)\s+(?:add|install|i)(?:\s|$)"
|
|
82
|
+
r"|gem\s+install"
|
|
83
|
+
r"|cargo\s+(?:install|add)"
|
|
84
|
+
r"|go\s+(?:install|get)(?:\s|$)"
|
|
85
|
+
r"|poetry\s+(?:add|install)"
|
|
86
|
+
r"|conda\s+install"
|
|
87
|
+
r")", re.I),
|
|
88
|
+
"language package install — runs code from a registry (supply-chain risk)"),
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def classify_command(command: str) -> Verdict:
|
|
93
|
+
"""Classify a single bash command into allow / ask / block."""
|
|
94
|
+
cmd = command.strip()
|
|
95
|
+
if not cmd:
|
|
96
|
+
return Verdict("allow", "", "")
|
|
97
|
+
if re.search(r"\brm\b.*-\w*[rR]", cmd) and _RM_ROOT.search(cmd):
|
|
98
|
+
return Verdict("block", "recursive force-remove of a filesystem/home/mount root", "rm-rf-root")
|
|
99
|
+
for pat, rx, why in _BLOCK:
|
|
100
|
+
if rx.search(cmd):
|
|
101
|
+
return Verdict("block", why, pat)
|
|
102
|
+
for pat, rx, why in _ASK:
|
|
103
|
+
if rx.search(cmd):
|
|
104
|
+
return Verdict("ask", why, pat)
|
|
105
|
+
return Verdict("allow", "", "")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def pre_scan(text: str) -> list[Verdict]:
|
|
109
|
+
"""Classify each line of a goal / plan; return the non-allow verdicts, deduped
|
|
110
|
+
by pattern (block first, then ask). Used for the goal-start pre-flight, where
|
|
111
|
+
'ask' verdicts become the one up-front approval batch."""
|
|
112
|
+
seen: dict[str, Verdict] = {}
|
|
113
|
+
for line in text.splitlines():
|
|
114
|
+
v = classify_command(line)
|
|
115
|
+
if v.action != "allow" and v.pattern not in seen:
|
|
116
|
+
seen[v.pattern] = v
|
|
117
|
+
return sorted(seen.values(), key=lambda v: 0 if v.action == "block" else 1)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
# Network-intent hints in an objective/plan — used to surface "this goal may
|
|
121
|
+
# need network" in the up-front approval, so the offline-by-default goal sandbox
|
|
122
|
+
# can be opted online before the user leaves. Heuristic + conservative: a miss
|
|
123
|
+
# just means the run stays offline and a network step fails gracefully.
|
|
124
|
+
_NET_HINT = re.compile(
|
|
125
|
+
r"(?i)\b("
|
|
126
|
+
r"pip3?\s+install|(apt|apt-get|yum|dnf|brew|pacman|snap|conda)\s+(install|update|add|-S)|"
|
|
127
|
+
r"npm\s+(i|install|ci)|yarn\s+add|pnpm\s+add|poetry\s+add|cargo\s+add|go\s+get|"
|
|
128
|
+
r"download|install\s+(the\s+)?\w+\s+(package|librar|dependen|module|toolkit)|"
|
|
129
|
+
r"\bclone\b|\bcurl\b|\bwget\b|https?://|\bpypi\b|from\s+the\s+internet|"
|
|
130
|
+
r"fetch\s+(from|the)|\bnetwork\b|\binternet\b"
|
|
131
|
+
r")\b"
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def network_intent(text: str) -> str | None:
|
|
136
|
+
"""A short reason string if *text* (an objective or plan) looks like it needs
|
|
137
|
+
network access, else None. Feeds the goal pre-flight network approval."""
|
|
138
|
+
m = _NET_HINT.search(text or "")
|
|
139
|
+
return m.group(0).strip() if m else None
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""Chat sandbox: a lightweight Docker container that runs the agent's tools
|
|
2
|
+
in isolation while sharing the project directory.
|
|
3
|
+
|
|
4
|
+
Same Session interface as container.DockerSession (exec / stop), so the
|
|
5
|
+
existing build_session_registry() works unchanged — only the container
|
|
6
|
+
internals differ (no conda, no /testbed, just /workspace).
|
|
7
|
+
|
|
8
|
+
Lifecycle: created on /sandbox on, destroyed on /sandbox off or app exit.
|
|
9
|
+
|
|
10
|
+
The default image is minimal (debian:bookworm-slim). Override via
|
|
11
|
+
ROCKYCODE_SANDBOX_IMAGE if Docker Hub is unreachable.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import asyncio
|
|
16
|
+
import os
|
|
17
|
+
import shlex
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Optional
|
|
20
|
+
|
|
21
|
+
from rockycode.engine.tools import (
|
|
22
|
+
GLOB_MAX_PATHS,
|
|
23
|
+
GREP_MAX_MATCHES,
|
|
24
|
+
RISK,
|
|
25
|
+
SCHEMAS,
|
|
26
|
+
Tool,
|
|
27
|
+
_truncate,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
# python-preinstalled so the agent can run/test code offline and an approved
|
|
31
|
+
# `pip install` doesn't have to apt-bootstrap the toolchain (the old .deb pile).
|
|
32
|
+
# Override with ROCKYCODE_SANDBOX_IMAGE for non-python projects.
|
|
33
|
+
DEFAULT_SANDBOX_IMAGE = "python:3.12-slim"
|
|
34
|
+
EXEC_TIMEOUT_S = 120
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class ChatSandbox:
|
|
38
|
+
"""One lightweight container per chat session; project dir mounted at /workspace."""
|
|
39
|
+
|
|
40
|
+
def __init__(self, container_id: str, workdir: Path) -> None:
|
|
41
|
+
self.container_id = container_id
|
|
42
|
+
self.workdir = workdir
|
|
43
|
+
self._running = True
|
|
44
|
+
|
|
45
|
+
@classmethod
|
|
46
|
+
async def start(cls, workdir: Path, *, image: str | None = None,
|
|
47
|
+
network: bool = False) -> "ChatSandbox":
|
|
48
|
+
img = image or os.getenv("ROCKYCODE_SANDBOX_IMAGE", DEFAULT_SANDBOX_IMAGE)
|
|
49
|
+
wd = workdir.resolve()
|
|
50
|
+
# network=False → `--network none`: no egress at all. This is the SAFE
|
|
51
|
+
# DEFAULT for every surface (no exfiltration, no apt/pip rabbit-holes);
|
|
52
|
+
# the container can only touch the mounted workspace. Callers that need
|
|
53
|
+
# the network (goal mode when the plan implies it, chat/exec on request)
|
|
54
|
+
# pass network=True explicitly and say so out loud.
|
|
55
|
+
net_args = [] if network else ["--network", "none"]
|
|
56
|
+
proc = await asyncio.create_subprocess_exec(
|
|
57
|
+
"docker", "run", "-d", "--rm",
|
|
58
|
+
*net_args,
|
|
59
|
+
"-v", f"{wd}:/workspace:rw",
|
|
60
|
+
"-w", "/workspace",
|
|
61
|
+
img, "tail", "-f", "/dev/null",
|
|
62
|
+
stdout=asyncio.subprocess.PIPE,
|
|
63
|
+
stderr=asyncio.subprocess.PIPE,
|
|
64
|
+
)
|
|
65
|
+
out, err = await proc.communicate()
|
|
66
|
+
if proc.returncode != 0:
|
|
67
|
+
msg = err.decode(errors="replace").strip()
|
|
68
|
+
hint = ""
|
|
69
|
+
if "pull access denied" in msg.lower() or "not found" in msg.lower():
|
|
70
|
+
hint = (
|
|
71
|
+
f"\n [hint] this image may not be pulled yet. "
|
|
72
|
+
f"try: docker pull {img}"
|
|
73
|
+
)
|
|
74
|
+
elif "Cannot connect" in msg or "Is the docker daemon" in msg:
|
|
75
|
+
hint = "\n [hint] docker daemon is not running or not reachable."
|
|
76
|
+
raise RuntimeError(f"sandbox start failed for {img}: {msg}{hint}")
|
|
77
|
+
return cls(out.decode().strip(), wd)
|
|
78
|
+
|
|
79
|
+
async def exec(
|
|
80
|
+
self,
|
|
81
|
+
script: str,
|
|
82
|
+
*,
|
|
83
|
+
stdin: Optional[bytes] = None,
|
|
84
|
+
timeout: int = EXEC_TIMEOUT_S,
|
|
85
|
+
) -> tuple[str, int]:
|
|
86
|
+
if not self._running:
|
|
87
|
+
return "[error] sandbox has been stopped", 1
|
|
88
|
+
proc = await asyncio.create_subprocess_exec(
|
|
89
|
+
"docker", "exec", "-i", self.container_id, "bash", "-c", script,
|
|
90
|
+
stdin=asyncio.subprocess.PIPE if stdin is not None else asyncio.subprocess.DEVNULL,
|
|
91
|
+
stdout=asyncio.subprocess.PIPE,
|
|
92
|
+
stderr=asyncio.subprocess.STDOUT,
|
|
93
|
+
)
|
|
94
|
+
try:
|
|
95
|
+
out, _ = await asyncio.wait_for(proc.communicate(input=stdin), timeout=timeout)
|
|
96
|
+
except asyncio.TimeoutError:
|
|
97
|
+
proc.kill()
|
|
98
|
+
return f"[timeout] command exceeded {timeout}s and was killed", 124
|
|
99
|
+
return out.decode(errors="replace"), proc.returncode or 0
|
|
100
|
+
|
|
101
|
+
async def stop(self) -> None:
|
|
102
|
+
self._running = False
|
|
103
|
+
proc = await asyncio.create_subprocess_exec(
|
|
104
|
+
"docker", "rm", "-f", self.container_id,
|
|
105
|
+
stdout=asyncio.subprocess.DEVNULL,
|
|
106
|
+
stderr=asyncio.subprocess.DEVNULL,
|
|
107
|
+
)
|
|
108
|
+
await proc.wait()
|
|
109
|
+
|
|
110
|
+
@property
|
|
111
|
+
def is_running(self) -> bool:
|
|
112
|
+
return self._running
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
# ── tool wrappers (same logic as container.py, minus conda / SWE-bench paths) ──
|
|
116
|
+
|
|
117
|
+
async def _bash(sandbox: ChatSandbox, command: str) -> str:
|
|
118
|
+
out, code = await sandbox.exec(command)
|
|
119
|
+
if out.startswith("[timeout]"):
|
|
120
|
+
return out
|
|
121
|
+
return _truncate(f"[exit {code}]\n{out}" if out else f"[exit {code}]")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
async def _read_file(sandbox: ChatSandbox, path: str, offset=None, limit=None) -> str:
|
|
125
|
+
q = shlex.quote(path)
|
|
126
|
+
# cat -n before sed: absolute line numbers survive windowed reads (same
|
|
127
|
+
# contract as the local and bench-container read_file).
|
|
128
|
+
window = ""
|
|
129
|
+
if offset or limit:
|
|
130
|
+
start = max(int(offset or 1), 1)
|
|
131
|
+
end = str(start + int(limit) - 1) if limit else "$"
|
|
132
|
+
window = f" | sed -n '{start},{end}p'"
|
|
133
|
+
out, _code = await sandbox.exec(
|
|
134
|
+
f'if [ -d {q} ]; then echo "[directory] {path}"; ls -p {q}; '
|
|
135
|
+
f'elif [ -e {q} ]; then cat -n {q}{window}; '
|
|
136
|
+
f'else echo "[error] file not found: {path}"; exit 1; fi'
|
|
137
|
+
)
|
|
138
|
+
return _truncate(out.rstrip("\n"))
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
async def _write_file(sandbox: ChatSandbox, path: str, content: str) -> str:
|
|
142
|
+
q = shlex.quote(path)
|
|
143
|
+
out, code = await sandbox.exec(
|
|
144
|
+
f'mkdir -p "$(dirname {q})" && cat > {q}', stdin=content.encode()
|
|
145
|
+
)
|
|
146
|
+
if code != 0:
|
|
147
|
+
return f"[error] write failed: {out.strip()}"
|
|
148
|
+
return f"[ok] wrote {len(content)} chars to {path}"
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
async def _edit_file(sandbox: ChatSandbox, path: str, old_string: str, new_string: str) -> str:
|
|
152
|
+
q = shlex.quote(path)
|
|
153
|
+
text, code = await sandbox.exec(f"cat {q}")
|
|
154
|
+
if code != 0:
|
|
155
|
+
return f"[error] file not found: {path}"
|
|
156
|
+
n = text.count(old_string)
|
|
157
|
+
if n == 0:
|
|
158
|
+
return "[error] old_string not found in file. read the file again — it may have changed."
|
|
159
|
+
if n > 1:
|
|
160
|
+
return f"[error] old_string appears {n} times; include more surrounding context to make it unique."
|
|
161
|
+
return await _write_file(sandbox, path, text.replace(old_string, new_string, 1))
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
async def _grep(sandbox: ChatSandbox, pattern: str, path: str = ".", include: str | None = None) -> str:
|
|
165
|
+
inc = f" --include={shlex.quote(include)}" if include else ""
|
|
166
|
+
excludes = " ".join(f"--exclude-dir={d}" for d in (".git", "__pycache__", ".venv", "node_modules"))
|
|
167
|
+
cmd = f"grep -rnIE {excludes}{inc} -e {shlex.quote(pattern)} {shlex.quote(path)}"
|
|
168
|
+
out, code = await sandbox.exec(cmd)
|
|
169
|
+
out = out.rstrip("\n")
|
|
170
|
+
if code == 1:
|
|
171
|
+
return "[no matches]"
|
|
172
|
+
if code != 0:
|
|
173
|
+
return f"[error] grep failed (bad pattern or path): {out[:200]}"
|
|
174
|
+
lines = out.splitlines()
|
|
175
|
+
if len(lines) > GREP_MAX_MATCHES:
|
|
176
|
+
lines = lines[:GREP_MAX_MATCHES]
|
|
177
|
+
lines.append(f"[truncated at {GREP_MAX_MATCHES} matches — narrow the pattern]")
|
|
178
|
+
return _truncate("\n".join(lines))
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
_GLOB_PY = (
|
|
182
|
+
"import glob, sys; "
|
|
183
|
+
f"hits = sorted(glob.glob(sys.argv[1], recursive=True))[:{GLOB_MAX_PATHS}]; "
|
|
184
|
+
"print('\\n'.join(hits) if hits else '[no matches]')"
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
async def _glob(sandbox: ChatSandbox, pattern: str) -> str:
|
|
189
|
+
out, code = await sandbox.exec(f"python3 -c {shlex.quote(_GLOB_PY)} {shlex.quote(pattern)}")
|
|
190
|
+
if code != 0:
|
|
191
|
+
return f"[error] glob failed: {out.strip()[:200]}"
|
|
192
|
+
return _truncate(out.rstrip("\n"))
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def build_sandbox_registry(sandbox: ChatSandbox, *, extras: dict[str, Tool] | None = None) -> dict[str, Tool]:
|
|
196
|
+
"""Build a tool registry bound to a chat sandbox (same schema, capsuled execution).
|
|
197
|
+
|
|
198
|
+
extras are merged on top — e.g. web_search/web_research/web_fetch that run on the
|
|
199
|
+
host, not inside the container.
|
|
200
|
+
"""
|
|
201
|
+
fns = {
|
|
202
|
+
"bash": lambda command: _bash(sandbox, command),
|
|
203
|
+
"read_file": lambda path, offset=None, limit=None: _read_file(sandbox, path, offset, limit),
|
|
204
|
+
"write_file": lambda path, content: _write_file(sandbox, path, content),
|
|
205
|
+
"edit_file": lambda path, old_string, new_string: _edit_file(
|
|
206
|
+
sandbox, path, old_string, new_string
|
|
207
|
+
),
|
|
208
|
+
"grep": lambda pattern, path=".", include=None: _grep(sandbox, pattern, path, include),
|
|
209
|
+
"glob": lambda pattern: _glob(sandbox, pattern),
|
|
210
|
+
}
|
|
211
|
+
# Same risk tiers as the host registry (read_file/grep/glob = "safe") so the
|
|
212
|
+
# loop's read-parallelism kicks in inside the sandbox too — goal mode and
|
|
213
|
+
# `chat --sandbox` both build their registry here. Without this the reads
|
|
214
|
+
# default to "risky" and every explore batch runs serially.
|
|
215
|
+
reg = {name: Tool(name=name, schema=SCHEMAS[name], fn=fn, risk=RISK.get(name, "risky"))
|
|
216
|
+
for name, fn in fns.items()}
|
|
217
|
+
if extras:
|
|
218
|
+
reg.update(extras)
|
|
219
|
+
return reg
|