rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"""Context compaction: keep long sessions inside the model's window.
|
|
2
|
+
|
|
3
|
+
Two stages, cheapest first:
|
|
4
|
+
|
|
5
|
+
1. prune — deterministic, free. Old tool outputs (the bulk of any agent
|
|
6
|
+
context) are stubbed down to a short head; the assistant's own messages
|
|
7
|
+
and tool_calls survive, so the model still sees *what* it did, just not
|
|
8
|
+
every byte it once read.
|
|
9
|
+
2. summarize — the recurrent state. If pruning can't get under the limit,
|
|
10
|
+
one non-streaming API call folds everything except a recent tail into a
|
|
11
|
+
dense state document, and history is rebuilt as [system, state, tail].
|
|
12
|
+
The call sends the *unmodified* history plus one instruction message, so
|
|
13
|
+
DeepSeek's automatic prefix cache makes its input nearly free.
|
|
14
|
+
|
|
15
|
+
Token math is deliberately conservative (3 chars ≈ 1 token): compacting a
|
|
16
|
+
little early costs a few cache misses; compacting late kills the turn.
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
CHARS_PER_TOKEN = 3
|
|
21
|
+
MSG_OVERHEAD_TOKENS = 8
|
|
22
|
+
|
|
23
|
+
KEEP_RECENT_TOKENS = 8_000 # protected tail budget (estimated tokens)
|
|
24
|
+
PRUNE_KEEP_CHARS = 200 # head kept when a tool output is stubbed
|
|
25
|
+
PRUNE_STUB = "… [old output dropped during compaction — re-run the tool if needed]"
|
|
26
|
+
SUMMARY_MAX_TOKENS = 4096
|
|
27
|
+
|
|
28
|
+
SUMMARIZE_INSTRUCTION = """\
|
|
29
|
+
Context is nearly full. Before continuing, write a compact state document
|
|
30
|
+
for this session so far; older messages will be dropped and replaced by it.
|
|
31
|
+
|
|
32
|
+
Cover, with concrete file paths, names, and line references:
|
|
33
|
+
1. Task — what we are trying to accomplish, in one or two sentences.
|
|
34
|
+
2. State — what has been done and learned so far (files read/edited, key
|
|
35
|
+
findings, commands run and their results).
|
|
36
|
+
3. Failures — approaches that did not work, so they are not retried.
|
|
37
|
+
4. Next — the immediate plan.
|
|
38
|
+
|
|
39
|
+
Write only the state document, no preamble. Be dense; facts over prose.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def estimate_msg_tokens(msg: dict) -> int:
|
|
44
|
+
chars = len(msg.get("content") or "")
|
|
45
|
+
for tc in msg.get("tool_calls") or []:
|
|
46
|
+
fn = tc.get("function", {})
|
|
47
|
+
chars += len(fn.get("name", "")) + len(fn.get("arguments", ""))
|
|
48
|
+
return MSG_OVERHEAD_TOKENS + chars // CHARS_PER_TOKEN
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def estimate_tokens(messages: list[dict]) -> int:
|
|
52
|
+
return sum(estimate_msg_tokens(m) for m in messages)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def tail_start(history: list[dict], budget_tokens: int) -> int:
|
|
56
|
+
"""Index where the protected recent tail begins.
|
|
57
|
+
|
|
58
|
+
Never 0 (the system prompt is handled separately), always compresses at
|
|
59
|
+
least the older half of the conversation, and never starts on a `tool`
|
|
60
|
+
message — a tool result without its assistant tool_calls parent is an
|
|
61
|
+
invalid conversation, so the tail grows backward to include the parent.
|
|
62
|
+
"""
|
|
63
|
+
i = len(history)
|
|
64
|
+
spent = 0
|
|
65
|
+
while i > 1:
|
|
66
|
+
cost = estimate_msg_tokens(history[i - 1])
|
|
67
|
+
if spent and spent + cost > budget_tokens:
|
|
68
|
+
break
|
|
69
|
+
spent += cost
|
|
70
|
+
i -= 1
|
|
71
|
+
floor = 1 + (len(history) - 1) // 2
|
|
72
|
+
if i < floor:
|
|
73
|
+
i = floor
|
|
74
|
+
while 1 < i < len(history) and history[i]["role"] == "tool":
|
|
75
|
+
i -= 1
|
|
76
|
+
return i
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _prune_gain_chars(content: str) -> int:
|
|
80
|
+
return len(content) - PRUNE_KEEP_CHARS - len(PRUNE_STUB) - 1
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def prune_savings(history: list[dict], protect_from: int) -> int:
|
|
84
|
+
"""Estimated tokens that prune_tool_outputs would free. No mutation."""
|
|
85
|
+
saved = 0
|
|
86
|
+
for msg in history[1:protect_from]:
|
|
87
|
+
if msg.get("role") == "tool":
|
|
88
|
+
gain = _prune_gain_chars(msg.get("content") or "")
|
|
89
|
+
if gain > 0:
|
|
90
|
+
saved += gain // CHARS_PER_TOKEN
|
|
91
|
+
return saved
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def prune_tool_outputs(history: list[dict], protect_from: int) -> int:
|
|
95
|
+
"""Stub bulky tool outputs in history[1:protect_from], in place.
|
|
96
|
+
|
|
97
|
+
Returns the number of messages stubbed. Deterministic given the same
|
|
98
|
+
history and protect_from, so a trajectory's compaction record is enough
|
|
99
|
+
to replay it.
|
|
100
|
+
"""
|
|
101
|
+
n = 0
|
|
102
|
+
for msg in history[1:protect_from]:
|
|
103
|
+
if msg.get("role") != "tool":
|
|
104
|
+
continue
|
|
105
|
+
content = msg.get("content") or ""
|
|
106
|
+
if _prune_gain_chars(content) <= 0:
|
|
107
|
+
continue
|
|
108
|
+
msg["content"] = content[:PRUNE_KEEP_CHARS] + "\n" + PRUNE_STUB
|
|
109
|
+
n += 1
|
|
110
|
+
return n
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
MAX_MSG_CHARS = 40_000 # ~13k tokens; a single message above this is truncated
|
|
114
|
+
# head+tail, so one giant paste / tool result can't defeat
|
|
115
|
+
# compaction or overflow the window.
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def truncate_oversized(history: list[dict], max_chars: int = MAX_MSG_CHARS) -> int:
|
|
119
|
+
"""Truncate any single message whose text content exceeds *max_chars* to a
|
|
120
|
+
head + tail with an elision marker, in place. Skips the system prompt.
|
|
121
|
+
Returns the count truncated.
|
|
122
|
+
|
|
123
|
+
The backstop for what prune and summarize miss: prune only shrinks `tool`
|
|
124
|
+
messages and summarize keeps the recent tail verbatim, so an oversized
|
|
125
|
+
NON-tool message (e.g. a pasted 300k-char log that tail_start always keeps)
|
|
126
|
+
survives both and keeps every step over the limit. This bounds it.
|
|
127
|
+
"""
|
|
128
|
+
n = 0
|
|
129
|
+
keep = max_chars // 2 - 60
|
|
130
|
+
for msg in history[1:]: # never the system prompt
|
|
131
|
+
content = msg.get("content")
|
|
132
|
+
if not isinstance(content, str) or len(content) <= max_chars:
|
|
133
|
+
continue
|
|
134
|
+
elided = len(content) - 2 * keep
|
|
135
|
+
msg["content"] = (
|
|
136
|
+
content[:keep]
|
|
137
|
+
+ f"\n… [{elided:,} chars elided to fit the context window] …\n"
|
|
138
|
+
+ content[-keep:]
|
|
139
|
+
)
|
|
140
|
+
n += 1
|
|
141
|
+
return n
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
async def summarize(client, model: str, history: list[dict], tools: list[dict]) -> tuple[str, dict]:
|
|
145
|
+
"""One non-streaming call: history + instruction → state document.
|
|
146
|
+
|
|
147
|
+
`tools` is passed through (with tool_choice="none") so the request body
|
|
148
|
+
matches the main loop's shape and the prefix cache can hit; thinking is
|
|
149
|
+
off — a summary needs no chain of thought.
|
|
150
|
+
"""
|
|
151
|
+
resp = await client.chat.completions.create(
|
|
152
|
+
model=model,
|
|
153
|
+
messages=history + [{"role": "user", "content": SUMMARIZE_INSTRUCTION}],
|
|
154
|
+
tools=tools,
|
|
155
|
+
tool_choice="none",
|
|
156
|
+
max_tokens=SUMMARY_MAX_TOKENS,
|
|
157
|
+
stream=False,
|
|
158
|
+
extra_body={"thinking": {"type": "disabled"}},
|
|
159
|
+
)
|
|
160
|
+
summary = (resp.choices[0].message.content or "").strip()
|
|
161
|
+
usage: dict = {}
|
|
162
|
+
if resp.usage is not None:
|
|
163
|
+
try:
|
|
164
|
+
u = resp.usage.model_dump()
|
|
165
|
+
except AttributeError:
|
|
166
|
+
u = dict(resp.usage)
|
|
167
|
+
usage = {k: v for k, v in u.items() if isinstance(v, int)}
|
|
168
|
+
return summary, usage
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def state_message(summary: str) -> dict:
|
|
172
|
+
return {
|
|
173
|
+
"role": "user",
|
|
174
|
+
"content": (
|
|
175
|
+
"[context compacted] Older messages were compressed into this "
|
|
176
|
+
"state document:\n\n"
|
|
177
|
+
f"{summary}\n\n"
|
|
178
|
+
"Continue the task from this state. The most recent messages "
|
|
179
|
+
"follow unmodified — do not redo work recorded as done above."
|
|
180
|
+
),
|
|
181
|
+
}
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""Run engine tools inside a SWE-bench task container.
|
|
2
|
+
|
|
3
|
+
A *session* is anything with `exec(script, stdin, timeout)`. DockerSession
|
|
4
|
+
talks to a long-lived container via `docker exec`; LocalSession runs the
|
|
5
|
+
same scripts in a local directory so the glue is testable without Docker.
|
|
6
|
+
|
|
7
|
+
SWE-bench image layout (constant across the official images):
|
|
8
|
+
repo checked out at /testbed at the task's base_commit
|
|
9
|
+
conda env "testbed" at /opt/miniconda3 with the repo's deps installed
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import asyncio
|
|
14
|
+
import shlex
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Optional
|
|
17
|
+
|
|
18
|
+
from rockycode.engine.tools import (
|
|
19
|
+
GLOB_MAX_PATHS,
|
|
20
|
+
GREP_MAX_MATCHES,
|
|
21
|
+
RISK,
|
|
22
|
+
SCHEMAS,
|
|
23
|
+
Tool,
|
|
24
|
+
_truncate,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
# Generous: real test suites run under emulation on Apple Silicon.
|
|
28
|
+
EXEC_TIMEOUT_S = 300
|
|
29
|
+
|
|
30
|
+
# Activate the task env and land in the repo before every command.
|
|
31
|
+
_WRAP = "source /opt/miniconda3/bin/activate testbed 2>/dev/null; cd /testbed 2>/dev/null; "
|
|
32
|
+
|
|
33
|
+
# `git add -A` respects .gitignore, so build junk stays out; --cached diff
|
|
34
|
+
# then captures edits AND new files the agent created.
|
|
35
|
+
GIT_DIFF_SCRIPT = "git add -A >/dev/null 2>&1; git -c core.fileMode=false diff --cached"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class DockerSession:
|
|
39
|
+
"""One long-lived container per task; commands go through docker exec."""
|
|
40
|
+
|
|
41
|
+
def __init__(self, container_id: str) -> None:
|
|
42
|
+
self.container_id = container_id
|
|
43
|
+
|
|
44
|
+
@classmethod
|
|
45
|
+
async def start(cls, image: str, *, platform: str = "linux/amd64") -> "DockerSession":
|
|
46
|
+
proc = await asyncio.create_subprocess_exec(
|
|
47
|
+
"docker", "run", "-d", "--platform", platform, image, "tail", "-f", "/dev/null",
|
|
48
|
+
stdout=asyncio.subprocess.PIPE,
|
|
49
|
+
stderr=asyncio.subprocess.PIPE,
|
|
50
|
+
)
|
|
51
|
+
out, err = await proc.communicate()
|
|
52
|
+
if proc.returncode != 0:
|
|
53
|
+
raise RuntimeError(f"docker run failed for {image}: {err.decode(errors='replace').strip()}")
|
|
54
|
+
return cls(out.decode().strip())
|
|
55
|
+
|
|
56
|
+
async def exec(
|
|
57
|
+
self,
|
|
58
|
+
script: str,
|
|
59
|
+
*,
|
|
60
|
+
stdin: Optional[bytes] = None,
|
|
61
|
+
timeout: int = EXEC_TIMEOUT_S,
|
|
62
|
+
) -> tuple[str, int]:
|
|
63
|
+
proc = await asyncio.create_subprocess_exec(
|
|
64
|
+
"docker", "exec", "-i", self.container_id, "bash", "-c", _WRAP + script,
|
|
65
|
+
stdin=asyncio.subprocess.PIPE if stdin is not None else asyncio.subprocess.DEVNULL,
|
|
66
|
+
stdout=asyncio.subprocess.PIPE,
|
|
67
|
+
stderr=asyncio.subprocess.STDOUT,
|
|
68
|
+
)
|
|
69
|
+
try:
|
|
70
|
+
out, _ = await asyncio.wait_for(proc.communicate(input=stdin), timeout=timeout)
|
|
71
|
+
except asyncio.TimeoutError:
|
|
72
|
+
proc.kill()
|
|
73
|
+
# NB: kills the host-side `docker exec`; the in-container process
|
|
74
|
+
# may linger. Acceptable for v1 — the container dies after the task.
|
|
75
|
+
return f"[timeout] command exceeded {timeout}s and was killed", 124
|
|
76
|
+
return out.decode(errors="replace"), proc.returncode or 0
|
|
77
|
+
|
|
78
|
+
async def stop(self) -> None:
|
|
79
|
+
proc = await asyncio.create_subprocess_exec(
|
|
80
|
+
"docker", "rm", "-f", self.container_id,
|
|
81
|
+
stdout=asyncio.subprocess.DEVNULL,
|
|
82
|
+
stderr=asyncio.subprocess.DEVNULL,
|
|
83
|
+
)
|
|
84
|
+
await proc.wait()
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class LocalSession:
|
|
88
|
+
"""Same interface, local bash in a directory. For tests."""
|
|
89
|
+
|
|
90
|
+
def __init__(self, workdir: Path) -> None:
|
|
91
|
+
self.workdir = workdir
|
|
92
|
+
|
|
93
|
+
async def exec(
|
|
94
|
+
self,
|
|
95
|
+
script: str,
|
|
96
|
+
*,
|
|
97
|
+
stdin: Optional[bytes] = None,
|
|
98
|
+
timeout: int = EXEC_TIMEOUT_S,
|
|
99
|
+
) -> tuple[str, int]:
|
|
100
|
+
proc = await asyncio.create_subprocess_exec(
|
|
101
|
+
"bash", "-c", script,
|
|
102
|
+
cwd=self.workdir,
|
|
103
|
+
stdin=asyncio.subprocess.PIPE if stdin is not None else asyncio.subprocess.DEVNULL,
|
|
104
|
+
stdout=asyncio.subprocess.PIPE,
|
|
105
|
+
stderr=asyncio.subprocess.STDOUT,
|
|
106
|
+
)
|
|
107
|
+
try:
|
|
108
|
+
out, _ = await asyncio.wait_for(proc.communicate(input=stdin), timeout=timeout)
|
|
109
|
+
except asyncio.TimeoutError:
|
|
110
|
+
proc.kill()
|
|
111
|
+
return f"[timeout] command exceeded {timeout}s and was killed", 124
|
|
112
|
+
return out.decode(errors="replace"), proc.returncode or 0
|
|
113
|
+
|
|
114
|
+
async def stop(self) -> None:
|
|
115
|
+
pass
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
async def _bash(session, command: str) -> str:
|
|
119
|
+
out, code = await session.exec(command)
|
|
120
|
+
if out.startswith("[timeout]"):
|
|
121
|
+
return out
|
|
122
|
+
return _truncate(f"[exit {code}]\n{out}" if out else f"[exit {code}]")
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
async def _read_file(session, path: str, offset=None, limit=None) -> str:
|
|
126
|
+
q = shlex.quote(path)
|
|
127
|
+
# cat -n BEFORE sed: numbering happens on the whole file, so a windowed
|
|
128
|
+
# read shows absolute line numbers the model can feed straight to edits.
|
|
129
|
+
window = ""
|
|
130
|
+
if offset or limit:
|
|
131
|
+
start = max(int(offset or 1), 1)
|
|
132
|
+
end = str(start + int(limit) - 1) if limit else "$"
|
|
133
|
+
window = f" | sed -n '{start},{end}p'"
|
|
134
|
+
out, _code = await session.exec(
|
|
135
|
+
f'if [ -d {q} ]; then echo "[directory] {path}"; ls -p {q}; '
|
|
136
|
+
f'elif [ -e {q} ]; then cat -n {q}{window}; '
|
|
137
|
+
f'else echo "[error] file not found: {path}"; exit 1; fi'
|
|
138
|
+
)
|
|
139
|
+
return _truncate(out.rstrip("\n"))
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
async def _write_file(session, path: str, content: str) -> str:
|
|
143
|
+
q = shlex.quote(path)
|
|
144
|
+
out, code = await session.exec(
|
|
145
|
+
f'mkdir -p "$(dirname {q})" && cat > {q}', stdin=content.encode()
|
|
146
|
+
)
|
|
147
|
+
if code != 0:
|
|
148
|
+
return f"[error] write failed: {out.strip()}"
|
|
149
|
+
return f"[ok] wrote {len(content)} chars to {path}"
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
async def _edit_file(session, path: str, old_string: str, new_string: str) -> str:
|
|
153
|
+
q = shlex.quote(path)
|
|
154
|
+
text, code = await session.exec(f"cat {q}")
|
|
155
|
+
if code != 0:
|
|
156
|
+
return f"[error] file not found: {path}"
|
|
157
|
+
n = text.count(old_string)
|
|
158
|
+
if n == 0:
|
|
159
|
+
return "[error] old_string not found in file. read the file again — it may have changed."
|
|
160
|
+
if n > 1:
|
|
161
|
+
return f"[error] old_string appears {n} times; include more surrounding context to make it unique."
|
|
162
|
+
return await _write_file(session, path, text.replace(old_string, new_string, 1))
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
async def _grep(session, pattern: str, path: str = ".", include: str | None = None) -> str:
|
|
166
|
+
inc = f" --include={shlex.quote(include)}" if include else ""
|
|
167
|
+
# -I skips binaries; junk dirs excluded to match the local tool's behavior.
|
|
168
|
+
# No pipe here: it would mask grep's exit code (1 = no match, 2 = bad
|
|
169
|
+
# pattern); matches are capped host-side instead.
|
|
170
|
+
excludes = " ".join(f"--exclude-dir={d}" for d in (".git", "__pycache__", ".venv", "node_modules"))
|
|
171
|
+
cmd = f"grep -rnIE {excludes}{inc} -e {shlex.quote(pattern)} {shlex.quote(path)}"
|
|
172
|
+
out, code = await session.exec(cmd)
|
|
173
|
+
out = out.rstrip("\n")
|
|
174
|
+
if code == 1:
|
|
175
|
+
return "[no matches]"
|
|
176
|
+
if code != 0:
|
|
177
|
+
return f"[error] grep failed (bad pattern or path): {out[:200]}"
|
|
178
|
+
lines = out.splitlines()
|
|
179
|
+
if len(lines) > GREP_MAX_MATCHES:
|
|
180
|
+
lines = lines[:GREP_MAX_MATCHES]
|
|
181
|
+
lines.append(f"[truncated at {GREP_MAX_MATCHES} matches — narrow the pattern]")
|
|
182
|
+
return _truncate("\n".join(lines))
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
_GLOB_PY = (
|
|
186
|
+
"import glob, sys; "
|
|
187
|
+
f"hits = sorted(glob.glob(sys.argv[1], recursive=True))[:{GLOB_MAX_PATHS}]; "
|
|
188
|
+
"print('\\n'.join(hits) if hits else '[no matches]')"
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
async def _glob(session, pattern: str) -> str:
|
|
193
|
+
out, code = await session.exec(f"python -c {shlex.quote(_GLOB_PY)} {shlex.quote(pattern)}")
|
|
194
|
+
if code != 0:
|
|
195
|
+
return f"[error] glob failed: {out.strip()[:200]}"
|
|
196
|
+
return _truncate(out.rstrip("\n"))
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def build_session_registry(session) -> dict[str, Tool]:
|
|
200
|
+
"""The same tools the chat TUI has, executing inside the session."""
|
|
201
|
+
fns = {
|
|
202
|
+
"bash": lambda command: _bash(session, command),
|
|
203
|
+
"read_file": lambda path, offset=None, limit=None: _read_file(session, path, offset, limit),
|
|
204
|
+
"write_file": lambda path, content: _write_file(session, path, content),
|
|
205
|
+
"edit_file": lambda path, old_string, new_string: _edit_file(
|
|
206
|
+
session, path, old_string, new_string
|
|
207
|
+
),
|
|
208
|
+
"grep": lambda pattern, path=".", include=None: _grep(session, pattern, path, include),
|
|
209
|
+
"glob": lambda pattern: _glob(session, pattern),
|
|
210
|
+
}
|
|
211
|
+
return {
|
|
212
|
+
name: Tool(name=name, schema=SCHEMAS[name], fn=fn, risk=RISK.get(name, "risky"))
|
|
213
|
+
for name, fn in fns.items()
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
async def extract_patch(session) -> str:
|
|
218
|
+
"""The agent's work as a git patch — this is the SWE-bench prediction."""
|
|
219
|
+
out, code = await session.exec(GIT_DIFF_SCRIPT)
|
|
220
|
+
if code != 0:
|
|
221
|
+
return ""
|
|
222
|
+
patch = out
|
|
223
|
+
if patch and not patch.endswith("\n"):
|
|
224
|
+
patch += "\n"
|
|
225
|
+
return patch
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""rocky's effort dial: off | high | xhigh | max.
|
|
2
|
+
|
|
3
|
+
The dial is rocky-owned and provider-neutral. Providers name their tiers
|
|
4
|
+
differently — DeepSeek V4 accepts only high|max, OpenAI-style models use
|
|
5
|
+
low|medium|high — so each provider gets a clamp table here and the dial value
|
|
6
|
+
never goes on the wire directly. Unknown values pass through untouched, so a
|
|
7
|
+
future model registry can bring its own tiers without touching callers.
|
|
8
|
+
|
|
9
|
+
`off` never reaches a request body: it means thinking disabled, and callers
|
|
10
|
+
keep the last effort so switching back on restores it.
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
EFFORT_LEVELS = ("off", "high", "xhigh", "max") # the user-facing dial
|
|
15
|
+
CLI_EFFORTS = ("high", "xhigh", "max") # flag values (`off` = --no-thinking)
|
|
16
|
+
|
|
17
|
+
# Per-reasoning-policy clamp: the dial value → what this provider accepts.
|
|
18
|
+
# DeepSeek V4 knows only high|max (xhigh clamps to max); OpenAI-style maps the
|
|
19
|
+
# rocky dial to low|medium|high. A provider's profile names its policy.
|
|
20
|
+
_CLAMP = {
|
|
21
|
+
"deepseek": {"high": "high", "xhigh": "max", "max": "max"},
|
|
22
|
+
"openai": {"high": "medium", "xhigh": "high", "max": "high"},
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def to_deepseek(effort: str) -> str: # kept for existing callers
|
|
27
|
+
return _CLAMP["deepseek"].get(effort, effort)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def build_extra_body(thinking: bool, effort: str, reasoning: str = "deepseek") -> dict:
|
|
31
|
+
"""The reasoning fields for this request, shaped by the provider's policy:
|
|
32
|
+
|
|
33
|
+
- "deepseek": `{thinking: {type}, reasoning_effort}` (DeepSeek's own knob)
|
|
34
|
+
- "openai": bare `{reasoning_effort}` (OpenAI-style), only when thinking
|
|
35
|
+
- "none": nothing — the provider has no reasoning param
|
|
36
|
+
|
|
37
|
+
The OpenAI SDK carries these via `extra_body`; unknown keys a provider
|
|
38
|
+
ignores are harmless, but "none" keeps the body clean for strict endpoints.
|
|
39
|
+
"""
|
|
40
|
+
if not thinking or reasoning == "none":
|
|
41
|
+
return {"thinking": {"type": "disabled"}} if reasoning == "deepseek" else {}
|
|
42
|
+
clamp = _CLAMP.get(reasoning, {})
|
|
43
|
+
tier = clamp.get(effort, effort)
|
|
44
|
+
if reasoning == "deepseek":
|
|
45
|
+
return {"thinking": {"type": "enabled"}, "reasoning_effort": tier}
|
|
46
|
+
return {"reasoning_effort": tier}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Engine events: the single contract between the agent core and everything
|
|
2
|
+
that watches it (TUI transcript, Rocky pet, trajectory logger, game layer).
|
|
3
|
+
|
|
4
|
+
Subscribers must tolerate unknown event types — match on isinstance and
|
|
5
|
+
ignore what you don't handle, so new events never break old UIs.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from enum import Enum
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class AgentState(str, Enum):
|
|
14
|
+
"""What Rocky is doing right now. Drives the pet animation."""
|
|
15
|
+
|
|
16
|
+
IDLE = "idle"
|
|
17
|
+
THINKING = "thinking"
|
|
18
|
+
RESPONDING = "responding"
|
|
19
|
+
TOOL = "tool"
|
|
20
|
+
COMPACTING = "compacting"
|
|
21
|
+
AMAZED = "amazed"
|
|
22
|
+
ERROR = "error"
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class Event:
|
|
27
|
+
pass
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class StateChanged(Event):
|
|
32
|
+
state: AgentState
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class TurnStarted(Event):
|
|
37
|
+
user_message: str
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class ThinkingDelta(Event):
|
|
42
|
+
"""A chunk of DeepSeek's reasoning_content stream."""
|
|
43
|
+
|
|
44
|
+
text: str
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass
|
|
48
|
+
class TextDelta(Event):
|
|
49
|
+
"""A chunk of the assistant's visible reply."""
|
|
50
|
+
|
|
51
|
+
text: str
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class ToolStarted(Event):
|
|
56
|
+
call_id: str
|
|
57
|
+
tool: str
|
|
58
|
+
args: dict = field(default_factory=dict)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class ToolFinished(Event):
|
|
63
|
+
call_id: str
|
|
64
|
+
tool: str
|
|
65
|
+
output: str
|
|
66
|
+
ok: bool
|
|
67
|
+
duration_s: float
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass
|
|
71
|
+
class Compacted(Event):
|
|
72
|
+
"""History was rewritten to fit the context window (mid-turn)."""
|
|
73
|
+
|
|
74
|
+
strategy: str # "prune" (old tool outputs stubbed) | "summarize" (state doc rebuild)
|
|
75
|
+
tokens_before: int # projected prompt tokens that tripped the limit
|
|
76
|
+
tokens_after: int # conservative estimate of the rebuilt history
|
|
77
|
+
messages_before: int
|
|
78
|
+
messages_after: int
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass
|
|
82
|
+
class ContextReminder(Event):
|
|
83
|
+
"""Soft, non-blocking nudge: context passed the 'model degrades past here'
|
|
84
|
+
mark (DeepSeek V4 ~50%). The user decides whether to /clear; auto-compaction
|
|
85
|
+
only kicks in near the window ceiling."""
|
|
86
|
+
|
|
87
|
+
pct: float # fraction of the window currently used
|
|
88
|
+
window: int
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass
|
|
92
|
+
class TurnFinished(Event):
|
|
93
|
+
"""End of a full user turn (all tool round-trips done)."""
|
|
94
|
+
|
|
95
|
+
steps: int
|
|
96
|
+
usage: dict = field(default_factory=dict)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
@dataclass
|
|
100
|
+
class EngineError(Event):
|
|
101
|
+
message: str
|