rockycode 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rockycode/__init__.py +1 -0
- rockycode/banner.py +37 -0
- rockycode/cli.py +1386 -0
- rockycode/config.py +178 -0
- rockycode/dream/__init__.py +9 -0
- rockycode/dream/core.py +523 -0
- rockycode/dream/judge.py +134 -0
- rockycode/dream/mining.py +152 -0
- rockycode/dream/proposals.py +440 -0
- rockycode/engine/__init__.py +10 -0
- rockycode/engine/artifact.py +367 -0
- rockycode/engine/budget.py +90 -0
- rockycode/engine/checks.py +157 -0
- rockycode/engine/compaction.py +181 -0
- rockycode/engine/container.py +225 -0
- rockycode/engine/effort.py +46 -0
- rockycode/engine/events.py +101 -0
- rockycode/engine/explore.py +592 -0
- rockycode/engine/goal.py +541 -0
- rockycode/engine/goal_review.py +161 -0
- rockycode/engine/goal_session.py +259 -0
- rockycode/engine/headless.py +481 -0
- rockycode/engine/loop.py +711 -0
- rockycode/engine/lsp.py +473 -0
- rockycode/engine/mcp.py +364 -0
- rockycode/engine/modes.py +123 -0
- rockycode/engine/outcome.py +81 -0
- rockycode/engine/permission.py +198 -0
- rockycode/engine/planmode.py +249 -0
- rockycode/engine/providers.py +196 -0
- rockycode/engine/redact.py +83 -0
- rockycode/engine/safety.py +139 -0
- rockycode/engine/sandbox.py +219 -0
- rockycode/engine/server.py +431 -0
- rockycode/engine/skills.py +178 -0
- rockycode/engine/titler.py +46 -0
- rockycode/engine/tools.py +479 -0
- rockycode/engine/trajectory.py +131 -0
- rockycode/engine/web.py +431 -0
- rockycode/engine/worktree.py +128 -0
- rockycode/memory/__init__.py +7 -0
- rockycode/memory/index.py +260 -0
- rockycode/memory/store.py +331 -0
- rockycode/modes/learn/learn.md +46 -0
- rockycode/modes/research/deep-research.md +53 -0
- rockycode/modes/research/paper-reading.md +49 -0
- rockycode/modes/research/prove.md +60 -0
- rockycode/modes/research/whiteboard.md +64 -0
- rockycode/onboarding.py +332 -0
- rockycode/palette.py +15 -0
- rockycode/pricing.py +178 -0
- rockycode/prompts/__init__.py +0 -0
- rockycode/prompts/rocky.py +257 -0
- rockycode/routines.py +287 -0
- rockycode/runners/__init__.py +0 -0
- rockycode/runners/agent.py +273 -0
- rockycode/runners/data.py +61 -0
- rockycode/runners/raw.py +176 -0
- rockycode/score.py +114 -0
- rockycode/session.py +298 -0
- rockycode/skills/architecture-viz/SKILL.md +71 -0
- rockycode/skills/architecture-viz/template.html +87 -0
- rockycode/skills/lean-prover/SKILL.md +155 -0
- rockycode/skills/lean-prover/torchlean-api.md +85 -0
- rockycode/tui/__init__.py +1 -0
- rockycode/tui/app.py +2450 -0
- rockycode/tui/exitsheet.py +181 -0
- rockycode/tui/goal_screen.py +315 -0
- rockycode/tui/mdterm.py +232 -0
- rockycode/tui/mdview.py +99 -0
- rockycode/tui/modepicker.py +103 -0
- rockycode/tui/permission.py +154 -0
- rockycode/tui/plangate.py +110 -0
- rockycode/tui/prompt_history.py +77 -0
- rockycode/tui/proposalcard.py +126 -0
- rockycode/tui/resume.py +142 -0
- rockycode/tui/rocky_pet.py +96 -0
- rockycode/tui/routinecard.py +123 -0
- rockycode-0.1.0.dist-info/METADATA +488 -0
- rockycode-0.1.0.dist-info/RECORD +83 -0
- rockycode-0.1.0.dist-info/WHEEL +4 -0
- rockycode-0.1.0.dist-info/entry_points.txt +2 -0
- rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
rockycode/engine/loop.py
ADDED
|
@@ -0,0 +1,711 @@
|
|
|
1
|
+
"""The agent loop: stream DeepSeek with native tool calls, execute tools,
|
|
2
|
+
repeat until the model answers without tools. Emits Events; owns history.
|
|
3
|
+
|
|
4
|
+
Before every API call the loop projects the next prompt size (last real
|
|
5
|
+
prompt_tokens + char-based estimates for newer messages) and, over the
|
|
6
|
+
threshold, compacts history via compaction.py (prune → state summary).
|
|
7
|
+
|
|
8
|
+
DeepSeek specifics handled here (see memory: reference_deepseek_api):
|
|
9
|
+
- thinking + reasoning_effort go in extra_body
|
|
10
|
+
- the stream carries delta.reasoning_content AND delta.content
|
|
11
|
+
- reasoning_content must NOT be sent back in history (HTTP 400) — we only
|
|
12
|
+
ever append {role, content, tool_calls} for assistant turns
|
|
13
|
+
- usage extras (cache hit/miss) are read via model_dump()
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import asyncio
|
|
18
|
+
import json
|
|
19
|
+
import os
|
|
20
|
+
import time
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import AsyncIterator, Awaitable, Callable, Optional
|
|
23
|
+
|
|
24
|
+
from openai import AsyncOpenAI
|
|
25
|
+
|
|
26
|
+
from rockycode.onboarding import current_key_source, require_base_url, require_key
|
|
27
|
+
|
|
28
|
+
from rockycode.engine import compaction
|
|
29
|
+
from rockycode.engine import planmode
|
|
30
|
+
from rockycode.engine import tools as tools_mod
|
|
31
|
+
from rockycode.engine.effort import build_extra_body
|
|
32
|
+
from rockycode.engine.redact import redact
|
|
33
|
+
from rockycode.engine.events import (
|
|
34
|
+
AgentState,
|
|
35
|
+
Compacted,
|
|
36
|
+
ContextReminder,
|
|
37
|
+
EngineError,
|
|
38
|
+
Event,
|
|
39
|
+
StateChanged,
|
|
40
|
+
TextDelta,
|
|
41
|
+
ThinkingDelta,
|
|
42
|
+
ToolFinished,
|
|
43
|
+
ToolStarted,
|
|
44
|
+
TurnFinished,
|
|
45
|
+
TurnStarted,
|
|
46
|
+
)
|
|
47
|
+
from rockycode.engine.outcome import SessionStats
|
|
48
|
+
from rockycode.engine.trajectory import TrajectoryLogger
|
|
49
|
+
from rockycode.prompts.rocky import ROCKY_SYSTEM
|
|
50
|
+
|
|
51
|
+
MAX_STEPS = 50
|
|
52
|
+
FINALIZE_STEPS = 3 # forced wrap-up window after the explore budget is spent
|
|
53
|
+
CONTEXT_WINDOW = 1_048_576 # DeepSeek V4's real 1M window (matches the bench config)
|
|
54
|
+
# Two-tier context handling. DeepSeek V4 degrades past ~50%, but forcing a
|
|
55
|
+
# compaction there interrupts the user — so 50% is a SOFT one-time reminder
|
|
56
|
+
# (ContextReminder), and automatic compaction only fires near the ceiling.
|
|
57
|
+
REMIND_THRESHOLD = 0.50 # non-blocking nudge: "past the half, /clear if you want"
|
|
58
|
+
COMPACT_THRESHOLD = 0.90 # hard auto-compact — the safety net before overflow
|
|
59
|
+
|
|
60
|
+
# Soft nudge while there's still explore budget. Evidence (dev10, twice):
|
|
61
|
+
# without pressure the model explores right through the cap and submits
|
|
62
|
+
# nothing — matplotlib-23476 burned all 50 steps on investigation, zero edits.
|
|
63
|
+
BUDGET_WARNINGS = {
|
|
64
|
+
10: (
|
|
65
|
+
"[harness] only 10 steps remain. stop exploring — commit to your best "
|
|
66
|
+
"fix now, make the edit, verify with one targeted test run, then finish."
|
|
67
|
+
),
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
# The finalize loop: once explore budget is spent, the model gets FINALIZE_STEPS
|
|
71
|
+
# extra turns with this hard instruction so it ALWAYS applies its fix instead of
|
|
72
|
+
# stopping empty-handed. This is the "make sure it finishes" mechanism.
|
|
73
|
+
FINALIZE_NOTICE = (
|
|
74
|
+
"[harness] EXPLORE BUDGET SPENT — this is your final chance. If your fix is "
|
|
75
|
+
"not yet written to disk, apply it NOW with edit_file/write_file, then stop. "
|
|
76
|
+
"Do not read or search anymore. When the edit is saved, reply DONE with a one-"
|
|
77
|
+
"line summary of what you changed."
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _merge_usage(total: dict[str, int], usage: dict) -> None:
|
|
82
|
+
for k, v in usage.items():
|
|
83
|
+
if isinstance(v, int):
|
|
84
|
+
total[k] = total.get(k, 0) + v
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
async def _always_allow(name: str, args: dict) -> bool:
|
|
88
|
+
"""Default approver: every tool call runs, no prompt. Keeps bench/headless
|
|
89
|
+
and the current TUI behavior unchanged unless a real approver is injected."""
|
|
90
|
+
return True
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class Engine:
|
|
94
|
+
def __init__(
|
|
95
|
+
self,
|
|
96
|
+
model: str,
|
|
97
|
+
*,
|
|
98
|
+
thinking: bool = True,
|
|
99
|
+
reasoning_effort: str = "max",
|
|
100
|
+
max_tokens: int = 384_000, # DeepSeek V4 max output (384K); never truncate
|
|
101
|
+
workdir: Optional[Path] = None,
|
|
102
|
+
allowed_roots: tuple[Path, ...] = (),
|
|
103
|
+
system_prompt: str = ROCKY_SYSTEM,
|
|
104
|
+
client: Optional[AsyncOpenAI] = None,
|
|
105
|
+
registry: Optional[dict[str, tools_mod.Tool]] = None,
|
|
106
|
+
trajectory_meta: Optional[dict] = None,
|
|
107
|
+
context_window: int = CONTEXT_WINDOW,
|
|
108
|
+
compact_threshold: float = COMPACT_THRESHOLD,
|
|
109
|
+
remind_threshold: float = REMIND_THRESHOLD,
|
|
110
|
+
max_steps: int = MAX_STEPS,
|
|
111
|
+
finalize_steps: int = FINALIZE_STEPS,
|
|
112
|
+
approver: Optional[Callable[[str, dict], Awaitable[bool]]] = None,
|
|
113
|
+
) -> None:
|
|
114
|
+
self.model = model
|
|
115
|
+
self.thinking = thinking
|
|
116
|
+
self.reasoning_effort = reasoning_effort
|
|
117
|
+
# Which reasoning-param shape the active provider wants (deepseek |
|
|
118
|
+
# openai | none). Set by a /model switch; defaults to DeepSeek, rocky's
|
|
119
|
+
# home provider, so existing behavior is byte-identical.
|
|
120
|
+
self.reasoning_policy = "deepseek"
|
|
121
|
+
self.provider_name = "deepseek"
|
|
122
|
+
self.tools_enabled = True # profile tools:off → drop tool schemas
|
|
123
|
+
self.max_tokens = max_tokens
|
|
124
|
+
self.context_window = context_window
|
|
125
|
+
self.compact_threshold = compact_threshold
|
|
126
|
+
self.remind_threshold = remind_threshold
|
|
127
|
+
self._reminded = False # fired the soft 50% nudge? re-arms below the mark
|
|
128
|
+
self.max_steps = max_steps
|
|
129
|
+
self.finalize_steps = finalize_steps
|
|
130
|
+
# Opt-in approval gate. Default = always-allow, so bench/headless and the
|
|
131
|
+
# current TUI stay unchanged until a real approver is injected (the TUI
|
|
132
|
+
# sets one that runs permission.decide() + a modal). Used by _approve.
|
|
133
|
+
self.approver = approver or _always_allow
|
|
134
|
+
# Plan mode: set to the session's plan-file path to make this engine
|
|
135
|
+
# read-only — planmode.gate() then runs on every tool call BEFORE the
|
|
136
|
+
# approver (the plan file itself is the one writable target). None = off.
|
|
137
|
+
# Host state only; the model never toggles it (docs/plan-mode-design.md).
|
|
138
|
+
self.plan_file: Optional[Path] = None
|
|
139
|
+
# Resolved paths a live human approval widened the READ jail to (read_file
|
|
140
|
+
# only). Mutable + held by reference in the registry, so an approval takes
|
|
141
|
+
# effect immediately. Never fed by a file — only a launch flag or a click.
|
|
142
|
+
self.read_grants: set[Path] = set()
|
|
143
|
+
# Compaction bookkeeping: the API's real prompt size for everything
|
|
144
|
+
# up to history[_sent_until], estimates for anything appended after.
|
|
145
|
+
self._last_prompt_tokens = 0
|
|
146
|
+
self._sent_until = 0
|
|
147
|
+
# Heuristic outcome counters (self-evolve phase 0), incremented at the
|
|
148
|
+
# exact branch points below and flushed as ONE `outcome` record by
|
|
149
|
+
# finalize_outcome() when the session ends.
|
|
150
|
+
self.stats = SessionStats()
|
|
151
|
+
self._outcome_written = False
|
|
152
|
+
self.workdir = workdir or Path.cwd()
|
|
153
|
+
self.allowed_roots = allowed_roots
|
|
154
|
+
# Explicit key AND endpoint (not the SDK's env fallbacks): an ambient
|
|
155
|
+
# OPENAI_API_KEY/OPENAI_BASE_URL must never decide what gets sent where.
|
|
156
|
+
self.client = client or AsyncOpenAI(api_key=require_key(), base_url=require_base_url(),
|
|
157
|
+
max_retries=5, timeout=300.0)
|
|
158
|
+
self.registry = (
|
|
159
|
+
registry if registry is not None
|
|
160
|
+
else tools_mod.build_registry(self.workdir, allowed_roots, read_grants=self.read_grants)
|
|
161
|
+
)
|
|
162
|
+
self.history: list[dict] = [{"role": "system", "content": system_prompt}]
|
|
163
|
+
# Collaboration mode (host-owned, like plan_file — the model never
|
|
164
|
+
# toggles it): set_mode() swaps a contract into the system prompt.
|
|
165
|
+
self._base_system = system_prompt
|
|
166
|
+
self._mode_contract: Optional[str] = None
|
|
167
|
+
self.mode_name: Optional[str] = None
|
|
168
|
+
self.resumed_mode: Optional[str] = None # mode seen in a resumed session's prompt
|
|
169
|
+
self.trajectory = TrajectoryLogger(
|
|
170
|
+
meta={
|
|
171
|
+
"model": model,
|
|
172
|
+
"thinking": thinking,
|
|
173
|
+
"reasoning_effort": reasoning_effort,
|
|
174
|
+
"max_tokens": max_tokens,
|
|
175
|
+
"context_window": context_window,
|
|
176
|
+
"max_steps": max_steps,
|
|
177
|
+
"workdir": str(self.workdir),
|
|
178
|
+
"base_url": require_base_url(),
|
|
179
|
+
**(trajectory_meta or {}),
|
|
180
|
+
}
|
|
181
|
+
)
|
|
182
|
+
self.trajectory.message(self.history[0])
|
|
183
|
+
|
|
184
|
+
def resume(self, messages: list[dict], *, from_session: Optional[str] = None) -> list[dict]:
|
|
185
|
+
"""Seed history from a prior session's messages. Keeps the CURRENT
|
|
186
|
+
system prompt (so updated memory/skills/instructions apply) and carries
|
|
187
|
+
the rest of the conversation. The carried messages are logged into the
|
|
188
|
+
new trajectory, so it stays self-contained for training."""
|
|
189
|
+
# A prior session's mode lives in ITS system prompt, which we drop —
|
|
190
|
+
# remember the name so the UI can offer re-entry instead of silence.
|
|
191
|
+
self.resumed_mode = None
|
|
192
|
+
for m in messages:
|
|
193
|
+
if m.get("role") == "system":
|
|
194
|
+
for line in str(m.get("content", "")).splitlines():
|
|
195
|
+
if line.startswith("# Collaboration mode: "):
|
|
196
|
+
self.resumed_mode = line.removeprefix("# Collaboration mode: ").strip()
|
|
197
|
+
carried = [m for m in messages if m.get("role") != "system"]
|
|
198
|
+
for m in carried:
|
|
199
|
+
self._append(m)
|
|
200
|
+
self.trajectory.note({"resumed_from": from_session, "carried": len(carried)})
|
|
201
|
+
return carried
|
|
202
|
+
|
|
203
|
+
def swap_registry(self, registry: dict[str, tools_mod.Tool]) -> None:
|
|
204
|
+
"""Hot-swap the tool registry at runtime (e.g. sandbox on/off)."""
|
|
205
|
+
self.registry = registry
|
|
206
|
+
|
|
207
|
+
def finalize_outcome(self) -> Optional[dict]:
|
|
208
|
+
"""Write the session's heuristic outcome record (self-evolve phase 0).
|
|
209
|
+
|
|
210
|
+
Called once at session end — cli: after the app closes, before the
|
|
211
|
+
exit card. Idempotent, and skipped when no user turn ever ran, so an
|
|
212
|
+
open-and-close session doesn't grow a meaningless reward line. The
|
|
213
|
+
judge-graded outcome (source="judge") is appended much later, by dream.
|
|
214
|
+
"""
|
|
215
|
+
if self._outcome_written or self.stats.turns == 0:
|
|
216
|
+
return None
|
|
217
|
+
self._outcome_written = True
|
|
218
|
+
data = {"source": "heuristic", **self.stats.as_data()}
|
|
219
|
+
self.trajectory.outcome(data)
|
|
220
|
+
return data
|
|
221
|
+
|
|
222
|
+
def set_base_system(self, text: str) -> None:
|
|
223
|
+
"""Finalize the base system prompt after startup composition.
|
|
224
|
+
|
|
225
|
+
The chat CLI can only generate the "# Tools this session" section
|
|
226
|
+
once every tool has registered (skills/memory/web/artifact/goal/
|
|
227
|
+
explore land AFTER Engine construction), so the final prompt —
|
|
228
|
+
tools + language + environment + date — is set here, before the
|
|
229
|
+
first user turn. Re-layers an active mode contract; logs the final
|
|
230
|
+
prompt to the trajectory so training data records what was actually
|
|
231
|
+
sent (the construction-time record is superseded, last-system-wins).
|
|
232
|
+
"""
|
|
233
|
+
self._base_system = text
|
|
234
|
+
if self.mode_name is not None and self._mode_contract is not None:
|
|
235
|
+
self.history[0]["content"] = (
|
|
236
|
+
f"{text}\n\n# Collaboration mode: {self.mode_name}\n\n"
|
|
237
|
+
f"{self._mode_contract.strip()}"
|
|
238
|
+
)
|
|
239
|
+
else:
|
|
240
|
+
self.history[0]["content"] = text
|
|
241
|
+
self.trajectory.message(self.history[0])
|
|
242
|
+
self.trajectory.note({"system_finalized": len(text)})
|
|
243
|
+
|
|
244
|
+
def set_mode(self, name: str, contract: str) -> None:
|
|
245
|
+
"""Swap a collaboration contract into the system prompt (a REAL swap,
|
|
246
|
+
not an appended note — models weight the system prompt hardest). Costs
|
|
247
|
+
one prefix-cache miss at the switch, which the user accepted; switching
|
|
248
|
+
at launch (config `mode`) costs nothing. One mode at a time: setting a
|
|
249
|
+
new one replaces the old."""
|
|
250
|
+
self.history[0]["content"] = (
|
|
251
|
+
f"{self._base_system}\n\n# Collaboration mode: {name}\n\n{contract.strip()}"
|
|
252
|
+
)
|
|
253
|
+
self.mode_name = name
|
|
254
|
+
self._mode_contract = contract
|
|
255
|
+
self.trajectory.note({"mode": name})
|
|
256
|
+
|
|
257
|
+
def clear_mode(self) -> None:
|
|
258
|
+
"""Back to the plain rocky contract."""
|
|
259
|
+
if self.mode_name is None:
|
|
260
|
+
return
|
|
261
|
+
self.history[0]["content"] = self._base_system
|
|
262
|
+
self.mode_name = None
|
|
263
|
+
self._mode_contract = None
|
|
264
|
+
self.trajectory.note({"mode": None})
|
|
265
|
+
|
|
266
|
+
def _extra_body(self) -> dict:
|
|
267
|
+
# self.reasoning_effort holds the dial value (high|xhigh|max) and is
|
|
268
|
+
# mutable mid-session (/effort); the provider clamp + param shape happen
|
|
269
|
+
# per call, keyed by the active provider's reasoning policy.
|
|
270
|
+
return build_extra_body(self.thinking, self.reasoning_effort, self.reasoning_policy)
|
|
271
|
+
|
|
272
|
+
def switch_provider(self, client, model: str, *, provider_name: str,
|
|
273
|
+
reasoning_policy: str, tools_enabled: bool = True) -> None:
|
|
274
|
+
"""Point the engine at a different OpenAI-compatible endpoint/model live
|
|
275
|
+
(a /model switch). The caller builds the client with the provider's
|
|
276
|
+
base_url + key; we swap model + reasoning policy. The prompt-cache prefix
|
|
277
|
+
is provider-specific, so the switch naturally starts a fresh cache — no
|
|
278
|
+
stale-hit risk. History and tools are untouched (same OpenAI protocol)."""
|
|
279
|
+
self.client = client
|
|
280
|
+
self.model = model
|
|
281
|
+
self.provider_name = provider_name
|
|
282
|
+
self.reasoning_policy = reasoning_policy
|
|
283
|
+
self.tools_enabled = tools_enabled
|
|
284
|
+
self.trajectory.note({"provider": provider_name, "model": model})
|
|
285
|
+
|
|
286
|
+
def _repair_history(self) -> None:
|
|
287
|
+
"""Inject synthetic tool responses for ANY orphaned tool_calls.
|
|
288
|
+
|
|
289
|
+
A tool_calls assistant message with no matching tool response for one of
|
|
290
|
+
its ids makes DeepSeek 400. This happens after a hard kill mid-tool, or
|
|
291
|
+
when a turn is cancelled (new submit / Esc). This is:
|
|
292
|
+
- idempotent: a call that already has a response is left alone, so it
|
|
293
|
+
is safe to call repeatedly (the turn's finally + the next turn's
|
|
294
|
+
pre-flight both call it) without ever producing a DUPLICATE response;
|
|
295
|
+
- in-position: a stub goes right after its own assistant message
|
|
296
|
+
(past any responses already there), never at the end — so it can't
|
|
297
|
+
land after a newly-appended user message and break ordering.
|
|
298
|
+
Fixes every orphan, not just the most recent (a multi-orphan resume
|
|
299
|
+
otherwise still 400s).
|
|
300
|
+
"""
|
|
301
|
+
i = 0
|
|
302
|
+
while i < len(self.history):
|
|
303
|
+
msg = self.history[i]
|
|
304
|
+
if msg.get("role") == "assistant" and msg.get("tool_calls"):
|
|
305
|
+
answered = {
|
|
306
|
+
self.history[j].get("tool_call_id")
|
|
307
|
+
for j in range(i + 1, len(self.history))
|
|
308
|
+
if self.history[j].get("role") == "tool"
|
|
309
|
+
}
|
|
310
|
+
insert_at = i + 1
|
|
311
|
+
while insert_at < len(self.history) and self.history[insert_at].get("role") == "tool":
|
|
312
|
+
insert_at += 1
|
|
313
|
+
for tc in msg["tool_calls"]:
|
|
314
|
+
cid = tc.get("id", "")
|
|
315
|
+
if cid and cid not in answered:
|
|
316
|
+
self.history.insert(insert_at, {
|
|
317
|
+
"role": "tool",
|
|
318
|
+
"tool_call_id": cid,
|
|
319
|
+
"content": "[error] tool execution was interrupted",
|
|
320
|
+
})
|
|
321
|
+
insert_at += 1
|
|
322
|
+
i += 1
|
|
323
|
+
|
|
324
|
+
def _append(self, msg: dict) -> None:
|
|
325
|
+
self.history.append(msg)
|
|
326
|
+
self.trajectory.message(msg)
|
|
327
|
+
|
|
328
|
+
def _is_read(self, name: str) -> bool:
|
|
329
|
+
"""A read-like tool — 'safe' risk tier, i.e. non-mutating (read_file /
|
|
330
|
+
grep / glob / check_code / recall_memory). Read calls can run concurrently
|
|
331
|
+
with one another; anything that writes or has side effects (edit / write /
|
|
332
|
+
bash / web / mcp) stays serial so two writes can never clobber."""
|
|
333
|
+
return getattr(self.registry.get(name), "risk", "risky") == "safe"
|
|
334
|
+
|
|
335
|
+
async def _approve(self, name: str, arguments_json: str) -> bool:
|
|
336
|
+
"""Ask the injected approver whether this tool call may run. Parses args
|
|
337
|
+
to a dict for the approver; a malformed payload becomes {} (execute will
|
|
338
|
+
surface the real error). Short-circuits the always-allow default so
|
|
339
|
+
headless/bench runs pay nothing."""
|
|
340
|
+
if self.approver is _always_allow:
|
|
341
|
+
return True
|
|
342
|
+
try:
|
|
343
|
+
args = json.loads(arguments_json) if arguments_json.strip() else {}
|
|
344
|
+
except (json.JSONDecodeError, AttributeError):
|
|
345
|
+
args = {}
|
|
346
|
+
if not isinstance(args, dict):
|
|
347
|
+
args = {}
|
|
348
|
+
return bool(await self.approver(name, args))
|
|
349
|
+
|
|
350
|
+
def _plan_gate(self, name: str, arguments_json: str) -> planmode.Verdict:
|
|
351
|
+
"""Plan-mode read-only gate, evaluated BEFORE the approver. Always
|
|
352
|
+
'pass' when plan mode is off, so normal sessions pay nothing."""
|
|
353
|
+
if self.plan_file is None:
|
|
354
|
+
return planmode.PASS
|
|
355
|
+
try:
|
|
356
|
+
args = json.loads(arguments_json) if arguments_json.strip() else {}
|
|
357
|
+
except (json.JSONDecodeError, AttributeError):
|
|
358
|
+
args = {}
|
|
359
|
+
if not isinstance(args, dict):
|
|
360
|
+
args = {}
|
|
361
|
+
risk = getattr(self.registry.get(name), "risk", "risky")
|
|
362
|
+
return planmode.gate(name, args, risk, self.plan_file, self.workdir)
|
|
363
|
+
|
|
364
|
+
def _projected_prompt_tokens(self) -> int:
|
|
365
|
+
"""Best guess at the next call's prompt size: the API's real count
|
|
366
|
+
for everything already sent, plus a conservative estimate for what
|
|
367
|
+
was appended since."""
|
|
368
|
+
return self._last_prompt_tokens + compaction.estimate_tokens(
|
|
369
|
+
self.history[self._sent_until :]
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
async def _maybe_compact(self, usage_total: dict[str, int]) -> AsyncIterator[Event]:
|
|
373
|
+
"""Shrink history if the next call would crowd the context window.
|
|
374
|
+
|
|
375
|
+
Free deterministic prune first; an LLM state summary only when
|
|
376
|
+
pruning can't get under the limit. Never raises — a failed summary
|
|
377
|
+
degrades to the prune and the turn continues.
|
|
378
|
+
"""
|
|
379
|
+
projected = self._projected_prompt_tokens()
|
|
380
|
+
# Soft, non-blocking nudge once we cross the degrade mark (~50%); re-arms
|
|
381
|
+
# after we drop back under it (a compaction, or a /clear).
|
|
382
|
+
if projected <= int(self.context_window * self.remind_threshold):
|
|
383
|
+
self._reminded = False
|
|
384
|
+
elif not self._reminded:
|
|
385
|
+
self._reminded = True
|
|
386
|
+
yield ContextReminder(pct=projected / self.context_window, window=self.context_window)
|
|
387
|
+
|
|
388
|
+
limit = int(self.context_window * self.compact_threshold)
|
|
389
|
+
if projected <= limit or len(self.history) < 3:
|
|
390
|
+
return
|
|
391
|
+
yield StateChanged(state=AgentState.COMPACTING)
|
|
392
|
+
|
|
393
|
+
before_msgs = len(self.history)
|
|
394
|
+
record: dict = {}
|
|
395
|
+
# Cap any single oversized message FIRST — a huge paste or tool result the
|
|
396
|
+
# tail keeps verbatim would otherwise survive prune (tool-only) and
|
|
397
|
+
# summarize (keeps the tail), keeping every step over the limit, or would
|
|
398
|
+
# overflow the summarize call itself. Head+tail truncation bounds it.
|
|
399
|
+
truncated = compaction.truncate_oversized(self.history)
|
|
400
|
+
if truncated:
|
|
401
|
+
record["truncated_oversized"] = truncated
|
|
402
|
+
projected = compaction.estimate_tokens(self.history)
|
|
403
|
+
tail = compaction.tail_start(
|
|
404
|
+
self.history,
|
|
405
|
+
min(compaction.KEEP_RECENT_TOKENS, compaction.estimate_tokens(self.history) // 2),
|
|
406
|
+
)
|
|
407
|
+
record["tail_start"] = tail
|
|
408
|
+
|
|
409
|
+
if projected - compaction.prune_savings(self.history, tail) <= limit:
|
|
410
|
+
strategy = "prune"
|
|
411
|
+
record["pruned_tool_outputs"] = compaction.prune_tool_outputs(self.history, tail)
|
|
412
|
+
else:
|
|
413
|
+
strategy = "summarize"
|
|
414
|
+
try:
|
|
415
|
+
summary, usage = await compaction.summarize(
|
|
416
|
+
self.client, self.model, self.history,
|
|
417
|
+
tools=[t.schema for t in self.registry.values()],
|
|
418
|
+
)
|
|
419
|
+
if not summary:
|
|
420
|
+
raise ValueError("model returned an empty summary")
|
|
421
|
+
except Exception as e: # noqa: BLE001 — degrade to prune, never kill the turn
|
|
422
|
+
strategy = "prune"
|
|
423
|
+
record["pruned_tool_outputs"] = compaction.prune_tool_outputs(self.history, tail)
|
|
424
|
+
yield EngineError(
|
|
425
|
+
message=f"compaction summary failed ({type(e).__name__}: {e}) — "
|
|
426
|
+
"pruned old tool outputs instead"
|
|
427
|
+
)
|
|
428
|
+
else:
|
|
429
|
+
_merge_usage(usage_total, usage)
|
|
430
|
+
_merge_usage(self.stats.usage, usage)
|
|
431
|
+
self.history = [self.history[0], compaction.state_message(summary)] + self.history[tail:]
|
|
432
|
+
record["summary"] = summary
|
|
433
|
+
record["new_history"] = list(self.history)
|
|
434
|
+
|
|
435
|
+
self._last_prompt_tokens = 0
|
|
436
|
+
self._sent_until = 0
|
|
437
|
+
self.stats.compactions += 1
|
|
438
|
+
tokens_after = compaction.estimate_tokens(self.history)
|
|
439
|
+
self.trajectory.compaction(
|
|
440
|
+
{
|
|
441
|
+
"strategy": strategy,
|
|
442
|
+
"tokens_before": projected,
|
|
443
|
+
"tokens_after_est": tokens_after,
|
|
444
|
+
"messages_before": before_msgs,
|
|
445
|
+
"messages_after": len(self.history),
|
|
446
|
+
**record,
|
|
447
|
+
}
|
|
448
|
+
)
|
|
449
|
+
yield Compacted(
|
|
450
|
+
strategy=strategy,
|
|
451
|
+
tokens_before=projected,
|
|
452
|
+
tokens_after=tokens_after,
|
|
453
|
+
messages_before=before_msgs,
|
|
454
|
+
messages_after=len(self.history),
|
|
455
|
+
)
|
|
456
|
+
|
|
457
|
+
async def run_turn(self, user_message: str) -> AsyncIterator[Event]:
|
|
458
|
+
"""One full user turn: stream → maybe tools → stream → … → answer."""
|
|
459
|
+
yield TurnStarted(user_message=user_message) # UI sees the clean message
|
|
460
|
+
self.stats.turns += 1
|
|
461
|
+
# Plan mode rides the USER turn (never the system prompt/tools → the cached
|
|
462
|
+
# prompt prefix stays byte-identical when the mode toggles).
|
|
463
|
+
if self.plan_file is not None:
|
|
464
|
+
user_message = planmode.marker(self.plan_file) + "\n\n" + user_message
|
|
465
|
+
self._append({"role": "user", "content": user_message})
|
|
466
|
+
|
|
467
|
+
usage_total: dict[str, int] = {}
|
|
468
|
+
used_tools = False
|
|
469
|
+
steps = 0
|
|
470
|
+
|
|
471
|
+
while True:
|
|
472
|
+
steps += 1
|
|
473
|
+
self.stats.steps += 1
|
|
474
|
+
|
|
475
|
+
# max_steps <= 0 means UNLIMITED — no cap, no finalize, no budget
|
|
476
|
+
# warnings. Used by interactive chat so the model runs until it's
|
|
477
|
+
# done; the human (and Esc/new-message interrupt) is the backstop.
|
|
478
|
+
if self.max_steps > 0:
|
|
479
|
+
over = steps - self.max_steps # >0 ⇒ in the forced finalize window
|
|
480
|
+
if over > self.finalize_steps:
|
|
481
|
+
self.stats.engine_errors += 1
|
|
482
|
+
yield EngineError(
|
|
483
|
+
message=f"step limit ({self.max_steps}+{self.finalize_steps} finalize) "
|
|
484
|
+
"reached — stopping this turn."
|
|
485
|
+
)
|
|
486
|
+
break
|
|
487
|
+
if over >= 1 and not used_tools:
|
|
488
|
+
# never acted — no fix to finalize, just stop
|
|
489
|
+
self.stats.engine_errors += 1
|
|
490
|
+
yield EngineError(message=f"step limit ({self.max_steps}) reached — stopping this turn.")
|
|
491
|
+
break
|
|
492
|
+
if over == 1:
|
|
493
|
+
# entering finalize: one hard instruction to apply the fix now
|
|
494
|
+
self._append({"role": "user", "content": FINALIZE_NOTICE})
|
|
495
|
+
elif over <= 0:
|
|
496
|
+
# soft pressure while explore budget remains
|
|
497
|
+
warning = BUDGET_WARNINGS.get(self.max_steps - steps)
|
|
498
|
+
if warning is not None and used_tools:
|
|
499
|
+
self._append({"role": "user", "content": warning})
|
|
500
|
+
|
|
501
|
+
async for ev in self._maybe_compact(usage_total):
|
|
502
|
+
yield ev
|
|
503
|
+
|
|
504
|
+
self._repair_history()
|
|
505
|
+
sent = len(self.history)
|
|
506
|
+
# Clamp output to what the window can still hold: input can run up to
|
|
507
|
+
# 90% before auto-compaction and max_tokens may be the full 384K, so
|
|
508
|
+
# input + output could exceed the window. Reserve the rest for output,
|
|
509
|
+
# never below a usable floor.
|
|
510
|
+
out_room = self.context_window - self._projected_prompt_tokens() - 512
|
|
511
|
+
eff_max_tokens = max(1024, min(self.max_tokens, out_room))
|
|
512
|
+
try:
|
|
513
|
+
# tools omitted entirely when the provider profile is tools:off
|
|
514
|
+
# (a model that calls tools badly) — the loop then runs as a
|
|
515
|
+
# plain responder rather than shipping malformed tool calls.
|
|
516
|
+
_tools = [t.schema for t in self.registry.values()] if self.tools_enabled else None
|
|
517
|
+
stream = await self.client.chat.completions.create(
|
|
518
|
+
model=self.model,
|
|
519
|
+
messages=self.history,
|
|
520
|
+
tools=_tools,
|
|
521
|
+
max_tokens=eff_max_tokens,
|
|
522
|
+
stream=True,
|
|
523
|
+
stream_options={"include_usage": True},
|
|
524
|
+
extra_body=self._extra_body(),
|
|
525
|
+
)
|
|
526
|
+
except Exception as e: # noqa: BLE001 — surfaced to UI, turn ends
|
|
527
|
+
self.stats.engine_errors += 1
|
|
528
|
+
yield StateChanged(state=AgentState.ERROR)
|
|
529
|
+
# Auth-class rejections: say WHICH credential source was used
|
|
530
|
+
# and WHERE it was sent — "which key is rocky even using?"
|
|
531
|
+
# must never need an archaeology session (it once cost an
|
|
532
|
+
# evening). Never the key itself, only its source name.
|
|
533
|
+
hint = ""
|
|
534
|
+
if getattr(e, "status_code", None) in (400, 401, 402, 403):
|
|
535
|
+
hint = (f" · key source: {current_key_source() or 'none'}"
|
|
536
|
+
f" → {require_base_url()}")
|
|
537
|
+
yield EngineError(message=redact(f"{type(e).__name__}: {e}") + hint)
|
|
538
|
+
yield StateChanged(state=AgentState.IDLE)
|
|
539
|
+
return
|
|
540
|
+
|
|
541
|
+
content_parts: list[str] = []
|
|
542
|
+
reasoning_parts: list[str] = []
|
|
543
|
+
tool_calls: dict[int, dict] = {}
|
|
544
|
+
thinking_seen = False
|
|
545
|
+
content_seen = False
|
|
546
|
+
|
|
547
|
+
async for chunk in stream:
|
|
548
|
+
if chunk.usage is not None:
|
|
549
|
+
try:
|
|
550
|
+
u = chunk.usage.model_dump()
|
|
551
|
+
except AttributeError:
|
|
552
|
+
u = dict(chunk.usage)
|
|
553
|
+
_merge_usage(usage_total, u)
|
|
554
|
+
_merge_usage(self.stats.usage, u)
|
|
555
|
+
self.trajectory.usage(u) # per-call: prompt/completion + cache hit/miss
|
|
556
|
+
if isinstance(u.get("prompt_tokens"), int):
|
|
557
|
+
self._last_prompt_tokens = u["prompt_tokens"]
|
|
558
|
+
if not chunk.choices:
|
|
559
|
+
continue
|
|
560
|
+
delta = chunk.choices[0].delta
|
|
561
|
+
|
|
562
|
+
reasoning = getattr(delta, "reasoning_content", None)
|
|
563
|
+
if reasoning:
|
|
564
|
+
if not thinking_seen:
|
|
565
|
+
thinking_seen = True
|
|
566
|
+
yield StateChanged(state=AgentState.THINKING)
|
|
567
|
+
reasoning_parts.append(reasoning)
|
|
568
|
+
yield ThinkingDelta(text=reasoning)
|
|
569
|
+
|
|
570
|
+
if delta.content:
|
|
571
|
+
if not content_seen:
|
|
572
|
+
content_seen = True
|
|
573
|
+
yield StateChanged(state=AgentState.RESPONDING)
|
|
574
|
+
content_parts.append(delta.content)
|
|
575
|
+
yield TextDelta(text=delta.content)
|
|
576
|
+
|
|
577
|
+
for tc in delta.tool_calls or []:
|
|
578
|
+
acc = tool_calls.setdefault(
|
|
579
|
+
tc.index, {"id": "", "name": "", "arguments": ""}
|
|
580
|
+
)
|
|
581
|
+
if tc.id:
|
|
582
|
+
acc["id"] = tc.id
|
|
583
|
+
if tc.function:
|
|
584
|
+
if tc.function.name:
|
|
585
|
+
acc["name"] += tc.function.name
|
|
586
|
+
if tc.function.arguments:
|
|
587
|
+
acc["arguments"] += tc.function.arguments
|
|
588
|
+
|
|
589
|
+
self._sent_until = sent
|
|
590
|
+
content = "".join(content_parts)
|
|
591
|
+
# Trajectory-only (stays OUT of history — see module doc): the
|
|
592
|
+
# thinking trace, for language-adherence checks and RL export.
|
|
593
|
+
self.trajectory.reasoning("".join(reasoning_parts))
|
|
594
|
+
|
|
595
|
+
if not tool_calls:
|
|
596
|
+
# Final answer — reasoning_content deliberately not stored.
|
|
597
|
+
self._append({"role": "assistant", "content": content})
|
|
598
|
+
if used_tools:
|
|
599
|
+
yield StateChanged(state=AgentState.AMAZED)
|
|
600
|
+
yield TurnFinished(steps=steps, usage=usage_total)
|
|
601
|
+
yield StateChanged(state=AgentState.IDLE)
|
|
602
|
+
return
|
|
603
|
+
|
|
604
|
+
used_tools = True
|
|
605
|
+
calls = [tool_calls[i] for i in sorted(tool_calls)]
|
|
606
|
+
self._append(
|
|
607
|
+
{
|
|
608
|
+
"role": "assistant",
|
|
609
|
+
"content": content or None,
|
|
610
|
+
"tool_calls": [
|
|
611
|
+
{
|
|
612
|
+
"id": c["id"],
|
|
613
|
+
"type": "function",
|
|
614
|
+
"function": {"name": c["name"], "arguments": c["arguments"]},
|
|
615
|
+
}
|
|
616
|
+
for c in calls
|
|
617
|
+
],
|
|
618
|
+
}
|
|
619
|
+
)
|
|
620
|
+
|
|
621
|
+
answered: set[str] = set()
|
|
622
|
+
# A read-only batch (every call is a non-mutating 'safe'-tier tool) runs
|
|
623
|
+
# CONCURRENTLY — reads can't conflict, and reading/grepping several files
|
|
624
|
+
# at once is the common explore step. Any batch containing a write / side
|
|
625
|
+
# effect stays SERIAL so two writes never clobber. Either way, responses
|
|
626
|
+
# are appended in the original tool_calls ORDER, which the API requires.
|
|
627
|
+
parallel = len(calls) > 1 and all(self._is_read(c["name"]) for c in calls)
|
|
628
|
+
try:
|
|
629
|
+
yield StateChanged(state=AgentState.TOOL)
|
|
630
|
+
if parallel:
|
|
631
|
+
for c in calls:
|
|
632
|
+
yield ToolStarted(call_id=c["id"], tool=c["name"], args={"raw": c["arguments"]})
|
|
633
|
+
# Approvals stay sequential (a modal can't run N-at-once); an
|
|
634
|
+
# in-workdir read auto-allows instantly, so this is usually a no-op.
|
|
635
|
+
allowed = {c["id"]: await self._approve(c["name"], c["arguments"]) for c in calls}
|
|
636
|
+
runnable = [c for c in calls if allowed[c["id"]]]
|
|
637
|
+
t0 = time.monotonic()
|
|
638
|
+
results = await asyncio.gather(*(
|
|
639
|
+
tools_mod.execute(self.registry, c["name"], c["arguments"]) for c in runnable
|
|
640
|
+
))
|
|
641
|
+
dt = time.monotonic() - t0
|
|
642
|
+
out_by_id = {c["id"]: r for c, r in zip(runnable, results)}
|
|
643
|
+
for c in calls: # original order — required by the API
|
|
644
|
+
if c["id"] in out_by_id:
|
|
645
|
+
output, ok = out_by_id[c["id"]]
|
|
646
|
+
self.stats.observe_tool(c["name"], c["arguments"], output, ok)
|
|
647
|
+
else:
|
|
648
|
+
output, ok = "[denied] user rejected this tool call", False
|
|
649
|
+
self.stats.denials += 1
|
|
650
|
+
yield ToolFinished(call_id=c["id"], tool=c["name"], output=output, ok=ok, duration_s=dt)
|
|
651
|
+
self._append({"role": "tool", "tool_call_id": c["id"], "content": output})
|
|
652
|
+
answered.add(c["id"])
|
|
653
|
+
else:
|
|
654
|
+
for c in calls:
|
|
655
|
+
yield ToolStarted(call_id=c["id"], tool=c["name"], args={"raw": c["arguments"]})
|
|
656
|
+
t0 = time.monotonic()
|
|
657
|
+
# Plan mode (if on) gates BEFORE the approver: a write to
|
|
658
|
+
# the plan file runs, other mutations are denied with a
|
|
659
|
+
# teaching message, read-only calls fall through to ask.
|
|
660
|
+
verdict = self._plan_gate(c["name"], c["arguments"])
|
|
661
|
+
if verdict.action == "deny":
|
|
662
|
+
output, ok = verdict.message, False
|
|
663
|
+
self.stats.plan_denials += 1
|
|
664
|
+
elif verdict.action == "allow" or await self._approve(c["name"], c["arguments"]):
|
|
665
|
+
output, ok = await tools_mod.execute(self.registry, c["name"], c["arguments"])
|
|
666
|
+
self.stats.observe_tool(c["name"], c["arguments"], output, ok)
|
|
667
|
+
else:
|
|
668
|
+
output, ok = "[denied] user rejected this tool call", False
|
|
669
|
+
self.stats.denials += 1
|
|
670
|
+
yield ToolFinished(
|
|
671
|
+
call_id=c["id"],
|
|
672
|
+
tool=c["name"],
|
|
673
|
+
output=output,
|
|
674
|
+
ok=ok,
|
|
675
|
+
duration_s=time.monotonic() - t0,
|
|
676
|
+
)
|
|
677
|
+
self._append({"role": "tool", "tool_call_id": c["id"], "content": output})
|
|
678
|
+
answered.add(c["id"])
|
|
679
|
+
finally:
|
|
680
|
+
# Invariant: every tool_call_id above MUST get a matching tool response
|
|
681
|
+
# or the next request (and --resume) 400s. On interrupt (a new submit or
|
|
682
|
+
# Esc cancels the worker → CancelledError at an await) some calls never
|
|
683
|
+
# ran. Two parts:
|
|
684
|
+
# 1) in-memory history — _repair_history backfills IN-POSITION and
|
|
685
|
+
# IDEMPOTENTLY. Critical: a concurrent new turn may already have
|
|
686
|
+
# appended its user message + run its own repair, so the OLD
|
|
687
|
+
# end-append produced a duplicate + misordered tool message that
|
|
688
|
+
# bricked the session with a permanent 400. Reusing repair can't
|
|
689
|
+
# double-insert, and it's sync (no await) so it finishes before
|
|
690
|
+
# control returns to the new turn.
|
|
691
|
+
# 2) the append-only trajectory — log a stub for each un-run call so a
|
|
692
|
+
# --resume reload also has the response. Best-effort; a failing
|
|
693
|
+
# trajectory write must never mask the cancellation.
|
|
694
|
+
self._repair_history()
|
|
695
|
+
if any(c["id"] not in answered for c in calls):
|
|
696
|
+
# Esc / a new submit landed mid-batch — a strong "the user
|
|
697
|
+
# wanted something else" signal for the outcome record.
|
|
698
|
+
self.stats.interrupts += 1
|
|
699
|
+
for c in calls:
|
|
700
|
+
if c["id"] not in answered:
|
|
701
|
+
try:
|
|
702
|
+
self.trajectory.message({
|
|
703
|
+
"role": "tool",
|
|
704
|
+
"tool_call_id": c["id"],
|
|
705
|
+
"content": "[error] tool execution was interrupted",
|
|
706
|
+
})
|
|
707
|
+
except Exception: # noqa: BLE001 — never mask the cancel with I/O
|
|
708
|
+
pass
|
|
709
|
+
|
|
710
|
+
yield TurnFinished(steps=steps, usage=usage_total)
|
|
711
|
+
yield StateChanged(state=AgentState.IDLE)
|