rockycode 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. rockycode/__init__.py +1 -0
  2. rockycode/banner.py +37 -0
  3. rockycode/cli.py +1386 -0
  4. rockycode/config.py +178 -0
  5. rockycode/dream/__init__.py +9 -0
  6. rockycode/dream/core.py +523 -0
  7. rockycode/dream/judge.py +134 -0
  8. rockycode/dream/mining.py +152 -0
  9. rockycode/dream/proposals.py +440 -0
  10. rockycode/engine/__init__.py +10 -0
  11. rockycode/engine/artifact.py +367 -0
  12. rockycode/engine/budget.py +90 -0
  13. rockycode/engine/checks.py +157 -0
  14. rockycode/engine/compaction.py +181 -0
  15. rockycode/engine/container.py +225 -0
  16. rockycode/engine/effort.py +46 -0
  17. rockycode/engine/events.py +101 -0
  18. rockycode/engine/explore.py +592 -0
  19. rockycode/engine/goal.py +541 -0
  20. rockycode/engine/goal_review.py +161 -0
  21. rockycode/engine/goal_session.py +259 -0
  22. rockycode/engine/headless.py +481 -0
  23. rockycode/engine/loop.py +711 -0
  24. rockycode/engine/lsp.py +473 -0
  25. rockycode/engine/mcp.py +364 -0
  26. rockycode/engine/modes.py +123 -0
  27. rockycode/engine/outcome.py +81 -0
  28. rockycode/engine/permission.py +198 -0
  29. rockycode/engine/planmode.py +249 -0
  30. rockycode/engine/providers.py +196 -0
  31. rockycode/engine/redact.py +83 -0
  32. rockycode/engine/safety.py +139 -0
  33. rockycode/engine/sandbox.py +219 -0
  34. rockycode/engine/server.py +431 -0
  35. rockycode/engine/skills.py +178 -0
  36. rockycode/engine/titler.py +46 -0
  37. rockycode/engine/tools.py +479 -0
  38. rockycode/engine/trajectory.py +131 -0
  39. rockycode/engine/web.py +431 -0
  40. rockycode/engine/worktree.py +128 -0
  41. rockycode/memory/__init__.py +7 -0
  42. rockycode/memory/index.py +260 -0
  43. rockycode/memory/store.py +331 -0
  44. rockycode/modes/learn/learn.md +46 -0
  45. rockycode/modes/research/deep-research.md +53 -0
  46. rockycode/modes/research/paper-reading.md +49 -0
  47. rockycode/modes/research/prove.md +60 -0
  48. rockycode/modes/research/whiteboard.md +64 -0
  49. rockycode/onboarding.py +332 -0
  50. rockycode/palette.py +15 -0
  51. rockycode/pricing.py +178 -0
  52. rockycode/prompts/__init__.py +0 -0
  53. rockycode/prompts/rocky.py +257 -0
  54. rockycode/routines.py +287 -0
  55. rockycode/runners/__init__.py +0 -0
  56. rockycode/runners/agent.py +273 -0
  57. rockycode/runners/data.py +61 -0
  58. rockycode/runners/raw.py +176 -0
  59. rockycode/score.py +114 -0
  60. rockycode/session.py +298 -0
  61. rockycode/skills/architecture-viz/SKILL.md +71 -0
  62. rockycode/skills/architecture-viz/template.html +87 -0
  63. rockycode/skills/lean-prover/SKILL.md +155 -0
  64. rockycode/skills/lean-prover/torchlean-api.md +85 -0
  65. rockycode/tui/__init__.py +1 -0
  66. rockycode/tui/app.py +2450 -0
  67. rockycode/tui/exitsheet.py +181 -0
  68. rockycode/tui/goal_screen.py +315 -0
  69. rockycode/tui/mdterm.py +232 -0
  70. rockycode/tui/mdview.py +99 -0
  71. rockycode/tui/modepicker.py +103 -0
  72. rockycode/tui/permission.py +154 -0
  73. rockycode/tui/plangate.py +110 -0
  74. rockycode/tui/prompt_history.py +77 -0
  75. rockycode/tui/proposalcard.py +126 -0
  76. rockycode/tui/resume.py +142 -0
  77. rockycode/tui/rocky_pet.py +96 -0
  78. rockycode/tui/routinecard.py +123 -0
  79. rockycode-0.1.0.dist-info/METADATA +488 -0
  80. rockycode-0.1.0.dist-info/RECORD +83 -0
  81. rockycode-0.1.0.dist-info/WHEEL +4 -0
  82. rockycode-0.1.0.dist-info/entry_points.txt +2 -0
  83. rockycode-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,711 @@
1
+ """The agent loop: stream DeepSeek with native tool calls, execute tools,
2
+ repeat until the model answers without tools. Emits Events; owns history.
3
+
4
+ Before every API call the loop projects the next prompt size (last real
5
+ prompt_tokens + char-based estimates for newer messages) and, over the
6
+ threshold, compacts history via compaction.py (prune → state summary).
7
+
8
+ DeepSeek specifics handled here (see memory: reference_deepseek_api):
9
+ - thinking + reasoning_effort go in extra_body
10
+ - the stream carries delta.reasoning_content AND delta.content
11
+ - reasoning_content must NOT be sent back in history (HTTP 400) — we only
12
+ ever append {role, content, tool_calls} for assistant turns
13
+ - usage extras (cache hit/miss) are read via model_dump()
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import asyncio
18
+ import json
19
+ import os
20
+ import time
21
+ from pathlib import Path
22
+ from typing import AsyncIterator, Awaitable, Callable, Optional
23
+
24
+ from openai import AsyncOpenAI
25
+
26
+ from rockycode.onboarding import current_key_source, require_base_url, require_key
27
+
28
+ from rockycode.engine import compaction
29
+ from rockycode.engine import planmode
30
+ from rockycode.engine import tools as tools_mod
31
+ from rockycode.engine.effort import build_extra_body
32
+ from rockycode.engine.redact import redact
33
+ from rockycode.engine.events import (
34
+ AgentState,
35
+ Compacted,
36
+ ContextReminder,
37
+ EngineError,
38
+ Event,
39
+ StateChanged,
40
+ TextDelta,
41
+ ThinkingDelta,
42
+ ToolFinished,
43
+ ToolStarted,
44
+ TurnFinished,
45
+ TurnStarted,
46
+ )
47
+ from rockycode.engine.outcome import SessionStats
48
+ from rockycode.engine.trajectory import TrajectoryLogger
49
+ from rockycode.prompts.rocky import ROCKY_SYSTEM
50
+
51
+ MAX_STEPS = 50
52
+ FINALIZE_STEPS = 3 # forced wrap-up window after the explore budget is spent
53
+ CONTEXT_WINDOW = 1_048_576 # DeepSeek V4's real 1M window (matches the bench config)
54
+ # Two-tier context handling. DeepSeek V4 degrades past ~50%, but forcing a
55
+ # compaction there interrupts the user — so 50% is a SOFT one-time reminder
56
+ # (ContextReminder), and automatic compaction only fires near the ceiling.
57
+ REMIND_THRESHOLD = 0.50 # non-blocking nudge: "past the half, /clear if you want"
58
+ COMPACT_THRESHOLD = 0.90 # hard auto-compact — the safety net before overflow
59
+
60
+ # Soft nudge while there's still explore budget. Evidence (dev10, twice):
61
+ # without pressure the model explores right through the cap and submits
62
+ # nothing — matplotlib-23476 burned all 50 steps on investigation, zero edits.
63
+ BUDGET_WARNINGS = {
64
+ 10: (
65
+ "[harness] only 10 steps remain. stop exploring — commit to your best "
66
+ "fix now, make the edit, verify with one targeted test run, then finish."
67
+ ),
68
+ }
69
+
70
+ # The finalize loop: once explore budget is spent, the model gets FINALIZE_STEPS
71
+ # extra turns with this hard instruction so it ALWAYS applies its fix instead of
72
+ # stopping empty-handed. This is the "make sure it finishes" mechanism.
73
+ FINALIZE_NOTICE = (
74
+ "[harness] EXPLORE BUDGET SPENT — this is your final chance. If your fix is "
75
+ "not yet written to disk, apply it NOW with edit_file/write_file, then stop. "
76
+ "Do not read or search anymore. When the edit is saved, reply DONE with a one-"
77
+ "line summary of what you changed."
78
+ )
79
+
80
+
81
+ def _merge_usage(total: dict[str, int], usage: dict) -> None:
82
+ for k, v in usage.items():
83
+ if isinstance(v, int):
84
+ total[k] = total.get(k, 0) + v
85
+
86
+
87
+ async def _always_allow(name: str, args: dict) -> bool:
88
+ """Default approver: every tool call runs, no prompt. Keeps bench/headless
89
+ and the current TUI behavior unchanged unless a real approver is injected."""
90
+ return True
91
+
92
+
93
+ class Engine:
94
+ def __init__(
95
+ self,
96
+ model: str,
97
+ *,
98
+ thinking: bool = True,
99
+ reasoning_effort: str = "max",
100
+ max_tokens: int = 384_000, # DeepSeek V4 max output (384K); never truncate
101
+ workdir: Optional[Path] = None,
102
+ allowed_roots: tuple[Path, ...] = (),
103
+ system_prompt: str = ROCKY_SYSTEM,
104
+ client: Optional[AsyncOpenAI] = None,
105
+ registry: Optional[dict[str, tools_mod.Tool]] = None,
106
+ trajectory_meta: Optional[dict] = None,
107
+ context_window: int = CONTEXT_WINDOW,
108
+ compact_threshold: float = COMPACT_THRESHOLD,
109
+ remind_threshold: float = REMIND_THRESHOLD,
110
+ max_steps: int = MAX_STEPS,
111
+ finalize_steps: int = FINALIZE_STEPS,
112
+ approver: Optional[Callable[[str, dict], Awaitable[bool]]] = None,
113
+ ) -> None:
114
+ self.model = model
115
+ self.thinking = thinking
116
+ self.reasoning_effort = reasoning_effort
117
+ # Which reasoning-param shape the active provider wants (deepseek |
118
+ # openai | none). Set by a /model switch; defaults to DeepSeek, rocky's
119
+ # home provider, so existing behavior is byte-identical.
120
+ self.reasoning_policy = "deepseek"
121
+ self.provider_name = "deepseek"
122
+ self.tools_enabled = True # profile tools:off → drop tool schemas
123
+ self.max_tokens = max_tokens
124
+ self.context_window = context_window
125
+ self.compact_threshold = compact_threshold
126
+ self.remind_threshold = remind_threshold
127
+ self._reminded = False # fired the soft 50% nudge? re-arms below the mark
128
+ self.max_steps = max_steps
129
+ self.finalize_steps = finalize_steps
130
+ # Opt-in approval gate. Default = always-allow, so bench/headless and the
131
+ # current TUI stay unchanged until a real approver is injected (the TUI
132
+ # sets one that runs permission.decide() + a modal). Used by _approve.
133
+ self.approver = approver or _always_allow
134
+ # Plan mode: set to the session's plan-file path to make this engine
135
+ # read-only — planmode.gate() then runs on every tool call BEFORE the
136
+ # approver (the plan file itself is the one writable target). None = off.
137
+ # Host state only; the model never toggles it (docs/plan-mode-design.md).
138
+ self.plan_file: Optional[Path] = None
139
+ # Resolved paths a live human approval widened the READ jail to (read_file
140
+ # only). Mutable + held by reference in the registry, so an approval takes
141
+ # effect immediately. Never fed by a file — only a launch flag or a click.
142
+ self.read_grants: set[Path] = set()
143
+ # Compaction bookkeeping: the API's real prompt size for everything
144
+ # up to history[_sent_until], estimates for anything appended after.
145
+ self._last_prompt_tokens = 0
146
+ self._sent_until = 0
147
+ # Heuristic outcome counters (self-evolve phase 0), incremented at the
148
+ # exact branch points below and flushed as ONE `outcome` record by
149
+ # finalize_outcome() when the session ends.
150
+ self.stats = SessionStats()
151
+ self._outcome_written = False
152
+ self.workdir = workdir or Path.cwd()
153
+ self.allowed_roots = allowed_roots
154
+ # Explicit key AND endpoint (not the SDK's env fallbacks): an ambient
155
+ # OPENAI_API_KEY/OPENAI_BASE_URL must never decide what gets sent where.
156
+ self.client = client or AsyncOpenAI(api_key=require_key(), base_url=require_base_url(),
157
+ max_retries=5, timeout=300.0)
158
+ self.registry = (
159
+ registry if registry is not None
160
+ else tools_mod.build_registry(self.workdir, allowed_roots, read_grants=self.read_grants)
161
+ )
162
+ self.history: list[dict] = [{"role": "system", "content": system_prompt}]
163
+ # Collaboration mode (host-owned, like plan_file — the model never
164
+ # toggles it): set_mode() swaps a contract into the system prompt.
165
+ self._base_system = system_prompt
166
+ self._mode_contract: Optional[str] = None
167
+ self.mode_name: Optional[str] = None
168
+ self.resumed_mode: Optional[str] = None # mode seen in a resumed session's prompt
169
+ self.trajectory = TrajectoryLogger(
170
+ meta={
171
+ "model": model,
172
+ "thinking": thinking,
173
+ "reasoning_effort": reasoning_effort,
174
+ "max_tokens": max_tokens,
175
+ "context_window": context_window,
176
+ "max_steps": max_steps,
177
+ "workdir": str(self.workdir),
178
+ "base_url": require_base_url(),
179
+ **(trajectory_meta or {}),
180
+ }
181
+ )
182
+ self.trajectory.message(self.history[0])
183
+
184
+ def resume(self, messages: list[dict], *, from_session: Optional[str] = None) -> list[dict]:
185
+ """Seed history from a prior session's messages. Keeps the CURRENT
186
+ system prompt (so updated memory/skills/instructions apply) and carries
187
+ the rest of the conversation. The carried messages are logged into the
188
+ new trajectory, so it stays self-contained for training."""
189
+ # A prior session's mode lives in ITS system prompt, which we drop —
190
+ # remember the name so the UI can offer re-entry instead of silence.
191
+ self.resumed_mode = None
192
+ for m in messages:
193
+ if m.get("role") == "system":
194
+ for line in str(m.get("content", "")).splitlines():
195
+ if line.startswith("# Collaboration mode: "):
196
+ self.resumed_mode = line.removeprefix("# Collaboration mode: ").strip()
197
+ carried = [m for m in messages if m.get("role") != "system"]
198
+ for m in carried:
199
+ self._append(m)
200
+ self.trajectory.note({"resumed_from": from_session, "carried": len(carried)})
201
+ return carried
202
+
203
+ def swap_registry(self, registry: dict[str, tools_mod.Tool]) -> None:
204
+ """Hot-swap the tool registry at runtime (e.g. sandbox on/off)."""
205
+ self.registry = registry
206
+
207
+ def finalize_outcome(self) -> Optional[dict]:
208
+ """Write the session's heuristic outcome record (self-evolve phase 0).
209
+
210
+ Called once at session end — cli: after the app closes, before the
211
+ exit card. Idempotent, and skipped when no user turn ever ran, so an
212
+ open-and-close session doesn't grow a meaningless reward line. The
213
+ judge-graded outcome (source="judge") is appended much later, by dream.
214
+ """
215
+ if self._outcome_written or self.stats.turns == 0:
216
+ return None
217
+ self._outcome_written = True
218
+ data = {"source": "heuristic", **self.stats.as_data()}
219
+ self.trajectory.outcome(data)
220
+ return data
221
+
222
+ def set_base_system(self, text: str) -> None:
223
+ """Finalize the base system prompt after startup composition.
224
+
225
+ The chat CLI can only generate the "# Tools this session" section
226
+ once every tool has registered (skills/memory/web/artifact/goal/
227
+ explore land AFTER Engine construction), so the final prompt —
228
+ tools + language + environment + date — is set here, before the
229
+ first user turn. Re-layers an active mode contract; logs the final
230
+ prompt to the trajectory so training data records what was actually
231
+ sent (the construction-time record is superseded, last-system-wins).
232
+ """
233
+ self._base_system = text
234
+ if self.mode_name is not None and self._mode_contract is not None:
235
+ self.history[0]["content"] = (
236
+ f"{text}\n\n# Collaboration mode: {self.mode_name}\n\n"
237
+ f"{self._mode_contract.strip()}"
238
+ )
239
+ else:
240
+ self.history[0]["content"] = text
241
+ self.trajectory.message(self.history[0])
242
+ self.trajectory.note({"system_finalized": len(text)})
243
+
244
+ def set_mode(self, name: str, contract: str) -> None:
245
+ """Swap a collaboration contract into the system prompt (a REAL swap,
246
+ not an appended note — models weight the system prompt hardest). Costs
247
+ one prefix-cache miss at the switch, which the user accepted; switching
248
+ at launch (config `mode`) costs nothing. One mode at a time: setting a
249
+ new one replaces the old."""
250
+ self.history[0]["content"] = (
251
+ f"{self._base_system}\n\n# Collaboration mode: {name}\n\n{contract.strip()}"
252
+ )
253
+ self.mode_name = name
254
+ self._mode_contract = contract
255
+ self.trajectory.note({"mode": name})
256
+
257
+ def clear_mode(self) -> None:
258
+ """Back to the plain rocky contract."""
259
+ if self.mode_name is None:
260
+ return
261
+ self.history[0]["content"] = self._base_system
262
+ self.mode_name = None
263
+ self._mode_contract = None
264
+ self.trajectory.note({"mode": None})
265
+
266
+ def _extra_body(self) -> dict:
267
+ # self.reasoning_effort holds the dial value (high|xhigh|max) and is
268
+ # mutable mid-session (/effort); the provider clamp + param shape happen
269
+ # per call, keyed by the active provider's reasoning policy.
270
+ return build_extra_body(self.thinking, self.reasoning_effort, self.reasoning_policy)
271
+
272
+ def switch_provider(self, client, model: str, *, provider_name: str,
273
+ reasoning_policy: str, tools_enabled: bool = True) -> None:
274
+ """Point the engine at a different OpenAI-compatible endpoint/model live
275
+ (a /model switch). The caller builds the client with the provider's
276
+ base_url + key; we swap model + reasoning policy. The prompt-cache prefix
277
+ is provider-specific, so the switch naturally starts a fresh cache — no
278
+ stale-hit risk. History and tools are untouched (same OpenAI protocol)."""
279
+ self.client = client
280
+ self.model = model
281
+ self.provider_name = provider_name
282
+ self.reasoning_policy = reasoning_policy
283
+ self.tools_enabled = tools_enabled
284
+ self.trajectory.note({"provider": provider_name, "model": model})
285
+
286
+ def _repair_history(self) -> None:
287
+ """Inject synthetic tool responses for ANY orphaned tool_calls.
288
+
289
+ A tool_calls assistant message with no matching tool response for one of
290
+ its ids makes DeepSeek 400. This happens after a hard kill mid-tool, or
291
+ when a turn is cancelled (new submit / Esc). This is:
292
+ - idempotent: a call that already has a response is left alone, so it
293
+ is safe to call repeatedly (the turn's finally + the next turn's
294
+ pre-flight both call it) without ever producing a DUPLICATE response;
295
+ - in-position: a stub goes right after its own assistant message
296
+ (past any responses already there), never at the end — so it can't
297
+ land after a newly-appended user message and break ordering.
298
+ Fixes every orphan, not just the most recent (a multi-orphan resume
299
+ otherwise still 400s).
300
+ """
301
+ i = 0
302
+ while i < len(self.history):
303
+ msg = self.history[i]
304
+ if msg.get("role") == "assistant" and msg.get("tool_calls"):
305
+ answered = {
306
+ self.history[j].get("tool_call_id")
307
+ for j in range(i + 1, len(self.history))
308
+ if self.history[j].get("role") == "tool"
309
+ }
310
+ insert_at = i + 1
311
+ while insert_at < len(self.history) and self.history[insert_at].get("role") == "tool":
312
+ insert_at += 1
313
+ for tc in msg["tool_calls"]:
314
+ cid = tc.get("id", "")
315
+ if cid and cid not in answered:
316
+ self.history.insert(insert_at, {
317
+ "role": "tool",
318
+ "tool_call_id": cid,
319
+ "content": "[error] tool execution was interrupted",
320
+ })
321
+ insert_at += 1
322
+ i += 1
323
+
324
+ def _append(self, msg: dict) -> None:
325
+ self.history.append(msg)
326
+ self.trajectory.message(msg)
327
+
328
+ def _is_read(self, name: str) -> bool:
329
+ """A read-like tool — 'safe' risk tier, i.e. non-mutating (read_file /
330
+ grep / glob / check_code / recall_memory). Read calls can run concurrently
331
+ with one another; anything that writes or has side effects (edit / write /
332
+ bash / web / mcp) stays serial so two writes can never clobber."""
333
+ return getattr(self.registry.get(name), "risk", "risky") == "safe"
334
+
335
+ async def _approve(self, name: str, arguments_json: str) -> bool:
336
+ """Ask the injected approver whether this tool call may run. Parses args
337
+ to a dict for the approver; a malformed payload becomes {} (execute will
338
+ surface the real error). Short-circuits the always-allow default so
339
+ headless/bench runs pay nothing."""
340
+ if self.approver is _always_allow:
341
+ return True
342
+ try:
343
+ args = json.loads(arguments_json) if arguments_json.strip() else {}
344
+ except (json.JSONDecodeError, AttributeError):
345
+ args = {}
346
+ if not isinstance(args, dict):
347
+ args = {}
348
+ return bool(await self.approver(name, args))
349
+
350
+ def _plan_gate(self, name: str, arguments_json: str) -> planmode.Verdict:
351
+ """Plan-mode read-only gate, evaluated BEFORE the approver. Always
352
+ 'pass' when plan mode is off, so normal sessions pay nothing."""
353
+ if self.plan_file is None:
354
+ return planmode.PASS
355
+ try:
356
+ args = json.loads(arguments_json) if arguments_json.strip() else {}
357
+ except (json.JSONDecodeError, AttributeError):
358
+ args = {}
359
+ if not isinstance(args, dict):
360
+ args = {}
361
+ risk = getattr(self.registry.get(name), "risk", "risky")
362
+ return planmode.gate(name, args, risk, self.plan_file, self.workdir)
363
+
364
+ def _projected_prompt_tokens(self) -> int:
365
+ """Best guess at the next call's prompt size: the API's real count
366
+ for everything already sent, plus a conservative estimate for what
367
+ was appended since."""
368
+ return self._last_prompt_tokens + compaction.estimate_tokens(
369
+ self.history[self._sent_until :]
370
+ )
371
+
372
+ async def _maybe_compact(self, usage_total: dict[str, int]) -> AsyncIterator[Event]:
373
+ """Shrink history if the next call would crowd the context window.
374
+
375
+ Free deterministic prune first; an LLM state summary only when
376
+ pruning can't get under the limit. Never raises — a failed summary
377
+ degrades to the prune and the turn continues.
378
+ """
379
+ projected = self._projected_prompt_tokens()
380
+ # Soft, non-blocking nudge once we cross the degrade mark (~50%); re-arms
381
+ # after we drop back under it (a compaction, or a /clear).
382
+ if projected <= int(self.context_window * self.remind_threshold):
383
+ self._reminded = False
384
+ elif not self._reminded:
385
+ self._reminded = True
386
+ yield ContextReminder(pct=projected / self.context_window, window=self.context_window)
387
+
388
+ limit = int(self.context_window * self.compact_threshold)
389
+ if projected <= limit or len(self.history) < 3:
390
+ return
391
+ yield StateChanged(state=AgentState.COMPACTING)
392
+
393
+ before_msgs = len(self.history)
394
+ record: dict = {}
395
+ # Cap any single oversized message FIRST — a huge paste or tool result the
396
+ # tail keeps verbatim would otherwise survive prune (tool-only) and
397
+ # summarize (keeps the tail), keeping every step over the limit, or would
398
+ # overflow the summarize call itself. Head+tail truncation bounds it.
399
+ truncated = compaction.truncate_oversized(self.history)
400
+ if truncated:
401
+ record["truncated_oversized"] = truncated
402
+ projected = compaction.estimate_tokens(self.history)
403
+ tail = compaction.tail_start(
404
+ self.history,
405
+ min(compaction.KEEP_RECENT_TOKENS, compaction.estimate_tokens(self.history) // 2),
406
+ )
407
+ record["tail_start"] = tail
408
+
409
+ if projected - compaction.prune_savings(self.history, tail) <= limit:
410
+ strategy = "prune"
411
+ record["pruned_tool_outputs"] = compaction.prune_tool_outputs(self.history, tail)
412
+ else:
413
+ strategy = "summarize"
414
+ try:
415
+ summary, usage = await compaction.summarize(
416
+ self.client, self.model, self.history,
417
+ tools=[t.schema for t in self.registry.values()],
418
+ )
419
+ if not summary:
420
+ raise ValueError("model returned an empty summary")
421
+ except Exception as e: # noqa: BLE001 — degrade to prune, never kill the turn
422
+ strategy = "prune"
423
+ record["pruned_tool_outputs"] = compaction.prune_tool_outputs(self.history, tail)
424
+ yield EngineError(
425
+ message=f"compaction summary failed ({type(e).__name__}: {e}) — "
426
+ "pruned old tool outputs instead"
427
+ )
428
+ else:
429
+ _merge_usage(usage_total, usage)
430
+ _merge_usage(self.stats.usage, usage)
431
+ self.history = [self.history[0], compaction.state_message(summary)] + self.history[tail:]
432
+ record["summary"] = summary
433
+ record["new_history"] = list(self.history)
434
+
435
+ self._last_prompt_tokens = 0
436
+ self._sent_until = 0
437
+ self.stats.compactions += 1
438
+ tokens_after = compaction.estimate_tokens(self.history)
439
+ self.trajectory.compaction(
440
+ {
441
+ "strategy": strategy,
442
+ "tokens_before": projected,
443
+ "tokens_after_est": tokens_after,
444
+ "messages_before": before_msgs,
445
+ "messages_after": len(self.history),
446
+ **record,
447
+ }
448
+ )
449
+ yield Compacted(
450
+ strategy=strategy,
451
+ tokens_before=projected,
452
+ tokens_after=tokens_after,
453
+ messages_before=before_msgs,
454
+ messages_after=len(self.history),
455
+ )
456
+
457
+ async def run_turn(self, user_message: str) -> AsyncIterator[Event]:
458
+ """One full user turn: stream → maybe tools → stream → … → answer."""
459
+ yield TurnStarted(user_message=user_message) # UI sees the clean message
460
+ self.stats.turns += 1
461
+ # Plan mode rides the USER turn (never the system prompt/tools → the cached
462
+ # prompt prefix stays byte-identical when the mode toggles).
463
+ if self.plan_file is not None:
464
+ user_message = planmode.marker(self.plan_file) + "\n\n" + user_message
465
+ self._append({"role": "user", "content": user_message})
466
+
467
+ usage_total: dict[str, int] = {}
468
+ used_tools = False
469
+ steps = 0
470
+
471
+ while True:
472
+ steps += 1
473
+ self.stats.steps += 1
474
+
475
+ # max_steps <= 0 means UNLIMITED — no cap, no finalize, no budget
476
+ # warnings. Used by interactive chat so the model runs until it's
477
+ # done; the human (and Esc/new-message interrupt) is the backstop.
478
+ if self.max_steps > 0:
479
+ over = steps - self.max_steps # >0 ⇒ in the forced finalize window
480
+ if over > self.finalize_steps:
481
+ self.stats.engine_errors += 1
482
+ yield EngineError(
483
+ message=f"step limit ({self.max_steps}+{self.finalize_steps} finalize) "
484
+ "reached — stopping this turn."
485
+ )
486
+ break
487
+ if over >= 1 and not used_tools:
488
+ # never acted — no fix to finalize, just stop
489
+ self.stats.engine_errors += 1
490
+ yield EngineError(message=f"step limit ({self.max_steps}) reached — stopping this turn.")
491
+ break
492
+ if over == 1:
493
+ # entering finalize: one hard instruction to apply the fix now
494
+ self._append({"role": "user", "content": FINALIZE_NOTICE})
495
+ elif over <= 0:
496
+ # soft pressure while explore budget remains
497
+ warning = BUDGET_WARNINGS.get(self.max_steps - steps)
498
+ if warning is not None and used_tools:
499
+ self._append({"role": "user", "content": warning})
500
+
501
+ async for ev in self._maybe_compact(usage_total):
502
+ yield ev
503
+
504
+ self._repair_history()
505
+ sent = len(self.history)
506
+ # Clamp output to what the window can still hold: input can run up to
507
+ # 90% before auto-compaction and max_tokens may be the full 384K, so
508
+ # input + output could exceed the window. Reserve the rest for output,
509
+ # never below a usable floor.
510
+ out_room = self.context_window - self._projected_prompt_tokens() - 512
511
+ eff_max_tokens = max(1024, min(self.max_tokens, out_room))
512
+ try:
513
+ # tools omitted entirely when the provider profile is tools:off
514
+ # (a model that calls tools badly) — the loop then runs as a
515
+ # plain responder rather than shipping malformed tool calls.
516
+ _tools = [t.schema for t in self.registry.values()] if self.tools_enabled else None
517
+ stream = await self.client.chat.completions.create(
518
+ model=self.model,
519
+ messages=self.history,
520
+ tools=_tools,
521
+ max_tokens=eff_max_tokens,
522
+ stream=True,
523
+ stream_options={"include_usage": True},
524
+ extra_body=self._extra_body(),
525
+ )
526
+ except Exception as e: # noqa: BLE001 — surfaced to UI, turn ends
527
+ self.stats.engine_errors += 1
528
+ yield StateChanged(state=AgentState.ERROR)
529
+ # Auth-class rejections: say WHICH credential source was used
530
+ # and WHERE it was sent — "which key is rocky even using?"
531
+ # must never need an archaeology session (it once cost an
532
+ # evening). Never the key itself, only its source name.
533
+ hint = ""
534
+ if getattr(e, "status_code", None) in (400, 401, 402, 403):
535
+ hint = (f" · key source: {current_key_source() or 'none'}"
536
+ f" → {require_base_url()}")
537
+ yield EngineError(message=redact(f"{type(e).__name__}: {e}") + hint)
538
+ yield StateChanged(state=AgentState.IDLE)
539
+ return
540
+
541
+ content_parts: list[str] = []
542
+ reasoning_parts: list[str] = []
543
+ tool_calls: dict[int, dict] = {}
544
+ thinking_seen = False
545
+ content_seen = False
546
+
547
+ async for chunk in stream:
548
+ if chunk.usage is not None:
549
+ try:
550
+ u = chunk.usage.model_dump()
551
+ except AttributeError:
552
+ u = dict(chunk.usage)
553
+ _merge_usage(usage_total, u)
554
+ _merge_usage(self.stats.usage, u)
555
+ self.trajectory.usage(u) # per-call: prompt/completion + cache hit/miss
556
+ if isinstance(u.get("prompt_tokens"), int):
557
+ self._last_prompt_tokens = u["prompt_tokens"]
558
+ if not chunk.choices:
559
+ continue
560
+ delta = chunk.choices[0].delta
561
+
562
+ reasoning = getattr(delta, "reasoning_content", None)
563
+ if reasoning:
564
+ if not thinking_seen:
565
+ thinking_seen = True
566
+ yield StateChanged(state=AgentState.THINKING)
567
+ reasoning_parts.append(reasoning)
568
+ yield ThinkingDelta(text=reasoning)
569
+
570
+ if delta.content:
571
+ if not content_seen:
572
+ content_seen = True
573
+ yield StateChanged(state=AgentState.RESPONDING)
574
+ content_parts.append(delta.content)
575
+ yield TextDelta(text=delta.content)
576
+
577
+ for tc in delta.tool_calls or []:
578
+ acc = tool_calls.setdefault(
579
+ tc.index, {"id": "", "name": "", "arguments": ""}
580
+ )
581
+ if tc.id:
582
+ acc["id"] = tc.id
583
+ if tc.function:
584
+ if tc.function.name:
585
+ acc["name"] += tc.function.name
586
+ if tc.function.arguments:
587
+ acc["arguments"] += tc.function.arguments
588
+
589
+ self._sent_until = sent
590
+ content = "".join(content_parts)
591
+ # Trajectory-only (stays OUT of history — see module doc): the
592
+ # thinking trace, for language-adherence checks and RL export.
593
+ self.trajectory.reasoning("".join(reasoning_parts))
594
+
595
+ if not tool_calls:
596
+ # Final answer — reasoning_content deliberately not stored.
597
+ self._append({"role": "assistant", "content": content})
598
+ if used_tools:
599
+ yield StateChanged(state=AgentState.AMAZED)
600
+ yield TurnFinished(steps=steps, usage=usage_total)
601
+ yield StateChanged(state=AgentState.IDLE)
602
+ return
603
+
604
+ used_tools = True
605
+ calls = [tool_calls[i] for i in sorted(tool_calls)]
606
+ self._append(
607
+ {
608
+ "role": "assistant",
609
+ "content": content or None,
610
+ "tool_calls": [
611
+ {
612
+ "id": c["id"],
613
+ "type": "function",
614
+ "function": {"name": c["name"], "arguments": c["arguments"]},
615
+ }
616
+ for c in calls
617
+ ],
618
+ }
619
+ )
620
+
621
+ answered: set[str] = set()
622
+ # A read-only batch (every call is a non-mutating 'safe'-tier tool) runs
623
+ # CONCURRENTLY — reads can't conflict, and reading/grepping several files
624
+ # at once is the common explore step. Any batch containing a write / side
625
+ # effect stays SERIAL so two writes never clobber. Either way, responses
626
+ # are appended in the original tool_calls ORDER, which the API requires.
627
+ parallel = len(calls) > 1 and all(self._is_read(c["name"]) for c in calls)
628
+ try:
629
+ yield StateChanged(state=AgentState.TOOL)
630
+ if parallel:
631
+ for c in calls:
632
+ yield ToolStarted(call_id=c["id"], tool=c["name"], args={"raw": c["arguments"]})
633
+ # Approvals stay sequential (a modal can't run N-at-once); an
634
+ # in-workdir read auto-allows instantly, so this is usually a no-op.
635
+ allowed = {c["id"]: await self._approve(c["name"], c["arguments"]) for c in calls}
636
+ runnable = [c for c in calls if allowed[c["id"]]]
637
+ t0 = time.monotonic()
638
+ results = await asyncio.gather(*(
639
+ tools_mod.execute(self.registry, c["name"], c["arguments"]) for c in runnable
640
+ ))
641
+ dt = time.monotonic() - t0
642
+ out_by_id = {c["id"]: r for c, r in zip(runnable, results)}
643
+ for c in calls: # original order — required by the API
644
+ if c["id"] in out_by_id:
645
+ output, ok = out_by_id[c["id"]]
646
+ self.stats.observe_tool(c["name"], c["arguments"], output, ok)
647
+ else:
648
+ output, ok = "[denied] user rejected this tool call", False
649
+ self.stats.denials += 1
650
+ yield ToolFinished(call_id=c["id"], tool=c["name"], output=output, ok=ok, duration_s=dt)
651
+ self._append({"role": "tool", "tool_call_id": c["id"], "content": output})
652
+ answered.add(c["id"])
653
+ else:
654
+ for c in calls:
655
+ yield ToolStarted(call_id=c["id"], tool=c["name"], args={"raw": c["arguments"]})
656
+ t0 = time.monotonic()
657
+ # Plan mode (if on) gates BEFORE the approver: a write to
658
+ # the plan file runs, other mutations are denied with a
659
+ # teaching message, read-only calls fall through to ask.
660
+ verdict = self._plan_gate(c["name"], c["arguments"])
661
+ if verdict.action == "deny":
662
+ output, ok = verdict.message, False
663
+ self.stats.plan_denials += 1
664
+ elif verdict.action == "allow" or await self._approve(c["name"], c["arguments"]):
665
+ output, ok = await tools_mod.execute(self.registry, c["name"], c["arguments"])
666
+ self.stats.observe_tool(c["name"], c["arguments"], output, ok)
667
+ else:
668
+ output, ok = "[denied] user rejected this tool call", False
669
+ self.stats.denials += 1
670
+ yield ToolFinished(
671
+ call_id=c["id"],
672
+ tool=c["name"],
673
+ output=output,
674
+ ok=ok,
675
+ duration_s=time.monotonic() - t0,
676
+ )
677
+ self._append({"role": "tool", "tool_call_id": c["id"], "content": output})
678
+ answered.add(c["id"])
679
+ finally:
680
+ # Invariant: every tool_call_id above MUST get a matching tool response
681
+ # or the next request (and --resume) 400s. On interrupt (a new submit or
682
+ # Esc cancels the worker → CancelledError at an await) some calls never
683
+ # ran. Two parts:
684
+ # 1) in-memory history — _repair_history backfills IN-POSITION and
685
+ # IDEMPOTENTLY. Critical: a concurrent new turn may already have
686
+ # appended its user message + run its own repair, so the OLD
687
+ # end-append produced a duplicate + misordered tool message that
688
+ # bricked the session with a permanent 400. Reusing repair can't
689
+ # double-insert, and it's sync (no await) so it finishes before
690
+ # control returns to the new turn.
691
+ # 2) the append-only trajectory — log a stub for each un-run call so a
692
+ # --resume reload also has the response. Best-effort; a failing
693
+ # trajectory write must never mask the cancellation.
694
+ self._repair_history()
695
+ if any(c["id"] not in answered for c in calls):
696
+ # Esc / a new submit landed mid-batch — a strong "the user
697
+ # wanted something else" signal for the outcome record.
698
+ self.stats.interrupts += 1
699
+ for c in calls:
700
+ if c["id"] not in answered:
701
+ try:
702
+ self.trajectory.message({
703
+ "role": "tool",
704
+ "tool_call_id": c["id"],
705
+ "content": "[error] tool execution was interrupted",
706
+ })
707
+ except Exception: # noqa: BLE001 — never mask the cancel with I/O
708
+ pass
709
+
710
+ yield TurnFinished(steps=steps, usage=usage_total)
711
+ yield StateChanged(state=AgentState.IDLE)