pcli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pcli/__init__.py +1 -0
- pcli/__main__.py +4 -0
- pcli/agent/__init__.py +0 -0
- pcli/agent/activity.py +116 -0
- pcli/agent/compaction.py +205 -0
- pcli/agent/context_pruning.py +88 -0
- pcli/agent/headless.py +209 -0
- pcli/agent/loop.py +442 -0
- pcli/agent/prompt.py +371 -0
- pcli/agent/runtime.py +240 -0
- pcli/browser/__init__.py +0 -0
- pcli/browser/session.py +135 -0
- pcli/cli.py +757 -0
- pcli/config/__init__.py +0 -0
- pcli/config/paths.py +95 -0
- pcli/config/settings.py +435 -0
- pcli/cost/__init__.py +0 -0
- pcli/cost/context.py +275 -0
- pcli/cost/context_detect.py +183 -0
- pcli/cost/pricing_table.py +141 -0
- pcli/cost/tracker.py +126 -0
- pcli/llm/__init__.py +0 -0
- pcli/llm/client.py +285 -0
- pcli/llm/errors.py +37 -0
- pcli/llm/models.py +100 -0
- pcli/llm/streaming.py +108 -0
- pcli/memory/__init__.py +0 -0
- pcli/memory/extraction.py +106 -0
- pcli/memory/models.py +103 -0
- pcli/memory/store.py +88 -0
- pcli/permissions/__init__.py +0 -0
- pcli/permissions/guardrails.py +219 -0
- pcli/permissions/manager.py +215 -0
- pcli/permissions/policy.py +70 -0
- pcli/sandbox/__init__.py +0 -0
- pcli/sandbox/base.py +50 -0
- pcli/sandbox/docker_backend.py +107 -0
- pcli/sandbox/limits.py +63 -0
- pcli/sandbox/null_backend.py +92 -0
- pcli/sandbox/selector.py +75 -0
- pcli/sandbox/subprocess_backend.py +376 -0
- pcli/scheduler/__init__.py +0 -0
- pcli/scheduler/daemon.py +194 -0
- pcli/scheduler/models.py +97 -0
- pcli/scheduler/runner.py +84 -0
- pcli/scheduler/store.py +75 -0
- pcli/scheduler/triggers.py +84 -0
- pcli/session/__init__.py +0 -0
- pcli/session/audit.py +122 -0
- pcli/session/directory_check.py +28 -0
- pcli/session/export.py +57 -0
- pcli/session/importer.py +92 -0
- pcli/session/models.py +168 -0
- pcli/session/store.py +127 -0
- pcli/telegram/__init__.py +0 -0
- pcli/telegram/bot.py +266 -0
- pcli/telegram/daemon.py +1197 -0
- pcli/telegram/permissions.py +131 -0
- pcli/telegram/sender.py +58 -0
- pcli/tools/__init__.py +0 -0
- pcli/tools/_nested_agent.py +204 -0
- pcli/tools/agent_tools.py +264 -0
- pcli/tools/agent_tools_store.py +69 -0
- pcli/tools/artifacts.py +47 -0
- pcli/tools/base.py +185 -0
- pcli/tools/builtin/__init__.py +0 -0
- pcli/tools/builtin/agent_tool_register_tool.py +100 -0
- pcli/tools/builtin/artifact_tool.py +212 -0
- pcli/tools/builtin/ask_tool.py +77 -0
- pcli/tools/builtin/browser_tool.py +253 -0
- pcli/tools/builtin/decision_tool.py +73 -0
- pcli/tools/builtin/describe_tool.py +389 -0
- pcli/tools/builtin/diff_tools.py +225 -0
- pcli/tools/builtin/fs_tools.py +371 -0
- pcli/tools/builtin/grep_tool.py +88 -0
- pcli/tools/builtin/memory_tool.py +108 -0
- pcli/tools/builtin/network_tools.py +107 -0
- pcli/tools/builtin/pip_tool.py +106 -0
- pcli/tools/builtin/shell_tool.py +240 -0
- pcli/tools/builtin/subagent_tool.py +146 -0
- pcli/tools/builtin/todo_tool.py +122 -0
- pcli/tools/builtin/toolbox_register_tool.py +76 -0
- pcli/tools/builtin/web_tools.py +322 -0
- pcli/tools/pydiscovery/__init__.py +0 -0
- pcli/tools/pydiscovery/cache.py +51 -0
- pcli/tools/pydiscovery/index.py +48 -0
- pcli/tools/pydiscovery/invoke.py +181 -0
- pcli/tools/pydiscovery/search.py +117 -0
- pcli/tools/registry.py +138 -0
- pcli/tools/toolbox/__init__.py +0 -0
- pcli/tools/toolbox/introspect.py +48 -0
- pcli/tools/toolbox/manager.py +336 -0
- pcli/tools/toolbox/plugin_base.py +51 -0
- pcli/tools/toolbox/plugins/__init__.py +6 -0
- pcli/tools/toolbox/plugins/httpd.py +99 -0
- pcli/tools/toolbox/plugins/kafka.py +162 -0
- pcli/tools/toolbox/plugins/kubectl.py +211 -0
- pcli/tools/toolbox/plugins/sge.py +146 -0
- pcli/tools/toolbox/store.py +65 -0
- pcli/tools/toolbox/synthesize.py +100 -0
- pcli/tui/__init__.py +0 -0
- pcli/tui/app.py +37 -0
- pcli/tui/screens/__init__.py +0 -0
- pcli/tui/screens/ask_question_modal.py +54 -0
- pcli/tui/screens/chat.py +2070 -0
- pcli/tui/screens/confirm_modal.py +39 -0
- pcli/tui/screens/models.py +43 -0
- pcli/tui/screens/permission_modal.py +71 -0
- pcli/tui/screens/sessions.py +162 -0
- pcli/tui/screens/subagent_activity_modal.py +71 -0
- pcli/tui/shell_passthrough.py +56 -0
- pcli/tui/styles/pcli.tcss +241 -0
- pcli/tui/themes.py +84 -0
- pcli/tui/widgets/__init__.py +0 -0
- pcli/tui/widgets/chat_input.py +240 -0
- pcli/tui/widgets/command_suggestions.py +33 -0
- pcli/tui/widgets/message_view.py +328 -0
- pcli/tui/widgets/paste_input.py +99 -0
- pcli/tui/widgets/paste_marker.py +69 -0
- pcli/tui/widgets/status_bar.py +133 -0
- pcli/tui/widgets/status_pane.py +58 -0
- pcli/util/__init__.py +0 -0
- pcli/util/ids.py +15 -0
- pcli/util/logging.py +18 -0
- pcli/util/text.py +10 -0
- pcli_agent-0.1.0.dist-info/METADATA +259 -0
- pcli_agent-0.1.0.dist-info/RECORD +130 -0
- pcli_agent-0.1.0.dist-info/WHEEL +4 -0
- pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
pcli/agent/loop.py
ADDED
|
@@ -0,0 +1,442 @@
|
|
|
1
|
+
"""Turn orchestration: message history -> gateway stream -> tool dispatch ->
|
|
2
|
+
gateway stream (repeat) -> caller.
|
|
3
|
+
|
|
4
|
+
The dispatch loop is deliberately decoupled from sessions/TUI: it consumes a
|
|
5
|
+
plain message list and an `ask` callback (see permissions/manager.py), and
|
|
6
|
+
yields events the caller renders and, in TurnCompleteEvent, folds back into
|
|
7
|
+
its own persisted message history.
|
|
8
|
+
|
|
9
|
+
It's also where large tool results get truncated out of the conversation
|
|
10
|
+
and archived to the artifact library (see tools/artifacts.py) — a single
|
|
11
|
+
choke point that every tool's output passes through, rather than each tool
|
|
12
|
+
having to implement its own truncation.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
from collections.abc import AsyncIterator, Callable
|
|
19
|
+
from typing import Literal
|
|
20
|
+
|
|
21
|
+
import jsonschema
|
|
22
|
+
from pydantic import BaseModel, Field
|
|
23
|
+
|
|
24
|
+
from pcli.llm.client import GatewayClient
|
|
25
|
+
from pcli.llm.models import (
|
|
26
|
+
ChatMessage,
|
|
27
|
+
FinishEvent,
|
|
28
|
+
StreamEvent,
|
|
29
|
+
TextDelta,
|
|
30
|
+
ToolCall,
|
|
31
|
+
ToolCallCompleteEvent,
|
|
32
|
+
Usage,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
_TRUNCATION_FINISH_REASONS = frozenset({"length", "max_tokens"})
|
|
36
|
+
"""What different OpenAI-compatible gateways send as finish_reason when a
|
|
37
|
+
response was cut off by hitting the token cap mid-generation, rather than
|
|
38
|
+
the model choosing to stop - "length" is the official OpenAI value; "max_
|
|
39
|
+
tokens" covers at least one observed local-gateway variant. Checked
|
|
40
|
+
case-sensitively against the raw value, same as every other finish_reason
|
|
41
|
+
comparison in this codebase (llm/streaming.py's own "tool_calls" check)."""
|
|
42
|
+
from pcli.permissions.manager import AskCallback, PermissionManager
|
|
43
|
+
from pcli.tools.base import ToolContext
|
|
44
|
+
from pcli.tools.registry import ToolRegistry
|
|
45
|
+
|
|
46
|
+
_DEFAULT_ARTIFACT_THRESHOLD_CHARS = 4000
|
|
47
|
+
_ARTIFACT_PREVIEW_CHARS = 2000
|
|
48
|
+
_MAX_IDENTICAL_TOOL_CALL_REPEATS = 3
|
|
49
|
+
"""Confirmed against a real debugged session: a heavily-quantized local
|
|
50
|
+
model called run_shell with byte-identical arguments 12 times in a row,
|
|
51
|
+
getting the exact same (unhelpful) result each time, before finally trying
|
|
52
|
+
something else — the system prompt's own "Recovering from a failed tool
|
|
53
|
+
call" guidance already says a third near-identical retry is never the right
|
|
54
|
+
move, but that's a suggestion the model has to choose to follow. This is
|
|
55
|
+
the mechanical backstop: independent of whether the model notices or
|
|
56
|
+
complies, pcli itself refuses to run an exact repeat a third time — see
|
|
57
|
+
_dispatch_tool_call's own repeat-tracking below."""
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class ToolStartEvent(BaseModel):
|
|
61
|
+
kind: Literal["tool_start"] = "tool_start"
|
|
62
|
+
tool_call: ToolCall
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class ToolResultEvent(BaseModel):
|
|
66
|
+
kind: Literal["tool_result"] = "tool_result"
|
|
67
|
+
tool_call: ToolCall
|
|
68
|
+
output: str
|
|
69
|
+
is_error: bool = False
|
|
70
|
+
extra_usage: list[Usage] = Field(default_factory=list)
|
|
71
|
+
artifact_id: str | None = None
|
|
72
|
+
"""Set when `output` is a truncated preview whose full content was
|
|
73
|
+
archived — the id fetch_artifact needs to retrieve the rest."""
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class TurnCompleteEvent(BaseModel):
|
|
77
|
+
kind: Literal["turn_complete"] = "turn_complete"
|
|
78
|
+
new_messages: list[ChatMessage]
|
|
79
|
+
terminated_early: bool = False
|
|
80
|
+
"""True if the turn was cut off by max_tool_iterations or the
|
|
81
|
+
max_tool_calls_per_turn guardrail rather than the model choosing to
|
|
82
|
+
stop on its own - the model's work here is genuinely incomplete, not
|
|
83
|
+
just finished. Consumers that treat a turn's outcome as a result (namely
|
|
84
|
+
spawn_subagent/make_agent_tool reporting back to a parent loop) use this
|
|
85
|
+
to mark that result as an error instead of a normal completion."""
|
|
86
|
+
response_truncated: bool = False
|
|
87
|
+
"""True if the turn's final (no-tool-call) response was cut off by
|
|
88
|
+
hitting the token/length cap mid-generation (finish_reason in
|
|
89
|
+
_TRUNCATION_FINISH_REASONS) rather than the model actually finishing -
|
|
90
|
+
distinct from terminated_early above (a different cause: a guardrail
|
|
91
|
+
stopping an otherwise-healthy turn, not the model getting cut off
|
|
92
|
+
mid-sentence). A real observed failure this exists to let a caller
|
|
93
|
+
detect: the model says "Let me implement X:" and stops there with no
|
|
94
|
+
tool call, because it ran out of room to actually make one - previously
|
|
95
|
+
indistinguishable from a deliberate, complete stop, so the turn just
|
|
96
|
+
silently ended with genuinely unfinished work and no explanation."""
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
AgentEvent = StreamEvent | ToolStartEvent | ToolResultEvent | TurnCompleteEvent
|
|
100
|
+
|
|
101
|
+
ToolContextFactory = Callable[[], ToolContext]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class AgentLoop:
|
|
105
|
+
def __init__(
|
|
106
|
+
self,
|
|
107
|
+
gateway_client: GatewayClient,
|
|
108
|
+
*,
|
|
109
|
+
model: str | None = None,
|
|
110
|
+
tool_registry: ToolRegistry | None = None,
|
|
111
|
+
permission_manager: PermissionManager | None = None,
|
|
112
|
+
tool_context_factory: ToolContextFactory | None = None,
|
|
113
|
+
max_tool_iterations: int | None = 25,
|
|
114
|
+
artifact_threshold_chars: int = _DEFAULT_ARTIFACT_THRESHOLD_CHARS,
|
|
115
|
+
max_response_tokens: int | None = None,
|
|
116
|
+
temperature: float | None = None,
|
|
117
|
+
) -> None:
|
|
118
|
+
"""`max_tool_iterations=None` means unlimited (local-api mode).
|
|
119
|
+
`max_response_tokens=None` means no cap is sent (the gateway's own
|
|
120
|
+
default applies) — see set_max_response_tokens. `temperature=None`
|
|
121
|
+
means no temperature field is sent at all (the gateway/model's own
|
|
122
|
+
default applies) — see set_temperature."""
|
|
123
|
+
self._client = gateway_client
|
|
124
|
+
self._model = model
|
|
125
|
+
self._tool_registry = tool_registry
|
|
126
|
+
self._permission_manager = permission_manager
|
|
127
|
+
self._tool_context_factory = tool_context_factory
|
|
128
|
+
self._max_tool_iterations = max_tool_iterations
|
|
129
|
+
self._artifact_threshold_chars = artifact_threshold_chars
|
|
130
|
+
self._max_response_tokens = max_response_tokens
|
|
131
|
+
self._temperature = temperature
|
|
132
|
+
# Identical-tool-call repeat tracking (see _MAX_IDENTICAL_TOOL_CALL_
|
|
133
|
+
# REPEATS above) - reset at the start of every run_turn so a fresh
|
|
134
|
+
# user turn never inherits a stale count from an unrelated one.
|
|
135
|
+
self._last_tool_call_signature: str | None = None
|
|
136
|
+
self._identical_tool_call_repeats: int = 0
|
|
137
|
+
|
|
138
|
+
@property
|
|
139
|
+
def model(self) -> str | None:
|
|
140
|
+
return self._model
|
|
141
|
+
|
|
142
|
+
def set_model(self, model: str | None) -> None:
|
|
143
|
+
self._model = model
|
|
144
|
+
|
|
145
|
+
@property
|
|
146
|
+
def max_response_tokens(self) -> int | None:
|
|
147
|
+
"""Read side of set_max_response_tokens — lets a caller building a
|
|
148
|
+
nested AgentLoop for a subagent (spawn_subagent, agent_tools.py's
|
|
149
|
+
make_agent_tool) inherit the parent's current cap instead of the
|
|
150
|
+
subagent silently running with none at all."""
|
|
151
|
+
return self._max_response_tokens
|
|
152
|
+
|
|
153
|
+
@property
|
|
154
|
+
def temperature(self) -> float | None:
|
|
155
|
+
"""Read side of set_temperature — same inheritance purpose as
|
|
156
|
+
max_response_tokens above."""
|
|
157
|
+
return self._temperature
|
|
158
|
+
|
|
159
|
+
def set_tool_registry(self, tool_registry: ToolRegistry | None) -> None:
|
|
160
|
+
self._tool_registry = tool_registry
|
|
161
|
+
|
|
162
|
+
def set_max_tool_iterations(self, max_tool_iterations: int | None) -> None:
|
|
163
|
+
"""None means unlimited (local-api mode) — see __init__."""
|
|
164
|
+
self._max_tool_iterations = max_tool_iterations
|
|
165
|
+
|
|
166
|
+
def set_artifact_threshold_chars(self, artifact_threshold_chars: int) -> None:
|
|
167
|
+
self._artifact_threshold_chars = artifact_threshold_chars
|
|
168
|
+
|
|
169
|
+
def set_max_response_tokens(self, max_response_tokens: int | None) -> None:
|
|
170
|
+
"""The dynamic per-request max_tokens cap (cost/context.py's
|
|
171
|
+
compute_max_response_tokens) — recomputed and set by the caller
|
|
172
|
+
(ChatScreen) once per user-submitted turn, from live session usage
|
|
173
|
+
AgentLoop itself has no access to (it's deliberately decoupled from
|
|
174
|
+
sessions/TUI — see the module docstring). Applied to every
|
|
175
|
+
chat_stream call this run_turn makes, including tool-call
|
|
176
|
+
round-trips within the same turn, so it doesn't shrink further as
|
|
177
|
+
those round-trips add their own usage — a reasonable simplification
|
|
178
|
+
given the bug this fixes (a single very long response, not a
|
|
179
|
+
many-tool-call turn) rather than a live per-call recomputation."""
|
|
180
|
+
self._max_response_tokens = max_response_tokens
|
|
181
|
+
|
|
182
|
+
def set_temperature(self, temperature: float | None) -> None:
|
|
183
|
+
"""None means no temperature field is sent at all — see __init__."""
|
|
184
|
+
self._temperature = temperature
|
|
185
|
+
|
|
186
|
+
async def run_turn(
|
|
187
|
+
self,
|
|
188
|
+
messages: list[ChatMessage],
|
|
189
|
+
*,
|
|
190
|
+
ask: AskCallback | None = None,
|
|
191
|
+
budget_check: Callable[[], str | None] | None = None,
|
|
192
|
+
) -> AsyncIterator[AgentEvent]:
|
|
193
|
+
"""budget_check, if given, is called at the top of every loop
|
|
194
|
+
iteration (same spot as the max_tool_iterations check below) -
|
|
195
|
+
returning None means proceed, a reason string means stop the turn
|
|
196
|
+
the same way max_tool_iterations does. A plain injected callable
|
|
197
|
+
(not a raw Session/cap value) rather than importing Session here -
|
|
198
|
+
matches the existing ask/tool_context_factory injection pattern and
|
|
199
|
+
keeps this module decoupled from what a "budget" even is (see the
|
|
200
|
+
module docstring); the caller closes over its own Session/Settings
|
|
201
|
+
to build it - see cost/tracker.py's cost_budget_reason."""
|
|
202
|
+
tools = self._tool_registry.to_openai_tools() if self._tool_registry else None
|
|
203
|
+
working_messages = list(messages)
|
|
204
|
+
original_len = len(working_messages)
|
|
205
|
+
iterations = 0
|
|
206
|
+
self._last_tool_call_signature = None
|
|
207
|
+
self._identical_tool_call_repeats = 0
|
|
208
|
+
tool_calls_dispatched = 0
|
|
209
|
+
max_tool_calls_per_turn = (
|
|
210
|
+
self._permission_manager.guardrails.max_tool_calls_per_turn
|
|
211
|
+
if self._permission_manager is not None
|
|
212
|
+
else None
|
|
213
|
+
)
|
|
214
|
+
terminated_early = False
|
|
215
|
+
response_truncated = False
|
|
216
|
+
|
|
217
|
+
while True:
|
|
218
|
+
iterations += 1
|
|
219
|
+
if budget_check is not None:
|
|
220
|
+
budget_reason = budget_check()
|
|
221
|
+
if budget_reason is not None:
|
|
222
|
+
note = f"\n[pcli] {budget_reason}"
|
|
223
|
+
yield TextDelta(text=note)
|
|
224
|
+
working_messages.append(ChatMessage(role="assistant", content=note))
|
|
225
|
+
terminated_early = True
|
|
226
|
+
break
|
|
227
|
+
if self._max_tool_iterations is not None and iterations > self._max_tool_iterations:
|
|
228
|
+
note = (
|
|
229
|
+
f"\n[pcli] Reached the max tool-call iteration limit "
|
|
230
|
+
f"({self._max_tool_iterations}) for this turn. If the model legitimately "
|
|
231
|
+
"needs more tool calls to finish, raise max_tool_iterations via "
|
|
232
|
+
"PCLI_MAX_TOOL_ITERATIONS or config.toml (--local-api removes this cap "
|
|
233
|
+
"entirely for a local gateway)."
|
|
234
|
+
)
|
|
235
|
+
yield TextDelta(text=note)
|
|
236
|
+
working_messages.append(ChatMessage(role="assistant", content=note))
|
|
237
|
+
terminated_early = True
|
|
238
|
+
break
|
|
239
|
+
|
|
240
|
+
text_parts: list[str] = []
|
|
241
|
+
tool_calls_collected: list[ToolCall] = []
|
|
242
|
+
finish_reason: str | None = None
|
|
243
|
+
async for event in self._client.chat_stream(
|
|
244
|
+
working_messages, model=self._model, tools=tools,
|
|
245
|
+
max_tokens=self._max_response_tokens, temperature=self._temperature,
|
|
246
|
+
):
|
|
247
|
+
if event.kind == "text_delta":
|
|
248
|
+
text_parts.append(event.text)
|
|
249
|
+
if isinstance(event, ToolCallCompleteEvent):
|
|
250
|
+
tool_calls_collected = event.tool_calls
|
|
251
|
+
if isinstance(event, FinishEvent):
|
|
252
|
+
finish_reason = event.reason
|
|
253
|
+
yield event
|
|
254
|
+
|
|
255
|
+
assistant_text = "".join(text_parts) or None
|
|
256
|
+
|
|
257
|
+
if not tool_calls_collected:
|
|
258
|
+
working_messages.append(ChatMessage(role="assistant", content=assistant_text))
|
|
259
|
+
response_truncated = finish_reason in _TRUNCATION_FINISH_REASONS
|
|
260
|
+
break
|
|
261
|
+
|
|
262
|
+
working_messages.append(
|
|
263
|
+
ChatMessage(role="assistant", content=assistant_text, tool_calls=tool_calls_collected)
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
limit_hit = False
|
|
267
|
+
for call in tool_calls_collected:
|
|
268
|
+
yield ToolStartEvent(tool_call=call)
|
|
269
|
+
if (
|
|
270
|
+
max_tool_calls_per_turn is not None
|
|
271
|
+
and max_tool_calls_per_turn > 0 # <= 0 means unlimited (local-api mode)
|
|
272
|
+
and tool_calls_dispatched >= max_tool_calls_per_turn
|
|
273
|
+
):
|
|
274
|
+
# Still respond to every tool_call_id in this batch (required
|
|
275
|
+
# by the chat-completions protocol) rather than executing it.
|
|
276
|
+
denial = (
|
|
277
|
+
f"Denied: reached the guardrail limit of {max_tool_calls_per_turn} "
|
|
278
|
+
"tool call(s) for this turn."
|
|
279
|
+
)
|
|
280
|
+
output, is_error, extra_usage, artifact_id = denial, True, [], None
|
|
281
|
+
limit_hit = True
|
|
282
|
+
else:
|
|
283
|
+
output, is_error, extra_usage, artifact_id = await self._dispatch_tool_call(
|
|
284
|
+
call, ask=ask
|
|
285
|
+
)
|
|
286
|
+
tool_calls_dispatched += 1
|
|
287
|
+
working_messages.append(
|
|
288
|
+
ChatMessage(
|
|
289
|
+
role="tool", tool_call_id=call.id, name=call.function.name, content=output
|
|
290
|
+
)
|
|
291
|
+
)
|
|
292
|
+
yield ToolResultEvent(
|
|
293
|
+
tool_call=call,
|
|
294
|
+
output=output,
|
|
295
|
+
is_error=is_error,
|
|
296
|
+
extra_usage=extra_usage,
|
|
297
|
+
artifact_id=artifact_id,
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
if limit_hit:
|
|
301
|
+
note = (
|
|
302
|
+
f"\n[pcli] Reached the guardrail limit of {max_tool_calls_per_turn} tool "
|
|
303
|
+
"call(s) for this turn. If this is expected, raise limits."
|
|
304
|
+
"max_tool_calls_per_turn in guardrails.toml (--local-api removes this cap "
|
|
305
|
+
"entirely for a local gateway)."
|
|
306
|
+
)
|
|
307
|
+
yield TextDelta(text=note)
|
|
308
|
+
working_messages.append(ChatMessage(role="assistant", content=note))
|
|
309
|
+
terminated_early = True
|
|
310
|
+
break
|
|
311
|
+
|
|
312
|
+
yield TurnCompleteEvent(
|
|
313
|
+
new_messages=working_messages[original_len:],
|
|
314
|
+
terminated_early=terminated_early,
|
|
315
|
+
response_truncated=response_truncated,
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
def _archive_if_large(self, output: str, ctx: ToolContext) -> tuple[str, str | None]:
|
|
319
|
+
if len(output) <= self._artifact_threshold_chars or ctx.artifact_store is None:
|
|
320
|
+
return output, None
|
|
321
|
+
|
|
322
|
+
artifact_id = ctx.artifact_store.put(output)
|
|
323
|
+
preview_chars = min(_ARTIFACT_PREVIEW_CHARS, self._artifact_threshold_chars)
|
|
324
|
+
preview = output[:preview_chars]
|
|
325
|
+
truncated = (
|
|
326
|
+
f"{preview}\n\n[...output truncated: {len(output)} chars total, archived as "
|
|
327
|
+
f"artifact_id='{artifact_id}'. Call fetch_artifact(artifact_id='{artifact_id}') "
|
|
328
|
+
"if you need the rest...]"
|
|
329
|
+
)
|
|
330
|
+
return truncated, artifact_id
|
|
331
|
+
|
|
332
|
+
async def _dispatch_tool_call(
|
|
333
|
+
self, call: ToolCall, *, ask: AskCallback | None
|
|
334
|
+
) -> tuple[str, bool, list[Usage], str | None]:
|
|
335
|
+
tool = self._tool_registry.get(call.function.name) if self._tool_registry else None
|
|
336
|
+
if tool is None:
|
|
337
|
+
return f"Unknown tool: {call.function.name}", True, [], None
|
|
338
|
+
|
|
339
|
+
try:
|
|
340
|
+
arguments = json.loads(call.function.arguments or "{}")
|
|
341
|
+
except json.JSONDecodeError as exc:
|
|
342
|
+
return f"Invalid arguments JSON: {exc}", True, [], None
|
|
343
|
+
if not isinstance(arguments, dict):
|
|
344
|
+
return "Tool arguments must be a JSON object.", True, [], None
|
|
345
|
+
|
|
346
|
+
# "purpose" only exists in the advertised schema (see
|
|
347
|
+
# ToolSpec.to_openai_tool) for the model's own benefit - it's never
|
|
348
|
+
# part of a tool's real parameters, so it must never reach schema
|
|
349
|
+
# validation or the handler. Popped from this parsed copy only; the
|
|
350
|
+
# original call.function.arguments JSON string (as persisted in the
|
|
351
|
+
# session's assistant message) is left untouched, which is what lets
|
|
352
|
+
# context_pruning.py re-extract it later for a pruned placeholder.
|
|
353
|
+
arguments.pop("purpose", None)
|
|
354
|
+
|
|
355
|
+
# Mechanical backstop for a real observed failure mode (see
|
|
356
|
+
# _MAX_IDENTICAL_TOOL_CALL_REPEATS above): a model that keeps
|
|
357
|
+
# resending the exact same call, byte-for-byte, expecting a
|
|
358
|
+
# different result. Signature is built from the same
|
|
359
|
+
# purpose-stripped `arguments` dict everything below uses, so
|
|
360
|
+
# changing only the "purpose" explanation while repeating the same
|
|
361
|
+
# real action still counts as a repeat. Checked before the
|
|
362
|
+
# permission gate specifically so an already-decided identical call
|
|
363
|
+
# can't re-trigger another interactive prompt for the same thing.
|
|
364
|
+
signature = f"{tool.name}:{json.dumps(arguments, sort_keys=True)}"
|
|
365
|
+
if signature == self._last_tool_call_signature:
|
|
366
|
+
self._identical_tool_call_repeats += 1
|
|
367
|
+
else:
|
|
368
|
+
self._last_tool_call_signature = signature
|
|
369
|
+
self._identical_tool_call_repeats = 1
|
|
370
|
+
if self._identical_tool_call_repeats >= _MAX_IDENTICAL_TOOL_CALL_REPEATS:
|
|
371
|
+
blocked_message = (
|
|
372
|
+
f"Blocked: this exact {tool.name} call (identical arguments) has now been "
|
|
373
|
+
f"attempted {self._identical_tool_call_repeats} times in a row with no change "
|
|
374
|
+
"in between — pcli is refusing to run it again, since repeating it will not "
|
|
375
|
+
"produce a different result. Stop and change strategy: re-read the actual "
|
|
376
|
+
"output from the previous attempts above, diagnose why it didn't help, and try "
|
|
377
|
+
"something meaningfully different — or use ask_user_question if you're stuck."
|
|
378
|
+
)
|
|
379
|
+
return blocked_message, True, [], None
|
|
380
|
+
|
|
381
|
+
try:
|
|
382
|
+
jsonschema.validate(arguments, tool.parameters)
|
|
383
|
+
except jsonschema.ValidationError as exc:
|
|
384
|
+
return f"Arguments failed schema validation: {exc.message}", True, [], None
|
|
385
|
+
|
|
386
|
+
if self._permission_manager is None:
|
|
387
|
+
return "No permission manager configured; tool execution is disabled.", True, [], None
|
|
388
|
+
|
|
389
|
+
if self._tool_context_factory is None:
|
|
390
|
+
return "No tool execution context configured.", True, [], None
|
|
391
|
+
# Built before the permission check (constructing it is side-effect
|
|
392
|
+
# free) so check() can attach a PermissionGrant to ctx.session when
|
|
393
|
+
# the user picks "remember for session/always".
|
|
394
|
+
ctx = self._tool_context_factory()
|
|
395
|
+
|
|
396
|
+
# Defense in depth: the registry not exposing a tool is necessary but
|
|
397
|
+
# not sufficient on its own — this is the real backstop against a
|
|
398
|
+
# stale/hallucinated tool call slipping through while plan mode is
|
|
399
|
+
# active, independent of whatever registry happens to be wired up.
|
|
400
|
+
if ctx.plan_mode and not tool.plan_mode_safe:
|
|
401
|
+
return "Denied: not available in plan mode.", True, [], None
|
|
402
|
+
|
|
403
|
+
command = arguments.get(tool.guardrail_command_arg) if tool.guardrail_command_arg else None
|
|
404
|
+
path = arguments.get(tool.guardrail_path_arg) if tool.guardrail_path_arg else None
|
|
405
|
+
python_module = (
|
|
406
|
+
arguments.get(tool.guardrail_python_module_arg)
|
|
407
|
+
if tool.guardrail_python_module_arg
|
|
408
|
+
else None
|
|
409
|
+
)
|
|
410
|
+
decision, deny_reason = await self._permission_manager.check_with_reason(
|
|
411
|
+
tool.name,
|
|
412
|
+
arguments,
|
|
413
|
+
command=command,
|
|
414
|
+
path=path,
|
|
415
|
+
python_module=python_module,
|
|
416
|
+
ask=ask,
|
|
417
|
+
risk_description=tool.risk_description,
|
|
418
|
+
default_allow=not tool.needs_permission,
|
|
419
|
+
session=ctx.session,
|
|
420
|
+
)
|
|
421
|
+
if decision == "deny":
|
|
422
|
+
message = f"Permission denied: {deny_reason}." if deny_reason else "Permission denied."
|
|
423
|
+
return message, True, [], None
|
|
424
|
+
|
|
425
|
+
try:
|
|
426
|
+
result = await tool.handler(arguments, ctx)
|
|
427
|
+
except Exception as exc: # noqa: BLE001 - surface any tool failure to the model
|
|
428
|
+
return f"Tool raised an exception: {exc}", True, [], None
|
|
429
|
+
|
|
430
|
+
# Global backstop on top of each tool's own (smaller) internal cap —
|
|
431
|
+
# guardrails.max_output_bytes is meant to bound every tool uniformly,
|
|
432
|
+
# not just the ones that happen to implement their own limit.
|
|
433
|
+
raw_output = result.output
|
|
434
|
+
max_output_bytes = self._permission_manager.guardrails.max_output_bytes
|
|
435
|
+
if len(raw_output) > max_output_bytes:
|
|
436
|
+
raw_output = (
|
|
437
|
+
f"{raw_output[:max_output_bytes]}\n"
|
|
438
|
+
f"[...output truncated to the guardrail limit of {max_output_bytes} chars...]"
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
output, artifact_id = self._archive_if_large(raw_output, ctx)
|
|
442
|
+
return output, result.is_error, result.extra_usage, artifact_id
|