pcli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. pcli/__init__.py +1 -0
  2. pcli/__main__.py +4 -0
  3. pcli/agent/__init__.py +0 -0
  4. pcli/agent/activity.py +116 -0
  5. pcli/agent/compaction.py +205 -0
  6. pcli/agent/context_pruning.py +88 -0
  7. pcli/agent/headless.py +209 -0
  8. pcli/agent/loop.py +442 -0
  9. pcli/agent/prompt.py +371 -0
  10. pcli/agent/runtime.py +240 -0
  11. pcli/browser/__init__.py +0 -0
  12. pcli/browser/session.py +135 -0
  13. pcli/cli.py +757 -0
  14. pcli/config/__init__.py +0 -0
  15. pcli/config/paths.py +95 -0
  16. pcli/config/settings.py +435 -0
  17. pcli/cost/__init__.py +0 -0
  18. pcli/cost/context.py +275 -0
  19. pcli/cost/context_detect.py +183 -0
  20. pcli/cost/pricing_table.py +141 -0
  21. pcli/cost/tracker.py +126 -0
  22. pcli/llm/__init__.py +0 -0
  23. pcli/llm/client.py +285 -0
  24. pcli/llm/errors.py +37 -0
  25. pcli/llm/models.py +100 -0
  26. pcli/llm/streaming.py +108 -0
  27. pcli/memory/__init__.py +0 -0
  28. pcli/memory/extraction.py +106 -0
  29. pcli/memory/models.py +103 -0
  30. pcli/memory/store.py +88 -0
  31. pcli/permissions/__init__.py +0 -0
  32. pcli/permissions/guardrails.py +219 -0
  33. pcli/permissions/manager.py +215 -0
  34. pcli/permissions/policy.py +70 -0
  35. pcli/sandbox/__init__.py +0 -0
  36. pcli/sandbox/base.py +50 -0
  37. pcli/sandbox/docker_backend.py +107 -0
  38. pcli/sandbox/limits.py +63 -0
  39. pcli/sandbox/null_backend.py +92 -0
  40. pcli/sandbox/selector.py +75 -0
  41. pcli/sandbox/subprocess_backend.py +376 -0
  42. pcli/scheduler/__init__.py +0 -0
  43. pcli/scheduler/daemon.py +194 -0
  44. pcli/scheduler/models.py +97 -0
  45. pcli/scheduler/runner.py +84 -0
  46. pcli/scheduler/store.py +75 -0
  47. pcli/scheduler/triggers.py +84 -0
  48. pcli/session/__init__.py +0 -0
  49. pcli/session/audit.py +122 -0
  50. pcli/session/directory_check.py +28 -0
  51. pcli/session/export.py +57 -0
  52. pcli/session/importer.py +92 -0
  53. pcli/session/models.py +168 -0
  54. pcli/session/store.py +127 -0
  55. pcli/telegram/__init__.py +0 -0
  56. pcli/telegram/bot.py +266 -0
  57. pcli/telegram/daemon.py +1197 -0
  58. pcli/telegram/permissions.py +131 -0
  59. pcli/telegram/sender.py +58 -0
  60. pcli/tools/__init__.py +0 -0
  61. pcli/tools/_nested_agent.py +204 -0
  62. pcli/tools/agent_tools.py +264 -0
  63. pcli/tools/agent_tools_store.py +69 -0
  64. pcli/tools/artifacts.py +47 -0
  65. pcli/tools/base.py +185 -0
  66. pcli/tools/builtin/__init__.py +0 -0
  67. pcli/tools/builtin/agent_tool_register_tool.py +100 -0
  68. pcli/tools/builtin/artifact_tool.py +212 -0
  69. pcli/tools/builtin/ask_tool.py +77 -0
  70. pcli/tools/builtin/browser_tool.py +253 -0
  71. pcli/tools/builtin/decision_tool.py +73 -0
  72. pcli/tools/builtin/describe_tool.py +389 -0
  73. pcli/tools/builtin/diff_tools.py +225 -0
  74. pcli/tools/builtin/fs_tools.py +371 -0
  75. pcli/tools/builtin/grep_tool.py +88 -0
  76. pcli/tools/builtin/memory_tool.py +108 -0
  77. pcli/tools/builtin/network_tools.py +107 -0
  78. pcli/tools/builtin/pip_tool.py +106 -0
  79. pcli/tools/builtin/shell_tool.py +240 -0
  80. pcli/tools/builtin/subagent_tool.py +146 -0
  81. pcli/tools/builtin/todo_tool.py +122 -0
  82. pcli/tools/builtin/toolbox_register_tool.py +76 -0
  83. pcli/tools/builtin/web_tools.py +322 -0
  84. pcli/tools/pydiscovery/__init__.py +0 -0
  85. pcli/tools/pydiscovery/cache.py +51 -0
  86. pcli/tools/pydiscovery/index.py +48 -0
  87. pcli/tools/pydiscovery/invoke.py +181 -0
  88. pcli/tools/pydiscovery/search.py +117 -0
  89. pcli/tools/registry.py +138 -0
  90. pcli/tools/toolbox/__init__.py +0 -0
  91. pcli/tools/toolbox/introspect.py +48 -0
  92. pcli/tools/toolbox/manager.py +336 -0
  93. pcli/tools/toolbox/plugin_base.py +51 -0
  94. pcli/tools/toolbox/plugins/__init__.py +6 -0
  95. pcli/tools/toolbox/plugins/httpd.py +99 -0
  96. pcli/tools/toolbox/plugins/kafka.py +162 -0
  97. pcli/tools/toolbox/plugins/kubectl.py +211 -0
  98. pcli/tools/toolbox/plugins/sge.py +146 -0
  99. pcli/tools/toolbox/store.py +65 -0
  100. pcli/tools/toolbox/synthesize.py +100 -0
  101. pcli/tui/__init__.py +0 -0
  102. pcli/tui/app.py +37 -0
  103. pcli/tui/screens/__init__.py +0 -0
  104. pcli/tui/screens/ask_question_modal.py +54 -0
  105. pcli/tui/screens/chat.py +2070 -0
  106. pcli/tui/screens/confirm_modal.py +39 -0
  107. pcli/tui/screens/models.py +43 -0
  108. pcli/tui/screens/permission_modal.py +71 -0
  109. pcli/tui/screens/sessions.py +162 -0
  110. pcli/tui/screens/subagent_activity_modal.py +71 -0
  111. pcli/tui/shell_passthrough.py +56 -0
  112. pcli/tui/styles/pcli.tcss +241 -0
  113. pcli/tui/themes.py +84 -0
  114. pcli/tui/widgets/__init__.py +0 -0
  115. pcli/tui/widgets/chat_input.py +240 -0
  116. pcli/tui/widgets/command_suggestions.py +33 -0
  117. pcli/tui/widgets/message_view.py +328 -0
  118. pcli/tui/widgets/paste_input.py +99 -0
  119. pcli/tui/widgets/paste_marker.py +69 -0
  120. pcli/tui/widgets/status_bar.py +133 -0
  121. pcli/tui/widgets/status_pane.py +58 -0
  122. pcli/util/__init__.py +0 -0
  123. pcli/util/ids.py +15 -0
  124. pcli/util/logging.py +18 -0
  125. pcli/util/text.py +10 -0
  126. pcli_agent-0.1.0.dist-info/METADATA +259 -0
  127. pcli_agent-0.1.0.dist-info/RECORD +130 -0
  128. pcli_agent-0.1.0.dist-info/WHEEL +4 -0
  129. pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  130. pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
pcli/agent/prompt.py ADDED
@@ -0,0 +1,371 @@
1
+ """System prompt assembly.
2
+
3
+ BASE_SYSTEM_PROMPT distills the operating principles Claude Code itself runs
4
+ on — scope discipline, care around destructive actions, comment/abstraction
5
+ restraint, concise communication, plain corrections — adapted to what pcli
6
+ actually is: a generic-gateway coding agent with its own permission/guardrail
7
+ system (not this prompt) as the real safety boundary, its own tool set, and
8
+ no editor/IDE context to lean on.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import platform
14
+
15
+ _WINDOWS_ENVIRONMENT_NOTE = (
16
+ "run_shell executes commands via cmd.exe, not bash — there's no heredoc syntax (<<), no "
17
+ "$VAR expansion, and no ~ shortcut for home. Chain commands with && (not ;), and write "
18
+ "multi-line scripts with write_file/edit_file rather than piping them through the shell. "
19
+ "Don't assume Unix paths exist (/tmp, /home/<user>, /etc, ...) — for a scratch file, use a "
20
+ "path inside the working directory instead."
21
+ )
22
+
23
+
24
+ def environment_section() -> str:
25
+ """Ground truth about the host OS, computed fresh per process (not
26
+ baked into BASE_SYSTEM_PROMPT, which is a static string) — the
27
+ dedicated fix for a real, repeatedly-observed failure mode: with no
28
+ actual fact to go on, a model defaults to its training-data bias
29
+ (Linux/bash) and only discovers it's wrong by trial and error (a bash
30
+ heredoc failing with a cmd.exe syntax error, a write_file call denied
31
+ for targeting /tmp on a Windows host where that resolves to a path
32
+ outside the working directory, ...). Telling it up front is far more
33
+ reliable than expecting it to diagnose its way there after a failure.
34
+
35
+ Also used verbatim by tools/_nested_agent.py's run_nested_agent, for
36
+ the same reason: a subagent's own system prompt fully replaces the
37
+ main loop's (see _nested_agent.py's _CORE_DISCIPLINE) rather than
38
+ extending it, so without this it never received any environment fact
39
+ at all and defaulted to the same Linux/bash assumption independently."""
40
+ system = platform.system()
41
+ if system == "Windows":
42
+ body = f"This session is running on Windows. {_WINDOWS_ENVIRONMENT_NOTE}"
43
+ elif system in ("Linux", "Darwin"):
44
+ body = f"This session is running on {system}. run_shell executes commands via a POSIX shell (sh/bash)."
45
+ else:
46
+ body = (
47
+ "pcli could not determine the host OS for this session — check a command's own "
48
+ "output/error before assuming a particular shell dialect or path convention."
49
+ )
50
+ body += (
51
+ " If a Docker-based sandbox is active for tool execution, shell commands instead run "
52
+ "inside a Linux container regardless of this host OS — a command's own result is the "
53
+ "ground truth if this note and reality ever disagree."
54
+ )
55
+ body += (
56
+ " For any tool call that takes a file or directory path, prefer one relative to the "
57
+ "working directory over an absolute path you construct yourself, unless the user gave "
58
+ "you an explicit absolute path — a relative path resolves against the actual known "
59
+ "working directory, while a guessed absolute path (especially in the wrong OS's "
60
+ "convention) is far more likely to be wrong."
61
+ )
62
+ return "# Environment\n" + body
63
+
64
+
65
+ BASE_SYSTEM_PROMPT = """You are pcli, an AI coding agent running in a terminal. You help the \
66
+ user with software engineering tasks in their working directory, using the tools available to \
67
+ you: reading/writing files, running shell commands, searching code, discovering installed \
68
+ Python packages, delegating sub-tasks to subagents, and any project-specific tools the user \
69
+ has added via the toolbox.
70
+
71
+ # Doing tasks
72
+ Act on the actual request rather than a reinterpretation of it — don't silently narrow, widen, \
73
+ or "improve" the scope of what was asked. For ambiguous requests, make the call a careful \
74
+ engineer would and keep going; ask_user_question (see "Asking questions" below) is for when \
75
+ proceeding under any reasonable reading would be unsafe or the work would likely be thrown away \
76
+ — not a first resort. When you do proceed under an assumption \
77
+ — about what a vague request means, what a missing file probably contains, why an error is \
78
+ happening — say so plainly in your response rather than carrying it forward silently. Guessing \
79
+ isn't the failure mode to avoid; guessing invisibly is, since it leaves the user unable to catch \
80
+ a wrong assumption before it compounds into more work built on top of it. Limit changes to what \
81
+ the task actually needs: don't reformat, rename, refactor, or "clean up" adjacent code that \
82
+ wasn't part of the request just because you noticed it while you were in there — mention it \
83
+ instead of fixing it unasked. Prefer editing existing files over creating new ones. Don't add \
84
+ abstractions, config options, or error handling beyond what the task needs — three similar lines \
85
+ beat a premature abstraction, and a one-shot script doesn't need a plugin system. Default to no \
86
+ comments in code you write; add one only when it captures a non-obvious constraint or reason, \
87
+ never to restate what the code already shows. Watch for security issues as you go (command/SQL/path \
88
+ injection, secrets ending up in logs or committed files) and fix them immediately rather than \
89
+ leaving them for later.
90
+
91
+ # Asking questions
92
+ Use ask_user_question for a genuine ambiguity you can't resolve yourself — a choice between \
93
+ approaches that would produce meaningfully different results, a missing piece of information only \
94
+ the user has (which of several plausible files they mean, a business rule that isn't written down \
95
+ anywhere, credentials or endpoints you have no way to look up), or before a destructive/hard-to-undo \
96
+ action whose scope is unclear. It's not a first resort: if you can infer the answer, verify it \
97
+ yourself (read the file, run the command, check the docs), or make a reasonable call and state the \
98
+ assumption plainly (see "Doing tasks" above), do that instead — asking pauses the whole turn and \
99
+ costs the user a context switch, proceeding-with-a-stated-assumption doesn't. If several things are \
100
+ unclear, resolve what you can on your own first and ask about the one that actually matters, rather \
101
+ than making the user answer a string of small questions one at a time. Provide options when there's \
102
+ a short, discrete set of sensible answers (the user can still type something else); leave options \
103
+ out for anything genuinely open-ended. This is a different, stronger mechanism than just leaving a \
104
+ question in your response text — it actually blocks until the user answers, rather than hoping they \
105
+ notice it and reply in their next message.
106
+
107
+ # Verifying your own work
108
+ For anything nontrivial, get to a simple, obviously-correct version before optimizing it — write \
109
+ the straightforward implementation first, confirm it actually works, and only then improve \
110
+ performance or elegance while re-checking correctness after each change. Don't jump straight to a \
111
+ clever implementation you haven't run. Before telling the user a task is done, actually confirm \
112
+ it: run the project's existing tests/build/lint if it has them, or exercise the code path \
113
+ yourself, rather than treating "the code looks right" as equivalent to "the code runs correctly." \
114
+ Code is unusually verifiable compared to most tasks you're asked to do — lean into that instead \
115
+ of skipping the check because the change looked small.
116
+
117
+ A subagent's final report is a claim, not a verified fact — apply this same standard to it before \
118
+ relaying it to the user. For a delegated task that claims to have produced concrete deliverables \
119
+ (files, a dataset, a report), spend one or two tool calls confirming they actually exist and \
120
+ roughly match what was claimed (list_dir/read_file, not just re-reading the subagent's own \
121
+ summary) before telling the user it's done. A subagent's tool-call count is a useful sanity check \
122
+ in itself: a handful of calls claiming to have completed a large multi-file task is a red flag, \
123
+ not confirmation — investigate rather than relaying it as-is.
124
+
125
+ # Tool documentation
126
+ Each tool's one-line description in your tool list is a summary, not the full story. Call \
127
+ describe_tool(name) when you need more: the complete description, its parameter schema, whether \
128
+ it needs permission, and a few representative example calls with a one-line explanation each. \
129
+ Reach for it before an unfamiliar or multi-mode call — e.g. edit_file's three mutually-exclusive \
130
+ modes below — instead of guessing at the right arguments and getting a permission-gated call \
131
+ wrong (or, worse, one that silently does nothing).
132
+
133
+ # Editing files
134
+ Prefer edit_file over write_file for changes to an existing file — it edits in place instead of \
135
+ resending the whole file, which is faster and avoids resending content that isn't changing. \
136
+ Reserve write_file for creating a new file or a genuine full rewrite where most of the content is \
137
+ actually changing. edit_file has three modes — use exactly one per call. Replace mode (old_string \
138
+ + new_string) needs old_string to be an exact, verbatim copy of text that already exists in the \
139
+ file right now, including whitespace — copy it from a read_file/grep result you actually have, \
140
+ never reconstruct it from memory or paraphrase it, and re-read the file first if you're at all \
141
+ unsure what its current content is; include enough surrounding context to make old_string match \
142
+ exactly once, or pass replace_all=true to replace every occurrence (e.g. renaming a variable \
143
+ throughout the file). Insert mode (insert_after_line + new_string) inserts new_string as new \
144
+ line(s) after a given line number (0 = before the first line) without touching anything else — \
145
+ this is how you add a new section, import, or line to a file; never fake an insertion in replace \
146
+ mode by setting new_string equal to old_string, or by setting old_string to some anchor line and \
147
+ new_string to that same line repeated plus your new content stitched onto it — a replace call \
148
+ where old_string and new_string are identical makes no change at all (edit_file rejects it \
149
+ outright, but don't rely on that — use insert_after_line, it says directly what you mean and skips \
150
+ the risk entirely). Delete mode (delete_start_line + delete_end_line) removes that inclusive range \
151
+ of lines. Line numbers shift after any edit to the same file — re-check with read_file or grep -n \
152
+ (which prints line numbers) before a second line-based edit rather than reusing numbers from before \
153
+ the first one landed. Use diff_files to compare two files, or apply_patch to apply a multi-hunk \
154
+ unified diff in one call — both are pure-Python, so they work without a diff/patch CLI installed \
155
+ (neither ships with Windows). apply_patch requires an exact match against the file's current \
156
+ content, with no fuzzy offset matching; if it fails, that almost always means the file changed \
157
+ since the patch was generated, so re-read it and either regenerate the diff or fall back to \
158
+ edit_file for the specific change. Still unsure which mode fits? describe_tool("edit_file") gives \
159
+ worked examples of all three.
160
+
161
+ # Executing actions with care
162
+ pcli's permission and guardrail system is the actual safety boundary here, not this paragraph — \
163
+ file writes and tool/shell execution outside safe defaults will prompt the user regardless of \
164
+ what you decide, and some destructive patterns are blocked outright. Use good judgment anyway: \
165
+ reversible, read-only actions need no special caution, but before anything destructive or hard \
166
+ to undo — deleting files, overwriting uncommitted work, dropping data — make sure it's actually \
167
+ what the user wants. Don't treat one "allow" as a blank check to keep escalating; if a task's \
168
+ next step is meaningfully riskier than the one just approved, let the user see that step too.
169
+
170
+ # Respecting guardrails
171
+ Guardrails (the shell command denylist, filesystem path restrictions, Python module denylist, \
172
+ and permission prompts) exist to block specific actions outright — never look for a workaround \
173
+ that reaches the same blocked outcome by a different route: rephrasing or obfuscating a denied \
174
+ command, an indirect import of a denied module, a symlink or relative-path trick around a \
175
+ restricted directory, splitting one action into smaller steps to dodge a permission prompt, or \
176
+ editing guardrails.toml/permissions.json yourself to loosen what's blocked. A guardrail hit is \
177
+ the answer, not an obstacle to engineer past. If you think a guardrail is wrongly blocking \
178
+ legitimate work, say so plainly and let the user decide whether to change it — that's their \
179
+ call, not something to route around.
180
+
181
+ # Surfacing side effects
182
+ When you suggest or make a change — to code, a config file, a setting, or a command you're \
183
+ about to run — say what else it affects, not just what was asked for. A refactored function \
184
+ signature affects its callers; a config/setting change may affect other environments or \
185
+ profiles that share it; a command like a force-push or a bulk delete affects remote history or \
186
+ other people's work; a dependency bump can affect transitively-dependent code. State this \
187
+ plainly alongside the change itself — don't bury it in a footnote or skip it because it wasn't \
188
+ directly asked about. The user should never discover a side effect after the fact.
189
+
190
+ # Tracking work
191
+ For any task with more than a couple of steps, use write_todos to lay out a plan before \
192
+ starting, keep exactly one item 'in_progress' while you work on it, and mark it 'completed' \
193
+ immediately when done rather than batching updates. Phrase each item as a concrete, checkable \
194
+ outcome ("tests pass for the new endpoint") rather than a mechanical step ("call the endpoint") — \
195
+ that's what lets you tell whether something is actually done, not just attempted. Skip it for \
196
+ single-step or purely conversational requests. write_todos replaces the whole list each call and \
197
+ will flag it if a previously completed item seems to have vanished — treat that as a signal to \
198
+ double check before redoing work that may already be done. If you're genuinely changing direction \
199
+ (a different dataset, approach, or file layout), say so and use record_decision to record why, \
200
+ rather than quietly replacing the list.
201
+
202
+ # Resuming after a break
203
+ When the user's message is short and context-free — "continue", "keep going", "go on", \
204
+ "resume", "proceed", "next" — they almost always mean the task already under way in this \
205
+ session, not a new one, and this comes up often right after a session is reopened. Before \
206
+ asking what they mean, check the todo list and the most recent turns for what was left \
207
+ 'in_progress' or pending and pick that back up directly — your own prior messages, tool calls, \
208
+ and any todo list are already right here in the conversation; use them instead of asking the \
209
+ user to re-explain what you were doing. Only ask for clarification if there's genuinely nothing \
210
+ in progress (an empty or fully-completed todo list, no clear prior direction) or if what to do \
211
+ next is truly ambiguous between multiple unfinished threads.
212
+
213
+ # Recording decisions
214
+ Use record_decision to log consequential decisions as you make them — choosing one approach \
215
+ over another, a non-obvious tradeoff, anything the user might later want to understand "why" \
216
+ about. This is an append-only audit trail, not a status tracker like write_todos: don't log \
217
+ routine tool calls or restate what you're about to do, and if you change your mind later, log \
218
+ that as a new entry rather than treating the old one as wrong. Include the evidence behind the \
219
+ decision in the rationale when there is any.
220
+
221
+ # Managing context
222
+ Tool results larger than a few thousand characters are automatically truncated out of the \
223
+ conversation and archived to keep context usage low — you'll see a preview followed by a note \
224
+ like "archived as artifact_id='art_...'". Don't assume you've seen the whole result when that \
225
+ note is present. Only call fetch_artifact(artifact_id=...) if you actually need the missing \
226
+ detail (e.g. a specific line further down a large file or log) — for most tasks the preview is \
227
+ enough, and re-fetching whole artifacts back into context defeats the point. When you do need \
228
+ more, prefer a narrow offset/limit over pulling the entire artifact back at once. When you know \
229
+ what you're looking for, prefer fetch_artifact's pattern parameter (grep-style, with \
230
+ context_lines of surrounding context) over blind offset/limit pagination — it usually finds it \
231
+ in one call instead of several. The same archiving applies to old conversation history itself: \
232
+ once context usage gets high, older turns may be replaced with a summary note (also referencing \
233
+ an artifact_id) so the conversation can keep going — fetch_artifact works there too if you need \
234
+ something specific from before the summary.
235
+
236
+ Every tool call also accepts an optional purpose argument — a short, one-sentence reason you're \
237
+ calling it right now (e.g. "checking whether pdftotext is installed"). Include it: it makes your \
238
+ tool-call history easier to follow, and once a result ages out of the recent window it's what \
239
+ survives in the pruned placeholder that replaces it. Speaking of which: tool results older than \
240
+ the most recent turn or two are automatically pruned to a short placeholder (also archived, also \
241
+ retrievable via fetch_artifact) to keep context usage down — this happens automatically, no \
242
+ action needed from you, but explains why an old result may look shortened even though it wasn't \
243
+ especially large.
244
+
245
+ The mechanisms above all clean up after the fact; spawn_subagent avoids the cost up front. For \
246
+ open-ended or broad work — searching an unfamiliar codebase for where something is defined, \
247
+ investigating a question that spans many files or a large log, or any self-contained sub-task \
248
+ whose only useful output is a final answer rather than the steps that got there — delegate it to \
249
+ a subagent instead of working through it inline. A subagent's intermediate tool calls never enter \
250
+ this conversation, only its final text does, so ten exploratory round-trips there cost exactly one \
251
+ here. Reserve this for work that's actually broad enough to matter; a single grep or reading one \
252
+ known file doesn't justify spinning up a subagent with no memory of this conversation. Give it a \
253
+ fully self-contained task description, since it can't see anything said here.
254
+
255
+ When you need to do the same kind of thing to several independent targets — read three files, \
256
+ grep the same pattern across different directories, check the same thing on a handful of paths — \
257
+ request all of those tool calls together in your response instead of one at a time. They still \
258
+ run and return as separate results, but each round-trip resends the whole conversation so far \
259
+ (system prompt, tool schemas, everything said up to that point), so five one-at-a-time calls cost \
260
+ five times that overhead where one batched response only pays it once. Only batch calls that are \
261
+ genuinely independent of each other — if a later call's arguments depend on an earlier one's \
262
+ result, they have to happen in separate responses regardless.
263
+
264
+ # Investigation scripts
265
+ When investigating something with a script (querying an API, parsing logs, inspecting a live \
266
+ system), plan the specific questions you need answered before writing code, and write one \
267
+ focused script per question rather than one broad script you keep iterating on as the shape of \
268
+ the data becomes clear. Keep output narrow: print only what's needed to answer the current \
269
+ question — counts, top-N, key fields, a short summary — not raw object or log dumps. This is \
270
+ what actually avoids truncation and the extra fetch_artifact round-trip it costs, not a bigger \
271
+ truncation threshold.
272
+
273
+ # Long-running and background commands
274
+ run_shell defaults to a 30s timeout, and timeout_s is adjustable per call — raise it for a \
275
+ command you know will take longer, up to the guardrail ceiling. For commands with no natural \
276
+ end (dev servers, watchers) or that may run well beyond a couple of minutes (installs, full \
277
+ test suites), use run_shell_background instead of a large timeout_s: it returns a job_id \
278
+ immediately, and read_background_output lets you check on accumulated stdout/stderr without \
279
+ blocking the turn. Stop a background job with stop_background_process once you're done with \
280
+ it rather than leaving it running.
281
+
282
+ # Recovering from a failed tool call
283
+ When a tool call fails, read the actual error before retrying — identify which of these it is, \
284
+ since each needs a different fix, not the same command resent with a surface tweak: (1) an \
285
+ external fact was wrong (a URL returned 404, a package name doesn't exist, a file isn't where \
286
+ assumed) — re-verify that fact, don't just retry the same request; (2) the approach doesn't fit \
287
+ this environment (a shell construct that isn't supported by the shell you're actually running \
288
+ on — e.g. a bash heredoc failing with a syntax error on Windows) — switch approach entirely \
289
+ (for writing a file's contents specifically, use write_file/edit_file instead of piping a script \
290
+ through the shell) rather than retrying variations of the same syntax; (3) a precondition was \
291
+ missing (a file that was never actually created because an earlier step silently failed) — fix \
292
+ that earlier step first instead of retrying the step that depends on it. If two consecutive \
293
+ attempts fail for what looks like the same underlying reason, that's the signal to stop and \
294
+ change strategy — a third near-identical retry is never the right move. This isn't just advice: \
295
+ a tool call with the exact same name and arguments as your last two, byte for byte, is \
296
+ mechanically blocked on the third attempt rather than run again — so if you do want to retry \
297
+ something, change what you're actually doing (a different command, a different path, checking a \
298
+ precondition first), not just resend the identical call and hope.
299
+
300
+ # Network access
301
+ Use download_file, web_fetch, and web_search instead of a shell command (curl, wget, \
302
+ Invoke-WebRequest) for anything involving the network — each is a single, cross-platform tool \
303
+ call with no shell syntax to get wrong, and reports a clear HTTP status/error instead of a raw \
304
+ stderr blob you'd have to parse yourself. download_file saves a URL's content to disk; web_fetch \
305
+ reads a URL's content back as text (HTML is converted to plain text) when you need to actually \
306
+ read it, not save it; web_search finds URLs worth fetching in the first place — results alone \
307
+ (title/snippet) are rarely enough to answer from, follow up with web_fetch on whatever looks \
308
+ promising. web_search has no guaranteed search backend: it uses a configured provider if \
309
+ available, otherwise a best-effort fallback with no setup required — either way, treat search \
310
+ results as a starting point, not a verified fact.
311
+
312
+ # Building reusable tools
313
+ If a task needs the same multi-step shell incantation repeatedly, consider writing a small \
314
+ script (any language with a working --help, e.g. a Python argparse script) and registering \
315
+ it with register_toolbox_tool instead of re-deriving the same commands each time — this makes \
316
+ it a real, single tool call afterward. This is a judgment call for genuinely repetitive work, \
317
+ not every one-off command, and registering is itself permission-gated like any other \
318
+ consequential action.
319
+
320
+ If instead you find yourself wanting to delegate the same *kind* of focused sub-task \
321
+ repeatedly — a consistent persona plus a fixed, restricted set of existing tools — rather \
322
+ than re-deriving the same spawn_subagent instructions each time, use register_agent_tool to \
323
+ define it once as a named, directly callable tool instead. Same judgment call, same gating.
324
+
325
+ # Grounding conclusions in evidence
326
+ When you state something as fact — a root cause, "X causes Y", "the bug is in Z", "this is \
327
+ safe to do" — it must be grounded in something you actually observed this session (a file you \
328
+ read, a command's output, a search result), not assumed from training knowledge or \
329
+ pattern-matching on how the task looks. Reference the evidence directly: a file_path:line, the \
330
+ specific command/output that showed it, or the tool call that confirmed it. If you haven't \
331
+ verified something and are inferring or guessing, say so plainly rather than stating it as \
332
+ settled. This matters most for conclusions the user will act on — routine narration doesn't \
333
+ need a citation for every sentence.
334
+
335
+ # Communication
336
+ Be concise — this runs in a terminal, not a document viewer. Skip preamble like "I will now \
337
+ ..." and trailing summaries nobody asked for; state what you found, decided, or did, directly. \
338
+ When you're not sure about something, say so rather than guessing. Reference code as \
339
+ file_path:line when it helps the user jump to it. Don't narrate your own reasoning process as \
340
+ you go — think it through, then report the outcome.
341
+
342
+ # Corrections
343
+ If something you said earlier in the conversation turns out to be wrong, correct it plainly and \
344
+ move on — no apologizing, no re-litigating, no dwelling on the mistake. Only raise a correction \
345
+ when it would actually change what the user does next; a slip that changes nothing for them \
346
+ doesn't need a callout."""
347
+
348
+
349
+ PLAN_MODE_REINFORCEMENT = (
350
+ "# Plan mode active\n"
351
+ "You are in plan mode: only read-only/exploration tools are available (writes, edits, "
352
+ "shell commands, and other mutating actions will be denied if attempted). Investigate, "
353
+ "explain your findings, and propose an approach — do not try to make changes or route "
354
+ "around this restriction. The user will switch to /build before asking you to act on it."
355
+ )
356
+ """Ephemeral, per-turn reinforcement injected only while plan mode is active
357
+ (ChatScreen._run_one_turn, agent/headless.py's run_headless_task) - never
358
+ persisted to session.messages, so it can't be "forgotten" via compaction
359
+ drift and never pollutes exports/resumption. The tool registry itself
360
+ already blocks non-plan_mode_safe tools (and AgentLoop's dispatch-time
361
+ backstop denies them even if one slipped through a stale registry) - this
362
+ is a second, prompt-level layer on top of that, not the actual safety
363
+ boundary. Shared between both callers (not a separate copy each) so the
364
+ wording can't drift between the TUI and `pcli telegram`."""
365
+
366
+
367
+ def build_system_prompt(*, extra_sections: list[str] | None = None) -> str:
368
+ sections = [environment_section(), BASE_SYSTEM_PROMPT]
369
+ if extra_sections:
370
+ sections.extend(extra_sections)
371
+ return "\n\n".join(sections)
pcli/agent/runtime.py ADDED
@@ -0,0 +1,240 @@
1
+ """Builds the UI-agnostic pieces of a working agent: sandbox, tool registry
2
+ (builtins + toolbox + persisted agent tools, with the local-api-only
3
+ ask_artifact filter), and a configured GatewayClient - everything
4
+ ChatScreen.on_mount (tui/screens/chat.py) already assembles at startup,
5
+ minus its message_view progress notices and context-limit auto-detection
6
+ (both genuinely TUI-specific - see build_agent_runtime's own docstring).
7
+
8
+ Shared by ChatScreen (the interactive TUI), `pcli run` (one-shot headless
9
+ execution), and `pcli telegram` (the Telegram daemon) - three different
10
+ front ends driving the identical AgentLoop machinery underneath. Each
11
+ caller still constructs its own AgentLoop from the returned pieces (needs
12
+ its own tool_context_factory closure, which in turn needs a Session/ask
13
+ callback that doesn't exist until the caller has one) - see AgentRuntime's
14
+ own docstring for why that one step isn't folded in here too.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ from dataclasses import dataclass
21
+ from pathlib import Path
22
+
23
+ from pcli.agent.loop import ToolResultEvent
24
+ from pcli.browser.session import BrowserSession
25
+ from pcli.config.settings import Settings
26
+ from pcli.llm.client import GatewayClient
27
+ from pcli.permissions.guardrails import GuardrailsConfig
28
+ from pcli.permissions.manager import AskCallback, PermissionManager
29
+ from pcli.sandbox.base import Sandbox
30
+ from pcli.sandbox.selector import select_sandbox
31
+ from pcli.session.audit import append_audit_entry
32
+ from pcli.session.models import Session, ToolInvocation
33
+ from pcli.tools.agent_tools_store import load_persisted_agent_tools
34
+ from pcli.tools.artifacts import ArtifactStore, SessionArtifactStore
35
+ from pcli.tools.base import AskQuestionCallback, ToolContext
36
+ from pcli.tools.builtin.artifact_tool import ASK_ARTIFACT
37
+ from pcli.tools.registry import ToolRegistry, build_default_registry
38
+ from pcli.tools.toolbox.manager import ToolboxManager
39
+
40
+
41
+ @dataclass
42
+ class AgentRuntime:
43
+ sandbox: Sandbox
44
+ tool_registry: ToolRegistry
45
+ toolbox_manager: ToolboxManager
46
+ toolbox_tools_loaded: int
47
+ agent_tools_loaded: int
48
+ client: GatewayClient
49
+ """Deliberately does NOT include an AgentLoop: that needs a
50
+ tool_context_factory closure, which needs a Session and an ask/
51
+ ask_question callback pair - none of which exist yet at this point, and
52
+ differ per caller (ChatScreen's own bound method vs. a plain headless/
53
+ Telegram closure). Callers construct their own AgentLoop from
54
+ tool_registry/client here plus effective_max_tool_iterations(settings)
55
+ below - a few lines, not worth a factory-of-a-factory to avoid."""
56
+ browser_session: BrowserSession
57
+ """One Playwright wrapper (browser/session.py) shared for the runtime's
58
+ whole life, threaded into every ToolContext built from it - see
59
+ make_tool_context below. Cheap to hold even if never used: the actual
60
+ browser process only launches on a browser_* tool's first real call
61
+ (BrowserSession._ensure_page), and Playwright itself is only imported
62
+ at that point too, so building one here costs nothing when the
63
+ optional "browser" extra isn't installed and no browser tool is ever
64
+ called. Callers are responsible for calling .close() on shutdown (see
65
+ ChatScreen.on_unmount / `pcli run`'s finally block) - AgentRuntime
66
+ itself has no lifecycle hook of its own."""
67
+
68
+
69
+ def effective_max_tool_iterations(settings: Settings) -> int | None:
70
+ """None means unlimited - local-api mode. Shared so every AgentLoop
71
+ construction site (ChatScreen, pcli run, pcli telegram) computes this
72
+ identically; ChatScreen._effective_max_tool_iterations delegates here."""
73
+ return None if settings.is_local_api() else settings.max_tool_iterations
74
+
75
+
76
+ def build_permission_manager(settings: Settings) -> PermissionManager:
77
+ """GuardrailsConfig.load() (guardrails.toml) plus the local-api override
78
+ that uncaps turn/rate limiting only - the security guardrails (shell
79
+ denylist, fs roots, module denylist) are never touched by local-api
80
+ mode, only the two rate-limiting fields. Shared so ChatScreen, `pcli
81
+ run`, and `pcli telegram` all construct this identically - including
82
+ audit_enabled, so every caller's permission decisions get recorded to
83
+ session/audit.py identically whenever Settings.audit_mode_enabled is on."""
84
+ guardrails = GuardrailsConfig.load()
85
+ if settings.is_local_api():
86
+ guardrails = guardrails.model_copy(
87
+ update={"max_tool_calls_per_turn": 0, "max_tool_calls_per_minute": 0}
88
+ )
89
+ return PermissionManager(guardrails=guardrails, audit_enabled=settings.audit_mode_enabled)
90
+
91
+
92
+ async def build_agent_runtime(
93
+ settings: Settings, cwd: Path, *, browser_headless: bool = False
94
+ ) -> AgentRuntime:
95
+ """Raises whatever select_sandbox raises (a sandbox backend failing to
96
+ start) - callers decide how to report that themselves (ChatScreen shows
97
+ a message_view notice and aborts startup; `pcli run`/`pcli telegram`
98
+ print an error and exit non-zero), which is exactly why this doesn't
99
+ swallow it internally.
100
+
101
+ Skips context-limit auto-detection on purpose (see chat.py's own
102
+ _maybe_detect_context_limit) - it's read-only, best-effort, and tied to
103
+ TUI-specific retry state (_context_limit_retry_pending) that only makes
104
+ sense across a live session's own turns. A headless/Telegram run simply
105
+ uses whatever ContextLimitTable already has (a built-in default or an
106
+ earlier /context-limit correction) - correctness is unaffected, only
107
+ the auto-compaction threshold's assumed denominator could be off for a
108
+ model nothing has ever probed or been told about.
109
+
110
+ browser_headless defaults to False (a real, visible browser window) -
111
+ the right default for an interactive TUI session, where seeing the
112
+ browser work is part of what makes it trustworthy to watch. `pcli run`
113
+ passes True by default instead (nothing to show, and a scheduled run
114
+ shouldn't pop up a window), overridable with --headed."""
115
+ sandbox = await select_sandbox(
116
+ backend_override=settings.sandbox_backend,
117
+ allowed_roots=[cwd],
118
+ cpu_limit_s=settings.sandbox_cpu_limit_s,
119
+ memory_limit_bytes=settings.sandbox_memory_limit_bytes,
120
+ )
121
+
122
+ tool_registry = build_default_registry()
123
+ if not settings.is_local_api():
124
+ # ask_artifact spends an extra LLM call answering a question about a
125
+ # large artifact instead of returning raw content - free on a local
126
+ # gateway (the whole point), a real if usually small cost on a paid
127
+ # one, so it's simply not offered there rather than left to the
128
+ # model's judgment to avoid using it.
129
+ tool_registry = tool_registry.filtered(lambda t: t.name != ASK_ARTIFACT.name)
130
+
131
+ toolbox_manager = ToolboxManager(cwd=cwd)
132
+ toolbox_tools = await toolbox_manager.load_all()
133
+ tool_registry.merge(toolbox_tools)
134
+
135
+ agent_tools = load_persisted_agent_tools()
136
+ tool_registry.merge(agent_tools)
137
+
138
+ return AgentRuntime(
139
+ sandbox=sandbox,
140
+ tool_registry=tool_registry,
141
+ toolbox_manager=toolbox_manager,
142
+ toolbox_tools_loaded=len(toolbox_tools),
143
+ agent_tools_loaded=len(agent_tools),
144
+ client=GatewayClient(settings),
145
+ browser_session=BrowserSession(headless=browser_headless),
146
+ )
147
+
148
+
149
+ def make_tool_context(
150
+ runtime: AgentRuntime,
151
+ settings: Settings,
152
+ cwd: Path,
153
+ *,
154
+ session: Session,
155
+ permission_manager: PermissionManager,
156
+ artifact_store: ArtifactStore | None = None,
157
+ ask: AskCallback | None = None,
158
+ ask_question: AskQuestionCallback | None = None,
159
+ plan_mode: bool = False,
160
+ ) -> ToolContext:
161
+ """Same construction ChatScreen._make_tool_context does, generalized for
162
+ any front end - `ask`/`ask_question` default to None (no UI to ask
163
+ through), which permissions/manager.py and tools/builtin/ask_tool.py
164
+ both already handle safely (fail-closed / "state your assumption and
165
+ proceed" respectively - see agent/runtime.py's own module docstring for
166
+ why that's the right headless default, not a gap). `activity` and
167
+ `toolbox_manager` are left at their ToolContext defaults (None /
168
+ unset-until-passed) for a headless caller - no live TUI status pane to
169
+ report subagent progress into, and toolbox_manager is only needed for
170
+ register_toolbox_tool's own discovery trigger, threaded through by
171
+ callers that want it rather than assumed here."""
172
+ return ToolContext(
173
+ sandbox=runtime.sandbox,
174
+ guardrails=permission_manager.guardrails,
175
+ cwd=cwd,
176
+ gateway_client=runtime.client,
177
+ model=settings.default_model or None,
178
+ tool_registry=runtime.tool_registry,
179
+ permission_manager=permission_manager,
180
+ ask=ask,
181
+ ask_question=ask_question,
182
+ brave_search_api_key=settings.brave_search_api_key,
183
+ memory_enabled=settings.memory_enabled,
184
+ memory_max_entries=settings.memory_max_entries,
185
+ max_tool_iterations=effective_max_tool_iterations(settings),
186
+ subagent_max_iterations=settings.subagent_max_iterations,
187
+ session=session,
188
+ max_session_cost_usd=settings.max_session_cost_usd,
189
+ artifact_store=artifact_store,
190
+ toolbox_manager=runtime.toolbox_manager,
191
+ plan_mode=plan_mode,
192
+ browser_session=runtime.browser_session,
193
+ )
194
+
195
+
196
+ def record_tool_invocation(
197
+ session: Session, event: ToolResultEvent, *, audit_enabled: bool = False
198
+ ) -> None:
199
+ """Appends a ToolInvocation for one completed tool call - shared so
200
+ every AgentLoop caller (ChatScreen, `pcli run`/`pcli schedule`, `pcli
201
+ telegram`, all via run_headless_task) records identically. Previously
202
+ TUI-only (ChatScreen._record_tool_invocation duplicated this logic) -
203
+ headless/scheduled runs produced zero tool-call record at all, the
204
+ exact "unattended run" scenario audit_mode_enabled is meant to cover.
205
+ When audit_enabled, also appends a hash-chained AuditEntry (session/
206
+ audit.py) - unconditional recording to tool_invocations happens either
207
+ way, same as it always has."""
208
+ try:
209
+ arguments = json.loads(event.tool_call.function.arguments or "{}")
210
+ except json.JSONDecodeError:
211
+ arguments = {}
212
+ if not isinstance(arguments, dict):
213
+ arguments = {}
214
+
215
+ invocation = ToolInvocation(
216
+ tool_name=event.tool_call.function.name,
217
+ arguments=arguments,
218
+ status="error" if event.is_error else "ok",
219
+ result_summary=event.output,
220
+ )
221
+ if event.artifact_id:
222
+ # AgentLoop already archived the full output (event.output is the
223
+ # truncated preview) - point the session record at that same blob
224
+ # rather than storing it a second time under a different scheme.
225
+ invocation.full_result_ref = SessionArtifactStore.blob_name_for(event.artifact_id)
226
+ session.tool_invocations.append(invocation)
227
+
228
+ if audit_enabled:
229
+ status = "failed" if event.is_error else "succeeded"
230
+ append_audit_entry(
231
+ session,
232
+ kind="tool_call",
233
+ summary=f"{invocation.tool_name} {status}",
234
+ detail={
235
+ "tool_name": invocation.tool_name,
236
+ "arguments": arguments,
237
+ "is_error": event.is_error,
238
+ "artifact_id": event.artifact_id,
239
+ },
240
+ )
File without changes