pcli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. pcli/__init__.py +1 -0
  2. pcli/__main__.py +4 -0
  3. pcli/agent/__init__.py +0 -0
  4. pcli/agent/activity.py +116 -0
  5. pcli/agent/compaction.py +205 -0
  6. pcli/agent/context_pruning.py +88 -0
  7. pcli/agent/headless.py +209 -0
  8. pcli/agent/loop.py +442 -0
  9. pcli/agent/prompt.py +371 -0
  10. pcli/agent/runtime.py +240 -0
  11. pcli/browser/__init__.py +0 -0
  12. pcli/browser/session.py +135 -0
  13. pcli/cli.py +757 -0
  14. pcli/config/__init__.py +0 -0
  15. pcli/config/paths.py +95 -0
  16. pcli/config/settings.py +435 -0
  17. pcli/cost/__init__.py +0 -0
  18. pcli/cost/context.py +275 -0
  19. pcli/cost/context_detect.py +183 -0
  20. pcli/cost/pricing_table.py +141 -0
  21. pcli/cost/tracker.py +126 -0
  22. pcli/llm/__init__.py +0 -0
  23. pcli/llm/client.py +285 -0
  24. pcli/llm/errors.py +37 -0
  25. pcli/llm/models.py +100 -0
  26. pcli/llm/streaming.py +108 -0
  27. pcli/memory/__init__.py +0 -0
  28. pcli/memory/extraction.py +106 -0
  29. pcli/memory/models.py +103 -0
  30. pcli/memory/store.py +88 -0
  31. pcli/permissions/__init__.py +0 -0
  32. pcli/permissions/guardrails.py +219 -0
  33. pcli/permissions/manager.py +215 -0
  34. pcli/permissions/policy.py +70 -0
  35. pcli/sandbox/__init__.py +0 -0
  36. pcli/sandbox/base.py +50 -0
  37. pcli/sandbox/docker_backend.py +107 -0
  38. pcli/sandbox/limits.py +63 -0
  39. pcli/sandbox/null_backend.py +92 -0
  40. pcli/sandbox/selector.py +75 -0
  41. pcli/sandbox/subprocess_backend.py +376 -0
  42. pcli/scheduler/__init__.py +0 -0
  43. pcli/scheduler/daemon.py +194 -0
  44. pcli/scheduler/models.py +97 -0
  45. pcli/scheduler/runner.py +84 -0
  46. pcli/scheduler/store.py +75 -0
  47. pcli/scheduler/triggers.py +84 -0
  48. pcli/session/__init__.py +0 -0
  49. pcli/session/audit.py +122 -0
  50. pcli/session/directory_check.py +28 -0
  51. pcli/session/export.py +57 -0
  52. pcli/session/importer.py +92 -0
  53. pcli/session/models.py +168 -0
  54. pcli/session/store.py +127 -0
  55. pcli/telegram/__init__.py +0 -0
  56. pcli/telegram/bot.py +266 -0
  57. pcli/telegram/daemon.py +1197 -0
  58. pcli/telegram/permissions.py +131 -0
  59. pcli/telegram/sender.py +58 -0
  60. pcli/tools/__init__.py +0 -0
  61. pcli/tools/_nested_agent.py +204 -0
  62. pcli/tools/agent_tools.py +264 -0
  63. pcli/tools/agent_tools_store.py +69 -0
  64. pcli/tools/artifacts.py +47 -0
  65. pcli/tools/base.py +185 -0
  66. pcli/tools/builtin/__init__.py +0 -0
  67. pcli/tools/builtin/agent_tool_register_tool.py +100 -0
  68. pcli/tools/builtin/artifact_tool.py +212 -0
  69. pcli/tools/builtin/ask_tool.py +77 -0
  70. pcli/tools/builtin/browser_tool.py +253 -0
  71. pcli/tools/builtin/decision_tool.py +73 -0
  72. pcli/tools/builtin/describe_tool.py +389 -0
  73. pcli/tools/builtin/diff_tools.py +225 -0
  74. pcli/tools/builtin/fs_tools.py +371 -0
  75. pcli/tools/builtin/grep_tool.py +88 -0
  76. pcli/tools/builtin/memory_tool.py +108 -0
  77. pcli/tools/builtin/network_tools.py +107 -0
  78. pcli/tools/builtin/pip_tool.py +106 -0
  79. pcli/tools/builtin/shell_tool.py +240 -0
  80. pcli/tools/builtin/subagent_tool.py +146 -0
  81. pcli/tools/builtin/todo_tool.py +122 -0
  82. pcli/tools/builtin/toolbox_register_tool.py +76 -0
  83. pcli/tools/builtin/web_tools.py +322 -0
  84. pcli/tools/pydiscovery/__init__.py +0 -0
  85. pcli/tools/pydiscovery/cache.py +51 -0
  86. pcli/tools/pydiscovery/index.py +48 -0
  87. pcli/tools/pydiscovery/invoke.py +181 -0
  88. pcli/tools/pydiscovery/search.py +117 -0
  89. pcli/tools/registry.py +138 -0
  90. pcli/tools/toolbox/__init__.py +0 -0
  91. pcli/tools/toolbox/introspect.py +48 -0
  92. pcli/tools/toolbox/manager.py +336 -0
  93. pcli/tools/toolbox/plugin_base.py +51 -0
  94. pcli/tools/toolbox/plugins/__init__.py +6 -0
  95. pcli/tools/toolbox/plugins/httpd.py +99 -0
  96. pcli/tools/toolbox/plugins/kafka.py +162 -0
  97. pcli/tools/toolbox/plugins/kubectl.py +211 -0
  98. pcli/tools/toolbox/plugins/sge.py +146 -0
  99. pcli/tools/toolbox/store.py +65 -0
  100. pcli/tools/toolbox/synthesize.py +100 -0
  101. pcli/tui/__init__.py +0 -0
  102. pcli/tui/app.py +37 -0
  103. pcli/tui/screens/__init__.py +0 -0
  104. pcli/tui/screens/ask_question_modal.py +54 -0
  105. pcli/tui/screens/chat.py +2070 -0
  106. pcli/tui/screens/confirm_modal.py +39 -0
  107. pcli/tui/screens/models.py +43 -0
  108. pcli/tui/screens/permission_modal.py +71 -0
  109. pcli/tui/screens/sessions.py +162 -0
  110. pcli/tui/screens/subagent_activity_modal.py +71 -0
  111. pcli/tui/shell_passthrough.py +56 -0
  112. pcli/tui/styles/pcli.tcss +241 -0
  113. pcli/tui/themes.py +84 -0
  114. pcli/tui/widgets/__init__.py +0 -0
  115. pcli/tui/widgets/chat_input.py +240 -0
  116. pcli/tui/widgets/command_suggestions.py +33 -0
  117. pcli/tui/widgets/message_view.py +328 -0
  118. pcli/tui/widgets/paste_input.py +99 -0
  119. pcli/tui/widgets/paste_marker.py +69 -0
  120. pcli/tui/widgets/status_bar.py +133 -0
  121. pcli/tui/widgets/status_pane.py +58 -0
  122. pcli/util/__init__.py +0 -0
  123. pcli/util/ids.py +15 -0
  124. pcli/util/logging.py +18 -0
  125. pcli/util/text.py +10 -0
  126. pcli_agent-0.1.0.dist-info/METADATA +259 -0
  127. pcli_agent-0.1.0.dist-info/RECORD +130 -0
  128. pcli_agent-0.1.0.dist-info/WHEEL +4 -0
  129. pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  130. pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,76 @@
1
+ """register_toolbox_tool: lets the LLM itself register a callable tool via
2
+ the toolbox mechanism (tools/toolbox/manager.py) instead of that being a
3
+ human-only, slash-command-triggered action. Aimed at the case where the
4
+ model has just written its own small reusable script (e.g. for something it
5
+ expects to repeat) and wants a real tool wrapping it, rather than re-deriving
6
+ the same shell incantation via run_shell every time.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from pcli.tools.base import ToolContext, ToolResult, ToolSpec
12
+ from pcli.tools.toolbox.manager import ToolboxDiscoveryError
13
+
14
+
15
+ async def _register_toolbox_tool(arguments: dict, ctx: ToolContext) -> ToolResult:
16
+ if ctx.toolbox_manager is None:
17
+ return ToolResult(output="Toolbox isn't available.", is_error=True)
18
+
19
+ name = arguments["name"]
20
+ path = arguments.get("path")
21
+ try:
22
+ summary = await ctx.toolbox_manager.discover(
23
+ name, gateway_client=ctx.gateway_client, model=ctx.model, path=path
24
+ )
25
+ except ToolboxDiscoveryError as exc:
26
+ message = str(exc)
27
+ if path is not None and ("doesn't exist" in message or "was not found" in message):
28
+ suggestion = "write_file the script first, or double-check the path is correct."
29
+ elif path is None:
30
+ suggestion = (
31
+ "if this is your own script rather than an installed CLI, pass path= instead "
32
+ "of relying on PATH lookup."
33
+ )
34
+ else:
35
+ suggestion = "make sure the script has a working --help that prints usage text."
36
+ return ToolResult(output=f"{message}\n[pcli] Suggestion: {suggestion}", is_error=True)
37
+
38
+ if ctx.tool_registry is not None:
39
+ loaded = await ctx.toolbox_manager.load_all()
40
+ ctx.tool_registry.merge(loaded)
41
+
42
+ return ToolResult(output=summary)
43
+
44
+
45
+ REGISTER_TOOLBOX_TOOL = ToolSpec(
46
+ name="register_toolbox_tool",
47
+ description="Registers a new callable tool from a script's --help output — either an "
48
+ "already-installed CLI on PATH (name only), or your own self-authored script (name + "
49
+ "path, e.g. one you just wrote with write_file and a working --help/argparse). Useful "
50
+ "when a task needs the same multi-step shell incantation repeatedly: write a small "
51
+ "script once, register it here, and call the resulting tool directly next time instead "
52
+ "of re-deriving the shell command. This is a judgment call for genuinely repetitive "
53
+ "work, not every one-off command — and it's permission-gated, so it isn't free.",
54
+ parameters={
55
+ "type": "object",
56
+ "properties": {
57
+ "name": {
58
+ "type": "string",
59
+ "description": "Short identifier for the new tool group (used as a prefix, "
60
+ "e.g. 'name_subcommand').",
61
+ },
62
+ "path": {
63
+ "type": "string",
64
+ "description": "Path to a self-authored script (relative to the working "
65
+ "directory, or absolute within it). Omit to look the name up on PATH "
66
+ "instead (for an already-installed CLI).",
67
+ },
68
+ },
69
+ "required": ["name"],
70
+ },
71
+ handler=_register_toolbox_tool,
72
+ needs_permission=True,
73
+ risk_description="Registers a new tool the model can call in later turns — a bigger "
74
+ "action than running one command, since it grants standing execution rights.",
75
+ read_only=False,
76
+ )
@@ -0,0 +1,322 @@
1
+ """web_fetch/web_search: pcli's only tools with real internet access (every
2
+ other tool is local — filesystem, shell, a configured LLM gateway). Built on
3
+ httpx (already a pcli dependency) plus stdlib html.parser, no new
4
+ dependencies.
5
+
6
+ web_search has no built-in, keyless "the" web search API — there isn't one.
7
+ If Settings.brave_search_api_key is configured, it's used (a real, supported
8
+ API). Otherwise this falls back to scraping DuckDuckGo's server-rendered
9
+ HTML results page (https://html.duckduckgo.com/html/), which needs no API
10
+ key and works out of the box, but is inherently best-effort: it depends on
11
+ DuckDuckGo's current HTML structure and a browser-like User-Agent (confirmed
12
+ necessary — httpx's default UA gets a soft-blocked empty response), and
13
+ could break if either changes. Treat it as a reasonable default, not a
14
+ guarantee — recommend brave_search_api_key for anything load-bearing.
15
+
16
+ Also confirmed directly against this endpoint: search operators
17
+ (site:/filetype:/-exclusion/OR/intitle:) don't work here — a query using
18
+ any of them comes back as a soft-blocked empty response (HTTP 202, no
19
+ results), not just a degraded/literal-text match. Only plain keywords and
20
+ quoted exact phrases reliably return results. This is why WEB_SEARCH's own
21
+ tool description steers the model away from operators rather than assuming
22
+ they're safe to try.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ from html.parser import HTMLParser
28
+
29
+ import httpx
30
+
31
+ from pcli.tools.base import ToolContext, ToolResult, ToolSpec
32
+
33
+ _DEFAULT_TIMEOUT_S = 20.0
34
+ _MAX_FETCH_BYTES = 5_000_000 # safety cap on what we'll read into memory
35
+ _DEFAULT_MAX_CHARS = 8_000
36
+ _DEFAULT_MAX_RESULTS = 5
37
+
38
+ # DuckDuckGo's HTML endpoint soft-blocks (returns the homepage, not results)
39
+ # for a generic/missing User-Agent - confirmed directly, not guessed.
40
+ _BROWSER_USER_AGENT = (
41
+ "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
42
+ "(KHTML, like Gecko) Chrome/122.0 Safari/537.36"
43
+ )
44
+
45
+
46
+ class _TextExtractor(HTMLParser):
47
+ """Best-effort HTML->text: drops script/style content, inserts a
48
+ newline at block-level element boundaries, keeps everything else.
49
+ Doesn't attempt readability-style boilerplate removal (nav/footer/ads
50
+ stay in) - good enough for "read what this page says", not a
51
+ replacement for a real readability extractor."""
52
+
53
+ _BLOCK_TAGS = frozenset(
54
+ {"p", "br", "div", "li", "tr", "h1", "h2", "h3", "h4", "h5", "h6", "section", "article"}
55
+ )
56
+ _SKIP_TAGS = frozenset({"script", "style", "noscript"})
57
+
58
+ def __init__(self) -> None:
59
+ super().__init__()
60
+ self._parts: list[str] = []
61
+ self._skip_depth = 0
62
+
63
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
64
+ if tag in self._SKIP_TAGS:
65
+ self._skip_depth += 1
66
+ elif tag in self._BLOCK_TAGS:
67
+ self._parts.append("\n")
68
+
69
+ def handle_endtag(self, tag: str) -> None:
70
+ if tag in self._SKIP_TAGS and self._skip_depth > 0:
71
+ self._skip_depth -= 1
72
+
73
+ def handle_data(self, data: str) -> None:
74
+ if self._skip_depth == 0:
75
+ self._parts.append(data)
76
+
77
+ def get_text(self) -> str:
78
+ lines = [line.strip() for line in "".join(self._parts).splitlines()]
79
+ return "\n".join(line for line in lines if line)
80
+
81
+
82
+ def _html_to_text(html: str) -> str:
83
+ extractor = _TextExtractor()
84
+ extractor.feed(html)
85
+ return extractor.get_text()
86
+
87
+
88
+ async def _web_fetch(arguments: dict, ctx: ToolContext) -> ToolResult:
89
+ url = arguments["url"]
90
+ max_chars = int(arguments.get("max_chars", _DEFAULT_MAX_CHARS))
91
+
92
+ try:
93
+ async with httpx.AsyncClient(follow_redirects=True, timeout=_DEFAULT_TIMEOUT_S) as client:
94
+ response = await client.get(url)
95
+ except httpx.TimeoutException:
96
+ return ToolResult(
97
+ output=f"Fetch timed out after {_DEFAULT_TIMEOUT_S:g}s for {url}\n"
98
+ "[pcli] Suggestion: the site may be slow or unreachable from here - try again, or "
99
+ "skip it and use another source.",
100
+ is_error=True,
101
+ )
102
+ except httpx.HTTPError as exc:
103
+ return ToolResult(
104
+ output=f"Fetch failed: {exc}\n"
105
+ "[pcli] Suggestion: check the URL is correct and reachable, and that network access "
106
+ "is actually available from this environment.",
107
+ is_error=True,
108
+ )
109
+
110
+ if response.status_code >= 400:
111
+ return ToolResult(
112
+ output=f"Fetch failed: HTTP {response.status_code} for {url}\n"
113
+ "[pcli] Suggestion: double-check the URL is correct and still live - a 404/403 won't "
114
+ "resolve by retrying the identical URL.",
115
+ is_error=True,
116
+ )
117
+
118
+ content_type = response.headers.get("content-type", "")
119
+ raw_bytes = response.content
120
+ if len(raw_bytes) > _MAX_FETCH_BYTES:
121
+ return ToolResult(
122
+ output=f"Response exceeded the {_MAX_FETCH_BYTES:,}-byte safety cap for {url} "
123
+ f"(content-type: {content_type or 'unknown'}).\n"
124
+ "[pcli] Suggestion: this looks like a large binary/media file, not a page to read - "
125
+ "use download_file instead if you actually need the raw content on disk.",
126
+ is_error=True,
127
+ )
128
+
129
+ text = raw_bytes.decode(errors="replace")
130
+ if "html" in content_type:
131
+ text = _html_to_text(text)
132
+ elif not content_type.startswith(("text/", "application/json")) and content_type:
133
+ return ToolResult(
134
+ output=f"{url} is {content_type}, not text/HTML ({len(raw_bytes):,} bytes) - not "
135
+ "something meaningful to read as text.\n"
136
+ "[pcli] Suggestion: use download_file instead if you need this content on disk.",
137
+ is_error=True,
138
+ )
139
+
140
+ if len(text) > max_chars:
141
+ text = text[:max_chars] + f"\n[...truncated to {max_chars:,} chars...]"
142
+ return ToolResult(output=text or "(empty response)")
143
+
144
+
145
+ WEB_FETCH = ToolSpec(
146
+ name="web_fetch",
147
+ description="Fetch a URL and return its readable text content (HTML is converted to plain "
148
+ "text; JSON/plain text is returned as-is). Use this to actually read the content of a page "
149
+ "you already have the URL for — e.g. one found via web_search, or one the user gave you. "
150
+ "Not for downloading files to disk (use download_file for that).",
151
+ parameters={
152
+ "type": "object",
153
+ "properties": {
154
+ "url": {"type": "string", "description": "URL to fetch."},
155
+ "max_chars": {
156
+ "type": "integer",
157
+ "description": f"Maximum characters of text to return (default "
158
+ f"{_DEFAULT_MAX_CHARS:,}).",
159
+ },
160
+ },
161
+ "required": ["url"],
162
+ },
163
+ handler=_web_fetch,
164
+ needs_permission=True,
165
+ risk_description="Fetches content from a URL over the network.",
166
+ plan_mode_safe=True,
167
+ read_only=True,
168
+ )
169
+
170
+
171
+ class _DuckDuckGoResultParser(HTMLParser):
172
+ """Parses DuckDuckGo's html.duckduckgo.com/html/ results page. Organic
173
+ results are `<div class="... web-result ...">` (ads use `result--ad`
174
+ instead and are skipped); within one, `<a class="result__a">` carries
175
+ the title+URL and `<a class="result__snippet">` the snippet - confirmed
176
+ directly against a real response, not guessed from memory."""
177
+
178
+ def __init__(self) -> None:
179
+ super().__init__()
180
+ self.results: list[dict[str, str]] = []
181
+ self._depth = 0
182
+ self._result_div_depth: int | None = None
183
+ self._current: dict[str, str] = {}
184
+ self._capturing: str | None = None # "title" | "snippet" | None
185
+
186
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
187
+ attrs_dict = dict(attrs)
188
+ if tag == "div":
189
+ self._depth += 1
190
+ classes = (attrs_dict.get("class") or "").split()
191
+ if self._result_div_depth is None and "web-result" in classes:
192
+ self._result_div_depth = self._depth
193
+ self._current = {"title": "", "url": "", "snippet": ""}
194
+ elif tag == "a" and self._result_div_depth is not None:
195
+ classes = (attrs_dict.get("class") or "").split()
196
+ if "result__a" in classes:
197
+ self._capturing = "title"
198
+ self._current["url"] = attrs_dict.get("href") or ""
199
+ elif "result__snippet" in classes:
200
+ self._capturing = "snippet"
201
+
202
+ def handle_endtag(self, tag: str) -> None:
203
+ if tag == "a":
204
+ self._capturing = None
205
+ elif tag == "div":
206
+ if self._result_div_depth == self._depth:
207
+ if self._current.get("title") and self._current.get("url"):
208
+ self.results.append(
209
+ {
210
+ "title": self._current["title"].strip(),
211
+ "url": self._current["url"],
212
+ "snippet": self._current["snippet"].strip(),
213
+ }
214
+ )
215
+ self._result_div_depth = None
216
+ self._depth -= 1
217
+
218
+ def handle_data(self, data: str) -> None:
219
+ if self._capturing == "title":
220
+ self._current["title"] = self._current["title"] + data
221
+ elif self._capturing == "snippet":
222
+ self._current["snippet"] = self._current["snippet"] + data
223
+
224
+
225
+ async def _search_duckduckgo(query: str, max_results: int) -> list[dict[str, str]]:
226
+ async with httpx.AsyncClient(timeout=_DEFAULT_TIMEOUT_S) as client:
227
+ response = await client.post(
228
+ "https://html.duckduckgo.com/html/",
229
+ data={"q": query},
230
+ headers={"User-Agent": _BROWSER_USER_AGENT},
231
+ )
232
+ response.raise_for_status()
233
+ parser = _DuckDuckGoResultParser()
234
+ parser.feed(response.text)
235
+ return parser.results[:max_results]
236
+
237
+
238
+ async def _search_brave(query: str, max_results: int, api_key: str) -> list[dict[str, str]]:
239
+ async with httpx.AsyncClient(timeout=_DEFAULT_TIMEOUT_S) as client:
240
+ response = await client.get(
241
+ "https://api.search.brave.com/res/v1/web/search",
242
+ params={"q": query, "count": max_results},
243
+ headers={"Accept": "application/json", "X-Subscription-Token": api_key},
244
+ )
245
+ response.raise_for_status()
246
+ data = response.json()
247
+ results = data.get("web", {}).get("results", [])
248
+ return [
249
+ {"title": r.get("title", ""), "url": r.get("url", ""), "snippet": r.get("description", "")}
250
+ for r in results[:max_results]
251
+ ]
252
+
253
+
254
+ def _format_results(results: list[dict[str, str]]) -> str:
255
+ if not results:
256
+ return "No results."
257
+ lines = []
258
+ for index, result in enumerate(results, start=1):
259
+ lines.append(f"{index}. {result['title']}\n {result['url']}\n {result['snippet']}")
260
+ return "\n\n".join(lines)
261
+
262
+
263
+ async def _web_search(arguments: dict, ctx: ToolContext) -> ToolResult:
264
+ query = arguments["query"]
265
+ max_results = int(arguments.get("max_results", _DEFAULT_MAX_RESULTS))
266
+
267
+ try:
268
+ if ctx.brave_search_api_key:
269
+ results = await _search_brave(query, max_results, ctx.brave_search_api_key)
270
+ else:
271
+ results = await _search_duckduckgo(query, max_results)
272
+ except httpx.TimeoutException:
273
+ return ToolResult(
274
+ output=f"Search timed out after {_DEFAULT_TIMEOUT_S:g}s for query: {query}\n"
275
+ "[pcli] Suggestion: try again, or narrow the query.",
276
+ is_error=True,
277
+ )
278
+ except httpx.HTTPError as exc:
279
+ return ToolResult(
280
+ output=f"Search failed: {exc}\n"
281
+ "[pcli] Suggestion: check network access is available from this environment. If "
282
+ "this keeps failing, the no-API-key DuckDuckGo fallback may be getting blocked - "
283
+ "configure brave_search_api_key for a real, supported search API instead.",
284
+ is_error=True,
285
+ )
286
+
287
+ return ToolResult(output=_format_results(results))
288
+
289
+
290
+ WEB_SEARCH = ToolSpec(
291
+ name="web_search",
292
+ description="Search the web and return a short list of results (title, URL, snippet). Use "
293
+ "web_fetch on a promising result's URL to read its actual content — search results alone "
294
+ "are rarely enough to answer from. Uses a configured search API if available, otherwise a "
295
+ "best-effort fallback with no setup required.\n\n"
296
+ "Query tips: use short, keyword-based queries (roughly 2-6 words), not a full question or "
297
+ "sentence — 'python asyncio cancel task' beats 'how do I cancel a task in python asyncio'. "
298
+ "Quote an exact phrase you need verbatim (an error message, a function name, a title) — "
299
+ "quoting is reliable on both backends. Avoid site:/filetype:/-exclusion/OR operators: they "
300
+ "only work with a configured search API (Brave) and are unreliable — confirmed to return "
301
+ "nothing, not just fewer results — on the no-setup DuckDuckGo fallback used by default. If "
302
+ "you need a specific site, search normally and pick the matching result, or web_fetch a "
303
+ "known URL directly instead. Start broad and re-search with narrower keywords based on what "
304
+ "the first results show, rather than building one heavily-qualified query upfront.",
305
+ parameters={
306
+ "type": "object",
307
+ "properties": {
308
+ "query": {"type": "string", "description": "Search query."},
309
+ "max_results": {
310
+ "type": "integer",
311
+ "description": f"Maximum number of results to return (default "
312
+ f"{_DEFAULT_MAX_RESULTS}).",
313
+ },
314
+ },
315
+ "required": ["query"],
316
+ },
317
+ handler=_web_search,
318
+ needs_permission=True,
319
+ risk_description="Sends a search query to a third-party service over the network.",
320
+ plan_mode_safe=True,
321
+ read_only=True,
322
+ )
File without changes
@@ -0,0 +1,51 @@
1
+ """Disk cache for the pydiscovery index, invalidated by an environment
2
+ fingerprint (interpreter path + installed distribution name/version pairs)
3
+ so it rebuilds automatically when the venv changes, but not on every launch."""
4
+
5
+ from __future__ import annotations
6
+
7
+ import hashlib
8
+ import json
9
+ import sys
10
+ from dataclasses import asdict
11
+ from importlib import metadata as importlib_metadata
12
+ from pathlib import Path
13
+
14
+ from pcli.config.paths import cache_dir
15
+ from pcli.tools.pydiscovery.index import ModuleEntry, build_index
16
+
17
+
18
+ def environment_fingerprint() -> str:
19
+ dist_pairs = sorted(
20
+ f"{(dist.metadata.get('Name') or dist.metadata.get('name') or '?')}=={dist.version}"
21
+ for dist in importlib_metadata.distributions()
22
+ )
23
+ raw = "\n".join([sys.executable, sys.version, *dist_pairs])
24
+ return hashlib.sha256(raw.encode()).hexdigest()[:16]
25
+
26
+
27
+ def index_cache_path() -> Path:
28
+ return cache_dir() / "pydiscovery_index.json"
29
+
30
+
31
+ def load_or_build_index(*, force_rebuild: bool = False) -> list[ModuleEntry]:
32
+ fingerprint = environment_fingerprint()
33
+ path = index_cache_path()
34
+
35
+ if not force_rebuild and path.exists():
36
+ try:
37
+ data = json.loads(path.read_text(encoding="utf-8"))
38
+ if data.get("fingerprint") == fingerprint:
39
+ return [ModuleEntry(**entry) for entry in data["entries"]]
40
+ except (json.JSONDecodeError, KeyError, TypeError):
41
+ pass # corrupt/stale cache -> fall through and rebuild
42
+
43
+ entries = build_index()
44
+ path.parent.mkdir(parents=True, exist_ok=True)
45
+ path.write_text(
46
+ json.dumps(
47
+ {"fingerprint": fingerprint, "entries": [asdict(e) for e in entries]}, indent=2
48
+ ),
49
+ encoding="utf-8",
50
+ )
51
+ return entries
@@ -0,0 +1,48 @@
1
+ """Enumerates importable modules/packages by NAME only — via pkgutil and
2
+ importlib.metadata, neither of which imports the module bodies. This is the
3
+ "tier 1" index: cheap, safe, always available, good for coarse search.
4
+ Docstrings/signatures ("tier 2") require actually importing a specific
5
+ module, done on demand by inspect_python_module (see search.py)."""
6
+
7
+ from __future__ import annotations
8
+
9
+ import pkgutil
10
+ import sys
11
+ from dataclasses import dataclass
12
+ from importlib import metadata as importlib_metadata
13
+
14
+
15
+ @dataclass
16
+ class ModuleEntry:
17
+ name: str
18
+ package: str | None
19
+ summary: str
20
+
21
+
22
+ def build_index() -> list[ModuleEntry]:
23
+ entries: dict[str, ModuleEntry] = {}
24
+
25
+ package_summaries: dict[str, str] = {}
26
+ for dist in importlib_metadata.distributions():
27
+ name = dist.metadata.get("Name") or dist.metadata.get("name")
28
+ if not name:
29
+ continue
30
+ summary = dist.metadata.get("Summary") or ""
31
+ package_summaries[name.lower().replace("-", "_")] = summary
32
+ entries.setdefault(name, ModuleEntry(name=name, package=name, summary=summary))
33
+
34
+ for name in sorted(getattr(sys, "stdlib_module_names", ())):
35
+ if name.startswith("_"):
36
+ continue
37
+ entries.setdefault(
38
+ name, ModuleEntry(name=name, package=None, summary="(Python standard library)")
39
+ )
40
+
41
+ for module_info in pkgutil.iter_modules():
42
+ name = module_info.name
43
+ if name.startswith("_"):
44
+ continue
45
+ summary = package_summaries.get(name.lower().replace("-", "_"), "")
46
+ entries.setdefault(name, ModuleEntry(name=name, package=None, summary=summary))
47
+
48
+ return sorted(entries.values(), key=lambda e: e.name)
@@ -0,0 +1,181 @@
1
+ """Runs arbitrary Python (module introspection or a function call) in an
2
+ isolated subprocess, never in the main pcli process.
3
+
4
+ Deliberately does NOT go through ctx.sandbox: that may be a Docker container
5
+ running a bare image with none of the host's installed packages, whereas
6
+ pydiscovery's whole point is to reach packages installed in the *host*
7
+ environment. So this always spawns the host interpreter (sys.executable)
8
+ through a dedicated RestrictedSubprocessSandbox instead — real process
9
+ isolation (scrubbed env, timeout, resource limits) without losing access to
10
+ the packages being introspected.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import sys
17
+
18
+ from pcli.sandbox.base import ExecRequest
19
+ from pcli.sandbox.subprocess_backend import RestrictedSubprocessSandbox
20
+ from pcli.tools.base import ToolContext, ToolResult, ToolSpec
21
+
22
+ _BOOTSTRAP_SCRIPT = r"""
23
+ import importlib, inspect, json, sys
24
+
25
+ def _resolve(qualified_name):
26
+ parts = qualified_name.split(".")
27
+ module = None
28
+ remaining = []
29
+ for i in range(len(parts), 0, -1):
30
+ candidate = ".".join(parts[:i])
31
+ try:
32
+ module = importlib.import_module(candidate)
33
+ remaining = parts[i:]
34
+ break
35
+ except ImportError:
36
+ continue
37
+ if module is None:
38
+ raise ImportError("Could not import any prefix of %r" % (qualified_name,))
39
+ obj = module
40
+ for attr in remaining:
41
+ obj = getattr(obj, attr)
42
+ return obj
43
+
44
+ def _json_safe(value):
45
+ try:
46
+ json.dumps(value)
47
+ return value
48
+ except TypeError:
49
+ return {"__repr__": repr(value)[:2000], "__type__": type(value).__name__}
50
+
51
+ def main():
52
+ payload = json.loads(sys.stdin.read())
53
+ mode = payload["mode"]
54
+ if mode == "inspect":
55
+ module = importlib.import_module(payload["module"])
56
+ query = (payload.get("query") or "").lower()
57
+ members = []
58
+ for name, obj in inspect.getmembers(module):
59
+ if name.startswith("_"):
60
+ continue
61
+ if query and query not in name.lower():
62
+ continue
63
+ if not (inspect.isfunction(obj) or inspect.isclass(obj)
64
+ or inspect.isbuiltin(obj) or inspect.ismethod(obj)):
65
+ continue
66
+ try:
67
+ sig = str(inspect.signature(obj))
68
+ except (ValueError, TypeError):
69
+ sig = "(...)"
70
+ doc_lines = (inspect.getdoc(obj) or "").strip().splitlines()
71
+ members.append({
72
+ "name": name,
73
+ "kind": "class" if inspect.isclass(obj) else "function",
74
+ "signature": sig,
75
+ "doc": doc_lines[0] if doc_lines else "",
76
+ })
77
+ if len(members) >= 200:
78
+ break
79
+ print(json.dumps({"ok": True, "members": members}))
80
+ elif mode == "call":
81
+ obj = _resolve(payload["qualified_name"])
82
+ if not callable(obj):
83
+ print(json.dumps({"ok": False, "error": "%r is not callable" % (payload["qualified_name"],)}))
84
+ return
85
+ result = obj(*(payload.get("args") or []), **(payload.get("kwargs") or {}))
86
+ print(json.dumps({"ok": True, "result": _json_safe(result)}))
87
+ else:
88
+ print(json.dumps({"ok": False, "error": "unknown mode %r" % (mode,)}))
89
+
90
+ try:
91
+ main()
92
+ except Exception as exc:
93
+ print(json.dumps({"ok": False, "error": "%s: %s" % (type(exc).__name__, exc)}))
94
+ """
95
+
96
+
97
+ async def run_bootstrap(ctx: ToolContext, payload: dict, *, timeout_s: float = 30.0) -> dict:
98
+ sandbox = RestrictedSubprocessSandbox(allowed_roots=[ctx.cwd])
99
+ request = ExecRequest(
100
+ command=[sys.executable, "-c", _BOOTSTRAP_SCRIPT],
101
+ cwd=ctx.cwd,
102
+ stdin=json.dumps(payload),
103
+ timeout_s=timeout_s,
104
+ )
105
+ result = await sandbox.execute(request)
106
+ if result.timed_out:
107
+ raise RuntimeError("timed out")
108
+ output_line = result.stdout.strip().splitlines()[-1] if result.stdout.strip() else ""
109
+ if not output_line:
110
+ raise RuntimeError(result.stderr.strip() or f"no output (exit_code={result.exit_code})")
111
+ try:
112
+ return json.loads(output_line)
113
+ except json.JSONDecodeError as exc:
114
+ raise RuntimeError(f"could not parse bootstrap output: {result.stdout!r}") from exc
115
+
116
+
117
+ async def _call_python(arguments: dict, ctx: ToolContext) -> ToolResult:
118
+ qualified_name = arguments["qualified_name"]
119
+ payload = {
120
+ "mode": "call",
121
+ "qualified_name": qualified_name,
122
+ "args": arguments.get("args") or [],
123
+ "kwargs": arguments.get("kwargs") or {},
124
+ }
125
+ try:
126
+ data = await run_bootstrap(ctx, payload)
127
+ except RuntimeError as exc:
128
+ return ToolResult(
129
+ output=f"Call to '{qualified_name}' failed: {exc}\n"
130
+ "[pcli] Suggestion: use inspect_python_module to confirm the callable actually "
131
+ "exists under that name before retrying.",
132
+ is_error=True,
133
+ )
134
+
135
+ if not data.get("ok"):
136
+ error = data.get("error", "unknown error")
137
+ if "ImportError" in error or "ModuleNotFoundError" in error:
138
+ suggestion = "run search_python first to confirm the exact module name is installed."
139
+ elif "is not callable" in error:
140
+ suggestion = "use inspect_python_module to see what's actually callable there."
141
+ else:
142
+ suggestion = (
143
+ "double-check the argument names/types against inspect_python_module's "
144
+ "signature output."
145
+ )
146
+ return ToolResult(output=f"{error}\n[pcli] Suggestion: {suggestion}", is_error=True)
147
+
148
+ result = data["result"]
149
+ if isinstance(result, dict) and set(result) == {"__repr__", "__type__"}:
150
+ text = f"<{result['__type__']}> {result['__repr__']}"
151
+ elif isinstance(result, str):
152
+ text = result
153
+ else:
154
+ text = json.dumps(result, indent=2)
155
+ return ToolResult(output=text)
156
+
157
+
158
+ CALL_PYTHON = ToolSpec(
159
+ name="call_python",
160
+ description="Call a Python function by its fully-qualified name (e.g. 'math.sqrt', "
161
+ "'json.dumps'), with JSON-serializable positional/keyword arguments. Runs in an "
162
+ "isolated subprocess with the host's installed packages, not the main pcli process.",
163
+ parameters={
164
+ "type": "object",
165
+ "properties": {
166
+ "qualified_name": {
167
+ "type": "string",
168
+ "description": "Fully-qualified function/callable name, e.g. 'math.sqrt'.",
169
+ },
170
+ "args": {"type": "array", "description": "Positional arguments (JSON values)."},
171
+ "kwargs": {"type": "object", "description": "Keyword arguments (JSON values)."},
172
+ },
173
+ "required": ["qualified_name"],
174
+ },
175
+ handler=_call_python,
176
+ needs_permission=True,
177
+ needs_sandbox=True,
178
+ risk_description="Executes an arbitrary Python function call in a subprocess.",
179
+ guardrail_python_module_arg="qualified_name",
180
+ read_only=False,
181
+ )