pcli-agent 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pcli/__init__.py +1 -0
- pcli/__main__.py +4 -0
- pcli/agent/__init__.py +0 -0
- pcli/agent/activity.py +116 -0
- pcli/agent/compaction.py +205 -0
- pcli/agent/context_pruning.py +88 -0
- pcli/agent/headless.py +209 -0
- pcli/agent/loop.py +442 -0
- pcli/agent/prompt.py +371 -0
- pcli/agent/runtime.py +240 -0
- pcli/browser/__init__.py +0 -0
- pcli/browser/session.py +135 -0
- pcli/cli.py +757 -0
- pcli/config/__init__.py +0 -0
- pcli/config/paths.py +95 -0
- pcli/config/settings.py +435 -0
- pcli/cost/__init__.py +0 -0
- pcli/cost/context.py +275 -0
- pcli/cost/context_detect.py +183 -0
- pcli/cost/pricing_table.py +141 -0
- pcli/cost/tracker.py +126 -0
- pcli/llm/__init__.py +0 -0
- pcli/llm/client.py +285 -0
- pcli/llm/errors.py +37 -0
- pcli/llm/models.py +100 -0
- pcli/llm/streaming.py +108 -0
- pcli/memory/__init__.py +0 -0
- pcli/memory/extraction.py +106 -0
- pcli/memory/models.py +103 -0
- pcli/memory/store.py +88 -0
- pcli/permissions/__init__.py +0 -0
- pcli/permissions/guardrails.py +219 -0
- pcli/permissions/manager.py +215 -0
- pcli/permissions/policy.py +70 -0
- pcli/sandbox/__init__.py +0 -0
- pcli/sandbox/base.py +50 -0
- pcli/sandbox/docker_backend.py +107 -0
- pcli/sandbox/limits.py +63 -0
- pcli/sandbox/null_backend.py +92 -0
- pcli/sandbox/selector.py +75 -0
- pcli/sandbox/subprocess_backend.py +376 -0
- pcli/scheduler/__init__.py +0 -0
- pcli/scheduler/daemon.py +194 -0
- pcli/scheduler/models.py +97 -0
- pcli/scheduler/runner.py +84 -0
- pcli/scheduler/store.py +75 -0
- pcli/scheduler/triggers.py +84 -0
- pcli/session/__init__.py +0 -0
- pcli/session/audit.py +122 -0
- pcli/session/directory_check.py +28 -0
- pcli/session/export.py +57 -0
- pcli/session/importer.py +92 -0
- pcli/session/models.py +168 -0
- pcli/session/store.py +127 -0
- pcli/telegram/__init__.py +0 -0
- pcli/telegram/bot.py +266 -0
- pcli/telegram/daemon.py +1197 -0
- pcli/telegram/permissions.py +131 -0
- pcli/telegram/sender.py +58 -0
- pcli/tools/__init__.py +0 -0
- pcli/tools/_nested_agent.py +204 -0
- pcli/tools/agent_tools.py +264 -0
- pcli/tools/agent_tools_store.py +69 -0
- pcli/tools/artifacts.py +47 -0
- pcli/tools/base.py +185 -0
- pcli/tools/builtin/__init__.py +0 -0
- pcli/tools/builtin/agent_tool_register_tool.py +100 -0
- pcli/tools/builtin/artifact_tool.py +212 -0
- pcli/tools/builtin/ask_tool.py +77 -0
- pcli/tools/builtin/browser_tool.py +253 -0
- pcli/tools/builtin/decision_tool.py +73 -0
- pcli/tools/builtin/describe_tool.py +389 -0
- pcli/tools/builtin/diff_tools.py +225 -0
- pcli/tools/builtin/fs_tools.py +371 -0
- pcli/tools/builtin/grep_tool.py +88 -0
- pcli/tools/builtin/memory_tool.py +108 -0
- pcli/tools/builtin/network_tools.py +107 -0
- pcli/tools/builtin/pip_tool.py +106 -0
- pcli/tools/builtin/shell_tool.py +240 -0
- pcli/tools/builtin/subagent_tool.py +146 -0
- pcli/tools/builtin/todo_tool.py +122 -0
- pcli/tools/builtin/toolbox_register_tool.py +76 -0
- pcli/tools/builtin/web_tools.py +322 -0
- pcli/tools/pydiscovery/__init__.py +0 -0
- pcli/tools/pydiscovery/cache.py +51 -0
- pcli/tools/pydiscovery/index.py +48 -0
- pcli/tools/pydiscovery/invoke.py +181 -0
- pcli/tools/pydiscovery/search.py +117 -0
- pcli/tools/registry.py +138 -0
- pcli/tools/toolbox/__init__.py +0 -0
- pcli/tools/toolbox/introspect.py +48 -0
- pcli/tools/toolbox/manager.py +336 -0
- pcli/tools/toolbox/plugin_base.py +51 -0
- pcli/tools/toolbox/plugins/__init__.py +6 -0
- pcli/tools/toolbox/plugins/httpd.py +99 -0
- pcli/tools/toolbox/plugins/kafka.py +162 -0
- pcli/tools/toolbox/plugins/kubectl.py +211 -0
- pcli/tools/toolbox/plugins/sge.py +146 -0
- pcli/tools/toolbox/store.py +65 -0
- pcli/tools/toolbox/synthesize.py +100 -0
- pcli/tui/__init__.py +0 -0
- pcli/tui/app.py +37 -0
- pcli/tui/screens/__init__.py +0 -0
- pcli/tui/screens/ask_question_modal.py +54 -0
- pcli/tui/screens/chat.py +2070 -0
- pcli/tui/screens/confirm_modal.py +39 -0
- pcli/tui/screens/models.py +43 -0
- pcli/tui/screens/permission_modal.py +71 -0
- pcli/tui/screens/sessions.py +162 -0
- pcli/tui/screens/subagent_activity_modal.py +71 -0
- pcli/tui/shell_passthrough.py +56 -0
- pcli/tui/styles/pcli.tcss +241 -0
- pcli/tui/themes.py +84 -0
- pcli/tui/widgets/__init__.py +0 -0
- pcli/tui/widgets/chat_input.py +240 -0
- pcli/tui/widgets/command_suggestions.py +33 -0
- pcli/tui/widgets/message_view.py +328 -0
- pcli/tui/widgets/paste_input.py +99 -0
- pcli/tui/widgets/paste_marker.py +69 -0
- pcli/tui/widgets/status_bar.py +133 -0
- pcli/tui/widgets/status_pane.py +58 -0
- pcli/util/__init__.py +0 -0
- pcli/util/ids.py +15 -0
- pcli/util/logging.py +18 -0
- pcli/util/text.py +10 -0
- pcli_agent-0.1.0.dist-info/METADATA +259 -0
- pcli_agent-0.1.0.dist-info/RECORD +130 -0
- pcli_agent-0.1.0.dist-info/WHEEL +4 -0
- pcli_agent-0.1.0.dist-info/entry_points.txt +2 -0
- pcli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""register_toolbox_tool: lets the LLM itself register a callable tool via
|
|
2
|
+
the toolbox mechanism (tools/toolbox/manager.py) instead of that being a
|
|
3
|
+
human-only, slash-command-triggered action. Aimed at the case where the
|
|
4
|
+
model has just written its own small reusable script (e.g. for something it
|
|
5
|
+
expects to repeat) and wants a real tool wrapping it, rather than re-deriving
|
|
6
|
+
the same shell incantation via run_shell every time.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pcli.tools.base import ToolContext, ToolResult, ToolSpec
|
|
12
|
+
from pcli.tools.toolbox.manager import ToolboxDiscoveryError
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
async def _register_toolbox_tool(arguments: dict, ctx: ToolContext) -> ToolResult:
|
|
16
|
+
if ctx.toolbox_manager is None:
|
|
17
|
+
return ToolResult(output="Toolbox isn't available.", is_error=True)
|
|
18
|
+
|
|
19
|
+
name = arguments["name"]
|
|
20
|
+
path = arguments.get("path")
|
|
21
|
+
try:
|
|
22
|
+
summary = await ctx.toolbox_manager.discover(
|
|
23
|
+
name, gateway_client=ctx.gateway_client, model=ctx.model, path=path
|
|
24
|
+
)
|
|
25
|
+
except ToolboxDiscoveryError as exc:
|
|
26
|
+
message = str(exc)
|
|
27
|
+
if path is not None and ("doesn't exist" in message or "was not found" in message):
|
|
28
|
+
suggestion = "write_file the script first, or double-check the path is correct."
|
|
29
|
+
elif path is None:
|
|
30
|
+
suggestion = (
|
|
31
|
+
"if this is your own script rather than an installed CLI, pass path= instead "
|
|
32
|
+
"of relying on PATH lookup."
|
|
33
|
+
)
|
|
34
|
+
else:
|
|
35
|
+
suggestion = "make sure the script has a working --help that prints usage text."
|
|
36
|
+
return ToolResult(output=f"{message}\n[pcli] Suggestion: {suggestion}", is_error=True)
|
|
37
|
+
|
|
38
|
+
if ctx.tool_registry is not None:
|
|
39
|
+
loaded = await ctx.toolbox_manager.load_all()
|
|
40
|
+
ctx.tool_registry.merge(loaded)
|
|
41
|
+
|
|
42
|
+
return ToolResult(output=summary)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
REGISTER_TOOLBOX_TOOL = ToolSpec(
|
|
46
|
+
name="register_toolbox_tool",
|
|
47
|
+
description="Registers a new callable tool from a script's --help output — either an "
|
|
48
|
+
"already-installed CLI on PATH (name only), or your own self-authored script (name + "
|
|
49
|
+
"path, e.g. one you just wrote with write_file and a working --help/argparse). Useful "
|
|
50
|
+
"when a task needs the same multi-step shell incantation repeatedly: write a small "
|
|
51
|
+
"script once, register it here, and call the resulting tool directly next time instead "
|
|
52
|
+
"of re-deriving the shell command. This is a judgment call for genuinely repetitive "
|
|
53
|
+
"work, not every one-off command — and it's permission-gated, so it isn't free.",
|
|
54
|
+
parameters={
|
|
55
|
+
"type": "object",
|
|
56
|
+
"properties": {
|
|
57
|
+
"name": {
|
|
58
|
+
"type": "string",
|
|
59
|
+
"description": "Short identifier for the new tool group (used as a prefix, "
|
|
60
|
+
"e.g. 'name_subcommand').",
|
|
61
|
+
},
|
|
62
|
+
"path": {
|
|
63
|
+
"type": "string",
|
|
64
|
+
"description": "Path to a self-authored script (relative to the working "
|
|
65
|
+
"directory, or absolute within it). Omit to look the name up on PATH "
|
|
66
|
+
"instead (for an already-installed CLI).",
|
|
67
|
+
},
|
|
68
|
+
},
|
|
69
|
+
"required": ["name"],
|
|
70
|
+
},
|
|
71
|
+
handler=_register_toolbox_tool,
|
|
72
|
+
needs_permission=True,
|
|
73
|
+
risk_description="Registers a new tool the model can call in later turns — a bigger "
|
|
74
|
+
"action than running one command, since it grants standing execution rights.",
|
|
75
|
+
read_only=False,
|
|
76
|
+
)
|
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
"""web_fetch/web_search: pcli's only tools with real internet access (every
|
|
2
|
+
other tool is local — filesystem, shell, a configured LLM gateway). Built on
|
|
3
|
+
httpx (already a pcli dependency) plus stdlib html.parser, no new
|
|
4
|
+
dependencies.
|
|
5
|
+
|
|
6
|
+
web_search has no built-in, keyless "the" web search API — there isn't one.
|
|
7
|
+
If Settings.brave_search_api_key is configured, it's used (a real, supported
|
|
8
|
+
API). Otherwise this falls back to scraping DuckDuckGo's server-rendered
|
|
9
|
+
HTML results page (https://html.duckduckgo.com/html/), which needs no API
|
|
10
|
+
key and works out of the box, but is inherently best-effort: it depends on
|
|
11
|
+
DuckDuckGo's current HTML structure and a browser-like User-Agent (confirmed
|
|
12
|
+
necessary — httpx's default UA gets a soft-blocked empty response), and
|
|
13
|
+
could break if either changes. Treat it as a reasonable default, not a
|
|
14
|
+
guarantee — recommend brave_search_api_key for anything load-bearing.
|
|
15
|
+
|
|
16
|
+
Also confirmed directly against this endpoint: search operators
|
|
17
|
+
(site:/filetype:/-exclusion/OR/intitle:) don't work here — a query using
|
|
18
|
+
any of them comes back as a soft-blocked empty response (HTTP 202, no
|
|
19
|
+
results), not just a degraded/literal-text match. Only plain keywords and
|
|
20
|
+
quoted exact phrases reliably return results. This is why WEB_SEARCH's own
|
|
21
|
+
tool description steers the model away from operators rather than assuming
|
|
22
|
+
they're safe to try.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from html.parser import HTMLParser
|
|
28
|
+
|
|
29
|
+
import httpx
|
|
30
|
+
|
|
31
|
+
from pcli.tools.base import ToolContext, ToolResult, ToolSpec
|
|
32
|
+
|
|
33
|
+
_DEFAULT_TIMEOUT_S = 20.0
|
|
34
|
+
_MAX_FETCH_BYTES = 5_000_000 # safety cap on what we'll read into memory
|
|
35
|
+
_DEFAULT_MAX_CHARS = 8_000
|
|
36
|
+
_DEFAULT_MAX_RESULTS = 5
|
|
37
|
+
|
|
38
|
+
# DuckDuckGo's HTML endpoint soft-blocks (returns the homepage, not results)
|
|
39
|
+
# for a generic/missing User-Agent - confirmed directly, not guessed.
|
|
40
|
+
_BROWSER_USER_AGENT = (
|
|
41
|
+
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
|
|
42
|
+
"(KHTML, like Gecko) Chrome/122.0 Safari/537.36"
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class _TextExtractor(HTMLParser):
|
|
47
|
+
"""Best-effort HTML->text: drops script/style content, inserts a
|
|
48
|
+
newline at block-level element boundaries, keeps everything else.
|
|
49
|
+
Doesn't attempt readability-style boilerplate removal (nav/footer/ads
|
|
50
|
+
stay in) - good enough for "read what this page says", not a
|
|
51
|
+
replacement for a real readability extractor."""
|
|
52
|
+
|
|
53
|
+
_BLOCK_TAGS = frozenset(
|
|
54
|
+
{"p", "br", "div", "li", "tr", "h1", "h2", "h3", "h4", "h5", "h6", "section", "article"}
|
|
55
|
+
)
|
|
56
|
+
_SKIP_TAGS = frozenset({"script", "style", "noscript"})
|
|
57
|
+
|
|
58
|
+
def __init__(self) -> None:
|
|
59
|
+
super().__init__()
|
|
60
|
+
self._parts: list[str] = []
|
|
61
|
+
self._skip_depth = 0
|
|
62
|
+
|
|
63
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
64
|
+
if tag in self._SKIP_TAGS:
|
|
65
|
+
self._skip_depth += 1
|
|
66
|
+
elif tag in self._BLOCK_TAGS:
|
|
67
|
+
self._parts.append("\n")
|
|
68
|
+
|
|
69
|
+
def handle_endtag(self, tag: str) -> None:
|
|
70
|
+
if tag in self._SKIP_TAGS and self._skip_depth > 0:
|
|
71
|
+
self._skip_depth -= 1
|
|
72
|
+
|
|
73
|
+
def handle_data(self, data: str) -> None:
|
|
74
|
+
if self._skip_depth == 0:
|
|
75
|
+
self._parts.append(data)
|
|
76
|
+
|
|
77
|
+
def get_text(self) -> str:
|
|
78
|
+
lines = [line.strip() for line in "".join(self._parts).splitlines()]
|
|
79
|
+
return "\n".join(line for line in lines if line)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _html_to_text(html: str) -> str:
|
|
83
|
+
extractor = _TextExtractor()
|
|
84
|
+
extractor.feed(html)
|
|
85
|
+
return extractor.get_text()
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
async def _web_fetch(arguments: dict, ctx: ToolContext) -> ToolResult:
|
|
89
|
+
url = arguments["url"]
|
|
90
|
+
max_chars = int(arguments.get("max_chars", _DEFAULT_MAX_CHARS))
|
|
91
|
+
|
|
92
|
+
try:
|
|
93
|
+
async with httpx.AsyncClient(follow_redirects=True, timeout=_DEFAULT_TIMEOUT_S) as client:
|
|
94
|
+
response = await client.get(url)
|
|
95
|
+
except httpx.TimeoutException:
|
|
96
|
+
return ToolResult(
|
|
97
|
+
output=f"Fetch timed out after {_DEFAULT_TIMEOUT_S:g}s for {url}\n"
|
|
98
|
+
"[pcli] Suggestion: the site may be slow or unreachable from here - try again, or "
|
|
99
|
+
"skip it and use another source.",
|
|
100
|
+
is_error=True,
|
|
101
|
+
)
|
|
102
|
+
except httpx.HTTPError as exc:
|
|
103
|
+
return ToolResult(
|
|
104
|
+
output=f"Fetch failed: {exc}\n"
|
|
105
|
+
"[pcli] Suggestion: check the URL is correct and reachable, and that network access "
|
|
106
|
+
"is actually available from this environment.",
|
|
107
|
+
is_error=True,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
if response.status_code >= 400:
|
|
111
|
+
return ToolResult(
|
|
112
|
+
output=f"Fetch failed: HTTP {response.status_code} for {url}\n"
|
|
113
|
+
"[pcli] Suggestion: double-check the URL is correct and still live - a 404/403 won't "
|
|
114
|
+
"resolve by retrying the identical URL.",
|
|
115
|
+
is_error=True,
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
content_type = response.headers.get("content-type", "")
|
|
119
|
+
raw_bytes = response.content
|
|
120
|
+
if len(raw_bytes) > _MAX_FETCH_BYTES:
|
|
121
|
+
return ToolResult(
|
|
122
|
+
output=f"Response exceeded the {_MAX_FETCH_BYTES:,}-byte safety cap for {url} "
|
|
123
|
+
f"(content-type: {content_type or 'unknown'}).\n"
|
|
124
|
+
"[pcli] Suggestion: this looks like a large binary/media file, not a page to read - "
|
|
125
|
+
"use download_file instead if you actually need the raw content on disk.",
|
|
126
|
+
is_error=True,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
text = raw_bytes.decode(errors="replace")
|
|
130
|
+
if "html" in content_type:
|
|
131
|
+
text = _html_to_text(text)
|
|
132
|
+
elif not content_type.startswith(("text/", "application/json")) and content_type:
|
|
133
|
+
return ToolResult(
|
|
134
|
+
output=f"{url} is {content_type}, not text/HTML ({len(raw_bytes):,} bytes) - not "
|
|
135
|
+
"something meaningful to read as text.\n"
|
|
136
|
+
"[pcli] Suggestion: use download_file instead if you need this content on disk.",
|
|
137
|
+
is_error=True,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
if len(text) > max_chars:
|
|
141
|
+
text = text[:max_chars] + f"\n[...truncated to {max_chars:,} chars...]"
|
|
142
|
+
return ToolResult(output=text or "(empty response)")
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
WEB_FETCH = ToolSpec(
|
|
146
|
+
name="web_fetch",
|
|
147
|
+
description="Fetch a URL and return its readable text content (HTML is converted to plain "
|
|
148
|
+
"text; JSON/plain text is returned as-is). Use this to actually read the content of a page "
|
|
149
|
+
"you already have the URL for — e.g. one found via web_search, or one the user gave you. "
|
|
150
|
+
"Not for downloading files to disk (use download_file for that).",
|
|
151
|
+
parameters={
|
|
152
|
+
"type": "object",
|
|
153
|
+
"properties": {
|
|
154
|
+
"url": {"type": "string", "description": "URL to fetch."},
|
|
155
|
+
"max_chars": {
|
|
156
|
+
"type": "integer",
|
|
157
|
+
"description": f"Maximum characters of text to return (default "
|
|
158
|
+
f"{_DEFAULT_MAX_CHARS:,}).",
|
|
159
|
+
},
|
|
160
|
+
},
|
|
161
|
+
"required": ["url"],
|
|
162
|
+
},
|
|
163
|
+
handler=_web_fetch,
|
|
164
|
+
needs_permission=True,
|
|
165
|
+
risk_description="Fetches content from a URL over the network.",
|
|
166
|
+
plan_mode_safe=True,
|
|
167
|
+
read_only=True,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
class _DuckDuckGoResultParser(HTMLParser):
|
|
172
|
+
"""Parses DuckDuckGo's html.duckduckgo.com/html/ results page. Organic
|
|
173
|
+
results are `<div class="... web-result ...">` (ads use `result--ad`
|
|
174
|
+
instead and are skipped); within one, `<a class="result__a">` carries
|
|
175
|
+
the title+URL and `<a class="result__snippet">` the snippet - confirmed
|
|
176
|
+
directly against a real response, not guessed from memory."""
|
|
177
|
+
|
|
178
|
+
def __init__(self) -> None:
|
|
179
|
+
super().__init__()
|
|
180
|
+
self.results: list[dict[str, str]] = []
|
|
181
|
+
self._depth = 0
|
|
182
|
+
self._result_div_depth: int | None = None
|
|
183
|
+
self._current: dict[str, str] = {}
|
|
184
|
+
self._capturing: str | None = None # "title" | "snippet" | None
|
|
185
|
+
|
|
186
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
187
|
+
attrs_dict = dict(attrs)
|
|
188
|
+
if tag == "div":
|
|
189
|
+
self._depth += 1
|
|
190
|
+
classes = (attrs_dict.get("class") or "").split()
|
|
191
|
+
if self._result_div_depth is None and "web-result" in classes:
|
|
192
|
+
self._result_div_depth = self._depth
|
|
193
|
+
self._current = {"title": "", "url": "", "snippet": ""}
|
|
194
|
+
elif tag == "a" and self._result_div_depth is not None:
|
|
195
|
+
classes = (attrs_dict.get("class") or "").split()
|
|
196
|
+
if "result__a" in classes:
|
|
197
|
+
self._capturing = "title"
|
|
198
|
+
self._current["url"] = attrs_dict.get("href") or ""
|
|
199
|
+
elif "result__snippet" in classes:
|
|
200
|
+
self._capturing = "snippet"
|
|
201
|
+
|
|
202
|
+
def handle_endtag(self, tag: str) -> None:
|
|
203
|
+
if tag == "a":
|
|
204
|
+
self._capturing = None
|
|
205
|
+
elif tag == "div":
|
|
206
|
+
if self._result_div_depth == self._depth:
|
|
207
|
+
if self._current.get("title") and self._current.get("url"):
|
|
208
|
+
self.results.append(
|
|
209
|
+
{
|
|
210
|
+
"title": self._current["title"].strip(),
|
|
211
|
+
"url": self._current["url"],
|
|
212
|
+
"snippet": self._current["snippet"].strip(),
|
|
213
|
+
}
|
|
214
|
+
)
|
|
215
|
+
self._result_div_depth = None
|
|
216
|
+
self._depth -= 1
|
|
217
|
+
|
|
218
|
+
def handle_data(self, data: str) -> None:
|
|
219
|
+
if self._capturing == "title":
|
|
220
|
+
self._current["title"] = self._current["title"] + data
|
|
221
|
+
elif self._capturing == "snippet":
|
|
222
|
+
self._current["snippet"] = self._current["snippet"] + data
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
async def _search_duckduckgo(query: str, max_results: int) -> list[dict[str, str]]:
|
|
226
|
+
async with httpx.AsyncClient(timeout=_DEFAULT_TIMEOUT_S) as client:
|
|
227
|
+
response = await client.post(
|
|
228
|
+
"https://html.duckduckgo.com/html/",
|
|
229
|
+
data={"q": query},
|
|
230
|
+
headers={"User-Agent": _BROWSER_USER_AGENT},
|
|
231
|
+
)
|
|
232
|
+
response.raise_for_status()
|
|
233
|
+
parser = _DuckDuckGoResultParser()
|
|
234
|
+
parser.feed(response.text)
|
|
235
|
+
return parser.results[:max_results]
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
async def _search_brave(query: str, max_results: int, api_key: str) -> list[dict[str, str]]:
|
|
239
|
+
async with httpx.AsyncClient(timeout=_DEFAULT_TIMEOUT_S) as client:
|
|
240
|
+
response = await client.get(
|
|
241
|
+
"https://api.search.brave.com/res/v1/web/search",
|
|
242
|
+
params={"q": query, "count": max_results},
|
|
243
|
+
headers={"Accept": "application/json", "X-Subscription-Token": api_key},
|
|
244
|
+
)
|
|
245
|
+
response.raise_for_status()
|
|
246
|
+
data = response.json()
|
|
247
|
+
results = data.get("web", {}).get("results", [])
|
|
248
|
+
return [
|
|
249
|
+
{"title": r.get("title", ""), "url": r.get("url", ""), "snippet": r.get("description", "")}
|
|
250
|
+
for r in results[:max_results]
|
|
251
|
+
]
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _format_results(results: list[dict[str, str]]) -> str:
|
|
255
|
+
if not results:
|
|
256
|
+
return "No results."
|
|
257
|
+
lines = []
|
|
258
|
+
for index, result in enumerate(results, start=1):
|
|
259
|
+
lines.append(f"{index}. {result['title']}\n {result['url']}\n {result['snippet']}")
|
|
260
|
+
return "\n\n".join(lines)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
async def _web_search(arguments: dict, ctx: ToolContext) -> ToolResult:
|
|
264
|
+
query = arguments["query"]
|
|
265
|
+
max_results = int(arguments.get("max_results", _DEFAULT_MAX_RESULTS))
|
|
266
|
+
|
|
267
|
+
try:
|
|
268
|
+
if ctx.brave_search_api_key:
|
|
269
|
+
results = await _search_brave(query, max_results, ctx.brave_search_api_key)
|
|
270
|
+
else:
|
|
271
|
+
results = await _search_duckduckgo(query, max_results)
|
|
272
|
+
except httpx.TimeoutException:
|
|
273
|
+
return ToolResult(
|
|
274
|
+
output=f"Search timed out after {_DEFAULT_TIMEOUT_S:g}s for query: {query}\n"
|
|
275
|
+
"[pcli] Suggestion: try again, or narrow the query.",
|
|
276
|
+
is_error=True,
|
|
277
|
+
)
|
|
278
|
+
except httpx.HTTPError as exc:
|
|
279
|
+
return ToolResult(
|
|
280
|
+
output=f"Search failed: {exc}\n"
|
|
281
|
+
"[pcli] Suggestion: check network access is available from this environment. If "
|
|
282
|
+
"this keeps failing, the no-API-key DuckDuckGo fallback may be getting blocked - "
|
|
283
|
+
"configure brave_search_api_key for a real, supported search API instead.",
|
|
284
|
+
is_error=True,
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
return ToolResult(output=_format_results(results))
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
WEB_SEARCH = ToolSpec(
|
|
291
|
+
name="web_search",
|
|
292
|
+
description="Search the web and return a short list of results (title, URL, snippet). Use "
|
|
293
|
+
"web_fetch on a promising result's URL to read its actual content — search results alone "
|
|
294
|
+
"are rarely enough to answer from. Uses a configured search API if available, otherwise a "
|
|
295
|
+
"best-effort fallback with no setup required.\n\n"
|
|
296
|
+
"Query tips: use short, keyword-based queries (roughly 2-6 words), not a full question or "
|
|
297
|
+
"sentence — 'python asyncio cancel task' beats 'how do I cancel a task in python asyncio'. "
|
|
298
|
+
"Quote an exact phrase you need verbatim (an error message, a function name, a title) — "
|
|
299
|
+
"quoting is reliable on both backends. Avoid site:/filetype:/-exclusion/OR operators: they "
|
|
300
|
+
"only work with a configured search API (Brave) and are unreliable — confirmed to return "
|
|
301
|
+
"nothing, not just fewer results — on the no-setup DuckDuckGo fallback used by default. If "
|
|
302
|
+
"you need a specific site, search normally and pick the matching result, or web_fetch a "
|
|
303
|
+
"known URL directly instead. Start broad and re-search with narrower keywords based on what "
|
|
304
|
+
"the first results show, rather than building one heavily-qualified query upfront.",
|
|
305
|
+
parameters={
|
|
306
|
+
"type": "object",
|
|
307
|
+
"properties": {
|
|
308
|
+
"query": {"type": "string", "description": "Search query."},
|
|
309
|
+
"max_results": {
|
|
310
|
+
"type": "integer",
|
|
311
|
+
"description": f"Maximum number of results to return (default "
|
|
312
|
+
f"{_DEFAULT_MAX_RESULTS}).",
|
|
313
|
+
},
|
|
314
|
+
},
|
|
315
|
+
"required": ["query"],
|
|
316
|
+
},
|
|
317
|
+
handler=_web_search,
|
|
318
|
+
needs_permission=True,
|
|
319
|
+
risk_description="Sends a search query to a third-party service over the network.",
|
|
320
|
+
plan_mode_safe=True,
|
|
321
|
+
read_only=True,
|
|
322
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Disk cache for the pydiscovery index, invalidated by an environment
|
|
2
|
+
fingerprint (interpreter path + installed distribution name/version pairs)
|
|
3
|
+
so it rebuilds automatically when the venv changes, but not on every launch."""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
import sys
|
|
10
|
+
from dataclasses import asdict
|
|
11
|
+
from importlib import metadata as importlib_metadata
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from pcli.config.paths import cache_dir
|
|
15
|
+
from pcli.tools.pydiscovery.index import ModuleEntry, build_index
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def environment_fingerprint() -> str:
|
|
19
|
+
dist_pairs = sorted(
|
|
20
|
+
f"{(dist.metadata.get('Name') or dist.metadata.get('name') or '?')}=={dist.version}"
|
|
21
|
+
for dist in importlib_metadata.distributions()
|
|
22
|
+
)
|
|
23
|
+
raw = "\n".join([sys.executable, sys.version, *dist_pairs])
|
|
24
|
+
return hashlib.sha256(raw.encode()).hexdigest()[:16]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def index_cache_path() -> Path:
|
|
28
|
+
return cache_dir() / "pydiscovery_index.json"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def load_or_build_index(*, force_rebuild: bool = False) -> list[ModuleEntry]:
|
|
32
|
+
fingerprint = environment_fingerprint()
|
|
33
|
+
path = index_cache_path()
|
|
34
|
+
|
|
35
|
+
if not force_rebuild and path.exists():
|
|
36
|
+
try:
|
|
37
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
38
|
+
if data.get("fingerprint") == fingerprint:
|
|
39
|
+
return [ModuleEntry(**entry) for entry in data["entries"]]
|
|
40
|
+
except (json.JSONDecodeError, KeyError, TypeError):
|
|
41
|
+
pass # corrupt/stale cache -> fall through and rebuild
|
|
42
|
+
|
|
43
|
+
entries = build_index()
|
|
44
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
45
|
+
path.write_text(
|
|
46
|
+
json.dumps(
|
|
47
|
+
{"fingerprint": fingerprint, "entries": [asdict(e) for e in entries]}, indent=2
|
|
48
|
+
),
|
|
49
|
+
encoding="utf-8",
|
|
50
|
+
)
|
|
51
|
+
return entries
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""Enumerates importable modules/packages by NAME only — via pkgutil and
|
|
2
|
+
importlib.metadata, neither of which imports the module bodies. This is the
|
|
3
|
+
"tier 1" index: cheap, safe, always available, good for coarse search.
|
|
4
|
+
Docstrings/signatures ("tier 2") require actually importing a specific
|
|
5
|
+
module, done on demand by inspect_python_module (see search.py)."""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import pkgutil
|
|
10
|
+
import sys
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from importlib import metadata as importlib_metadata
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class ModuleEntry:
|
|
17
|
+
name: str
|
|
18
|
+
package: str | None
|
|
19
|
+
summary: str
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def build_index() -> list[ModuleEntry]:
|
|
23
|
+
entries: dict[str, ModuleEntry] = {}
|
|
24
|
+
|
|
25
|
+
package_summaries: dict[str, str] = {}
|
|
26
|
+
for dist in importlib_metadata.distributions():
|
|
27
|
+
name = dist.metadata.get("Name") or dist.metadata.get("name")
|
|
28
|
+
if not name:
|
|
29
|
+
continue
|
|
30
|
+
summary = dist.metadata.get("Summary") or ""
|
|
31
|
+
package_summaries[name.lower().replace("-", "_")] = summary
|
|
32
|
+
entries.setdefault(name, ModuleEntry(name=name, package=name, summary=summary))
|
|
33
|
+
|
|
34
|
+
for name in sorted(getattr(sys, "stdlib_module_names", ())):
|
|
35
|
+
if name.startswith("_"):
|
|
36
|
+
continue
|
|
37
|
+
entries.setdefault(
|
|
38
|
+
name, ModuleEntry(name=name, package=None, summary="(Python standard library)")
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
for module_info in pkgutil.iter_modules():
|
|
42
|
+
name = module_info.name
|
|
43
|
+
if name.startswith("_"):
|
|
44
|
+
continue
|
|
45
|
+
summary = package_summaries.get(name.lower().replace("-", "_"), "")
|
|
46
|
+
entries.setdefault(name, ModuleEntry(name=name, package=None, summary=summary))
|
|
47
|
+
|
|
48
|
+
return sorted(entries.values(), key=lambda e: e.name)
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"""Runs arbitrary Python (module introspection or a function call) in an
|
|
2
|
+
isolated subprocess, never in the main pcli process.
|
|
3
|
+
|
|
4
|
+
Deliberately does NOT go through ctx.sandbox: that may be a Docker container
|
|
5
|
+
running a bare image with none of the host's installed packages, whereas
|
|
6
|
+
pydiscovery's whole point is to reach packages installed in the *host*
|
|
7
|
+
environment. So this always spawns the host interpreter (sys.executable)
|
|
8
|
+
through a dedicated RestrictedSubprocessSandbox instead — real process
|
|
9
|
+
isolation (scrubbed env, timeout, resource limits) without losing access to
|
|
10
|
+
the packages being introspected.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import sys
|
|
17
|
+
|
|
18
|
+
from pcli.sandbox.base import ExecRequest
|
|
19
|
+
from pcli.sandbox.subprocess_backend import RestrictedSubprocessSandbox
|
|
20
|
+
from pcli.tools.base import ToolContext, ToolResult, ToolSpec
|
|
21
|
+
|
|
22
|
+
_BOOTSTRAP_SCRIPT = r"""
|
|
23
|
+
import importlib, inspect, json, sys
|
|
24
|
+
|
|
25
|
+
def _resolve(qualified_name):
|
|
26
|
+
parts = qualified_name.split(".")
|
|
27
|
+
module = None
|
|
28
|
+
remaining = []
|
|
29
|
+
for i in range(len(parts), 0, -1):
|
|
30
|
+
candidate = ".".join(parts[:i])
|
|
31
|
+
try:
|
|
32
|
+
module = importlib.import_module(candidate)
|
|
33
|
+
remaining = parts[i:]
|
|
34
|
+
break
|
|
35
|
+
except ImportError:
|
|
36
|
+
continue
|
|
37
|
+
if module is None:
|
|
38
|
+
raise ImportError("Could not import any prefix of %r" % (qualified_name,))
|
|
39
|
+
obj = module
|
|
40
|
+
for attr in remaining:
|
|
41
|
+
obj = getattr(obj, attr)
|
|
42
|
+
return obj
|
|
43
|
+
|
|
44
|
+
def _json_safe(value):
|
|
45
|
+
try:
|
|
46
|
+
json.dumps(value)
|
|
47
|
+
return value
|
|
48
|
+
except TypeError:
|
|
49
|
+
return {"__repr__": repr(value)[:2000], "__type__": type(value).__name__}
|
|
50
|
+
|
|
51
|
+
def main():
|
|
52
|
+
payload = json.loads(sys.stdin.read())
|
|
53
|
+
mode = payload["mode"]
|
|
54
|
+
if mode == "inspect":
|
|
55
|
+
module = importlib.import_module(payload["module"])
|
|
56
|
+
query = (payload.get("query") or "").lower()
|
|
57
|
+
members = []
|
|
58
|
+
for name, obj in inspect.getmembers(module):
|
|
59
|
+
if name.startswith("_"):
|
|
60
|
+
continue
|
|
61
|
+
if query and query not in name.lower():
|
|
62
|
+
continue
|
|
63
|
+
if not (inspect.isfunction(obj) or inspect.isclass(obj)
|
|
64
|
+
or inspect.isbuiltin(obj) or inspect.ismethod(obj)):
|
|
65
|
+
continue
|
|
66
|
+
try:
|
|
67
|
+
sig = str(inspect.signature(obj))
|
|
68
|
+
except (ValueError, TypeError):
|
|
69
|
+
sig = "(...)"
|
|
70
|
+
doc_lines = (inspect.getdoc(obj) or "").strip().splitlines()
|
|
71
|
+
members.append({
|
|
72
|
+
"name": name,
|
|
73
|
+
"kind": "class" if inspect.isclass(obj) else "function",
|
|
74
|
+
"signature": sig,
|
|
75
|
+
"doc": doc_lines[0] if doc_lines else "",
|
|
76
|
+
})
|
|
77
|
+
if len(members) >= 200:
|
|
78
|
+
break
|
|
79
|
+
print(json.dumps({"ok": True, "members": members}))
|
|
80
|
+
elif mode == "call":
|
|
81
|
+
obj = _resolve(payload["qualified_name"])
|
|
82
|
+
if not callable(obj):
|
|
83
|
+
print(json.dumps({"ok": False, "error": "%r is not callable" % (payload["qualified_name"],)}))
|
|
84
|
+
return
|
|
85
|
+
result = obj(*(payload.get("args") or []), **(payload.get("kwargs") or {}))
|
|
86
|
+
print(json.dumps({"ok": True, "result": _json_safe(result)}))
|
|
87
|
+
else:
|
|
88
|
+
print(json.dumps({"ok": False, "error": "unknown mode %r" % (mode,)}))
|
|
89
|
+
|
|
90
|
+
try:
|
|
91
|
+
main()
|
|
92
|
+
except Exception as exc:
|
|
93
|
+
print(json.dumps({"ok": False, "error": "%s: %s" % (type(exc).__name__, exc)}))
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
async def run_bootstrap(ctx: ToolContext, payload: dict, *, timeout_s: float = 30.0) -> dict:
|
|
98
|
+
sandbox = RestrictedSubprocessSandbox(allowed_roots=[ctx.cwd])
|
|
99
|
+
request = ExecRequest(
|
|
100
|
+
command=[sys.executable, "-c", _BOOTSTRAP_SCRIPT],
|
|
101
|
+
cwd=ctx.cwd,
|
|
102
|
+
stdin=json.dumps(payload),
|
|
103
|
+
timeout_s=timeout_s,
|
|
104
|
+
)
|
|
105
|
+
result = await sandbox.execute(request)
|
|
106
|
+
if result.timed_out:
|
|
107
|
+
raise RuntimeError("timed out")
|
|
108
|
+
output_line = result.stdout.strip().splitlines()[-1] if result.stdout.strip() else ""
|
|
109
|
+
if not output_line:
|
|
110
|
+
raise RuntimeError(result.stderr.strip() or f"no output (exit_code={result.exit_code})")
|
|
111
|
+
try:
|
|
112
|
+
return json.loads(output_line)
|
|
113
|
+
except json.JSONDecodeError as exc:
|
|
114
|
+
raise RuntimeError(f"could not parse bootstrap output: {result.stdout!r}") from exc
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
async def _call_python(arguments: dict, ctx: ToolContext) -> ToolResult:
|
|
118
|
+
qualified_name = arguments["qualified_name"]
|
|
119
|
+
payload = {
|
|
120
|
+
"mode": "call",
|
|
121
|
+
"qualified_name": qualified_name,
|
|
122
|
+
"args": arguments.get("args") or [],
|
|
123
|
+
"kwargs": arguments.get("kwargs") or {},
|
|
124
|
+
}
|
|
125
|
+
try:
|
|
126
|
+
data = await run_bootstrap(ctx, payload)
|
|
127
|
+
except RuntimeError as exc:
|
|
128
|
+
return ToolResult(
|
|
129
|
+
output=f"Call to '{qualified_name}' failed: {exc}\n"
|
|
130
|
+
"[pcli] Suggestion: use inspect_python_module to confirm the callable actually "
|
|
131
|
+
"exists under that name before retrying.",
|
|
132
|
+
is_error=True,
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
if not data.get("ok"):
|
|
136
|
+
error = data.get("error", "unknown error")
|
|
137
|
+
if "ImportError" in error or "ModuleNotFoundError" in error:
|
|
138
|
+
suggestion = "run search_python first to confirm the exact module name is installed."
|
|
139
|
+
elif "is not callable" in error:
|
|
140
|
+
suggestion = "use inspect_python_module to see what's actually callable there."
|
|
141
|
+
else:
|
|
142
|
+
suggestion = (
|
|
143
|
+
"double-check the argument names/types against inspect_python_module's "
|
|
144
|
+
"signature output."
|
|
145
|
+
)
|
|
146
|
+
return ToolResult(output=f"{error}\n[pcli] Suggestion: {suggestion}", is_error=True)
|
|
147
|
+
|
|
148
|
+
result = data["result"]
|
|
149
|
+
if isinstance(result, dict) and set(result) == {"__repr__", "__type__"}:
|
|
150
|
+
text = f"<{result['__type__']}> {result['__repr__']}"
|
|
151
|
+
elif isinstance(result, str):
|
|
152
|
+
text = result
|
|
153
|
+
else:
|
|
154
|
+
text = json.dumps(result, indent=2)
|
|
155
|
+
return ToolResult(output=text)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
CALL_PYTHON = ToolSpec(
|
|
159
|
+
name="call_python",
|
|
160
|
+
description="Call a Python function by its fully-qualified name (e.g. 'math.sqrt', "
|
|
161
|
+
"'json.dumps'), with JSON-serializable positional/keyword arguments. Runs in an "
|
|
162
|
+
"isolated subprocess with the host's installed packages, not the main pcli process.",
|
|
163
|
+
parameters={
|
|
164
|
+
"type": "object",
|
|
165
|
+
"properties": {
|
|
166
|
+
"qualified_name": {
|
|
167
|
+
"type": "string",
|
|
168
|
+
"description": "Fully-qualified function/callable name, e.g. 'math.sqrt'.",
|
|
169
|
+
},
|
|
170
|
+
"args": {"type": "array", "description": "Positional arguments (JSON values)."},
|
|
171
|
+
"kwargs": {"type": "object", "description": "Keyword arguments (JSON values)."},
|
|
172
|
+
},
|
|
173
|
+
"required": ["qualified_name"],
|
|
174
|
+
},
|
|
175
|
+
handler=_call_python,
|
|
176
|
+
needs_permission=True,
|
|
177
|
+
needs_sandbox=True,
|
|
178
|
+
risk_description="Executes an arbitrary Python function call in a subprocess.",
|
|
179
|
+
guardrail_python_module_arg="qualified_name",
|
|
180
|
+
read_only=False,
|
|
181
|
+
)
|