pi-web-access-py 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,45 @@
1
+ """pi-web-access — Web search and URL fetching extension for pi-python.
2
+
3
+ Provides two tools:
4
+ - ``web_search`` — Search the web via Brave, Tavily, or SearXNG
5
+ - ``fetch_url`` — Fetch and extract text from a URL
6
+
7
+ Install: ``pip install pi-web-access-py``
8
+
9
+ Configuration (environment variables):
10
+ - ``BRAVE_API_KEY`` — Brave Search API key
11
+ - ``TAVILY_API_KEY`` — Tavily Search API key
12
+ - ``SEARXNG_URL`` — SearXNG instance URL
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from typing import TYPE_CHECKING
18
+
19
+ if TYPE_CHECKING:
20
+ from pi_agent_core.extensions import ExtensionAPI
21
+
22
+
23
+ def activate(pi: ExtensionAPI) -> None:
24
+ """Extension entry point — called by the ExtensionLoader."""
25
+ from pi_web_access.fetch_url import create_fetch_url_tool
26
+ from pi_web_access.web_search import create_web_search_tool
27
+
28
+ pi.register_tool(create_web_search_tool())
29
+ pi.register_tool(create_fetch_url_tool())
30
+
31
+ # Passthrough: advertised for autocomplete but forwarded to LLM as a
32
+ # regular prompt so the model invokes the web_search / fetch_url tool.
33
+ _noop = lambda args: None # noqa: E731
34
+ pi.register_command(
35
+ "web_search",
36
+ description="Search the web for real-time information",
37
+ handler=_noop,
38
+ passthrough=True,
39
+ )
40
+ pi.register_command(
41
+ "fetch_url",
42
+ description="Fetch and read the contents of a URL",
43
+ handler=_noop,
44
+ passthrough=True,
45
+ )
@@ -0,0 +1,122 @@
1
+ """fetch_url tool — fetch and read the contents of a URL."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from typing import Any
7
+
8
+ import httpx
9
+ from pydantic import BaseModel, Field
10
+
11
+ from pi_agent_core.coding_tools.truncate import DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncate_head
12
+ from pi_agent_core.types import AgentToolResult
13
+
14
+
15
+ class FetchUrlParams(BaseModel):
16
+ url: str = Field(description="The URL to fetch")
17
+ extract_text: bool = Field(
18
+ default=True,
19
+ description="Extract readable text from HTML (strip tags). Set false for raw content.",
20
+ )
21
+ max_length: int | None = Field(
22
+ default=None,
23
+ description="Maximum number of lines to return (default: 2000)",
24
+ )
25
+
26
+
27
+ _TAG_RE = re.compile(r"<script[^>]*>.*?</script>|<style[^>]*>.*?</style>", re.DOTALL | re.I)
28
+ _HTML_TAG_RE = re.compile(r"<[^>]+>")
29
+ _WS_RE = re.compile(r"\n{3,}")
30
+
31
+
32
+ def _simple_html_to_text(html: str) -> str:
33
+ """Lightweight HTML-to-text (no external deps)."""
34
+ text = _TAG_RE.sub("", html)
35
+ text = _HTML_TAG_RE.sub("", text)
36
+ text = text.replace("&amp;", "&").replace("&lt;", "<").replace("&gt;", ">")
37
+ text = text.replace("&quot;", '"').replace("&#39;", "'").replace("&nbsp;", " ")
38
+ text = _WS_RE.sub("\n\n", text)
39
+ return text.strip()
40
+
41
+
42
+ def _extract_text(html: str) -> str:
43
+ """Try trafilatura first, fall back to simple tag stripping."""
44
+ try:
45
+ import trafilatura
46
+
47
+ result = trafilatura.extract(html, include_comments=False, include_tables=True)
48
+ if result:
49
+ return result
50
+ except ImportError:
51
+ pass
52
+ return _simple_html_to_text(html)
53
+
54
+
55
+ async def fetch_url_execute(
56
+ tool_call_id: str,
57
+ params: Any,
58
+ signal: Any = None,
59
+ on_update: Any = None,
60
+ ) -> AgentToolResult:
61
+ try:
62
+ async with httpx.AsyncClient(
63
+ timeout=30,
64
+ follow_redirects=True,
65
+ headers={"User-Agent": "pi-python/0.1 (web-access extension)"},
66
+ ) as client:
67
+ resp = await client.get(params.url)
68
+ resp.raise_for_status()
69
+ raw = resp.text
70
+
71
+ if params.extract_text and "html" in resp.headers.get("content-type", "").lower():
72
+ text = _extract_text(raw)
73
+ else:
74
+ text = raw
75
+
76
+ max_lines = params.max_length or DEFAULT_MAX_LINES
77
+ result = truncate_head(text, max_lines=max_lines, max_bytes=DEFAULT_MAX_BYTES)
78
+
79
+ notice = ""
80
+ if result.truncated:
81
+ notice = (
82
+ f"\n[Showing {result.outputLines} of {result.totalLines} lines. "
83
+ f"Content truncated at {max_lines} lines.]"
84
+ )
85
+
86
+ return AgentToolResult(
87
+ content=[{"type": "text", "text": result.content + notice}],
88
+ details={
89
+ "url": params.url,
90
+ "statusCode": resp.status_code,
91
+ "contentType": resp.headers.get("content-type", ""),
92
+ "truncated": result.truncated,
93
+ },
94
+ )
95
+ except httpx.HTTPStatusError as e:
96
+ return AgentToolResult(
97
+ content=[
98
+ {"type": "text", "text": f"HTTP {e.response.status_code} fetching {params.url}"}
99
+ ]
100
+ )
101
+ except Exception as e:
102
+ return AgentToolResult(
103
+ content=[{"type": "text", "text": f"Failed to fetch {params.url}: {e}"}]
104
+ )
105
+
106
+
107
+ def create_fetch_url_tool() -> Any:
108
+ """Return a ToolDefinition for the fetch_url tool."""
109
+ from pi_agent_core.extensions.types import ToolDefinition
110
+
111
+ return ToolDefinition(
112
+ name="fetch_url",
113
+ description="Fetch and read the contents of a URL. Extracts readable text from HTML pages.",
114
+ parameters=FetchUrlParams,
115
+ execute=fetch_url_execute,
116
+ label="Fetch URL",
117
+ prompt_snippet="Fetch and read the contents of a URL",
118
+ prompt_guidelines=[
119
+ "Use fetch_url to read the full content of a web page when you have a specific URL.",
120
+ "Combine with web_search: search first, then fetch relevant URLs for details.",
121
+ ],
122
+ )
@@ -0,0 +1,7 @@
1
+ """Search provider adapters for pi-web-access."""
2
+
3
+ from pi_web_access.providers.brave import BraveProvider
4
+ from pi_web_access.providers.searxng import SearXNGProvider
5
+ from pi_web_access.providers.tavily import TavilyProvider
6
+
7
+ __all__ = ["BraveProvider", "SearXNGProvider", "TavilyProvider"]
@@ -0,0 +1,49 @@
1
+ """Brave Search API provider."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from typing import Any
7
+
8
+ import httpx
9
+
10
+
11
+ @dataclass
12
+ class SearchResult:
13
+ title: str
14
+ url: str
15
+ snippet: str
16
+
17
+
18
+ class BraveProvider:
19
+ """Brave Search API (https://api.search.brave.com)."""
20
+
21
+ BASE_URL = "https://api.search.brave.com/res/v1/web/search"
22
+
23
+ def __init__(self, api_key: str) -> None:
24
+ self.api_key = api_key
25
+
26
+ async def search(self, query: str, max_results: int = 5) -> list[SearchResult]:
27
+ async with httpx.AsyncClient(timeout=30) as client:
28
+ resp = await client.get(
29
+ self.BASE_URL,
30
+ params={"q": query, "count": min(max_results, 20)},
31
+ headers={
32
+ "Accept": "application/json",
33
+ "Accept-Encoding": "gzip",
34
+ "X-Subscription-Token": self.api_key,
35
+ },
36
+ )
37
+ resp.raise_for_status()
38
+ data: dict[str, Any] = resp.json()
39
+
40
+ results: list[SearchResult] = []
41
+ for item in (data.get("web", {}).get("results") or [])[:max_results]:
42
+ results.append(
43
+ SearchResult(
44
+ title=item.get("title", ""),
45
+ url=item.get("url", ""),
46
+ snippet=item.get("description", ""),
47
+ )
48
+ )
49
+ return results
@@ -0,0 +1,37 @@
1
+ """SearXNG self-hosted provider."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ import httpx
8
+
9
+ from pi_web_access.providers.brave import SearchResult
10
+
11
+
12
+ class SearXNGProvider:
13
+ """SearXNG self-hosted instance (JSON API)."""
14
+
15
+ def __init__(self, base_url: str) -> None:
16
+ self.base_url = base_url.rstrip("/")
17
+
18
+ async def search(self, query: str, max_results: int = 5) -> list[SearchResult]:
19
+ url = f"{self.base_url}/search"
20
+ async with httpx.AsyncClient(timeout=30) as client:
21
+ resp = await client.get(
22
+ url,
23
+ params={"q": query, "format": "json", "pageno": 1},
24
+ )
25
+ resp.raise_for_status()
26
+ data: dict[str, Any] = resp.json()
27
+
28
+ results: list[SearchResult] = []
29
+ for item in (data.get("results") or [])[:max_results]:
30
+ results.append(
31
+ SearchResult(
32
+ title=item.get("title", ""),
33
+ url=item.get("url", ""),
34
+ snippet=item.get("content", ""),
35
+ )
36
+ )
37
+ return results
@@ -0,0 +1,43 @@
1
+ """Tavily Search API provider."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ import httpx
8
+
9
+ from pi_web_access.providers.brave import SearchResult
10
+
11
+
12
+ class TavilyProvider:
13
+ """Tavily Search API (https://api.tavily.com)."""
14
+
15
+ BASE_URL = "https://api.tavily.com/search"
16
+
17
+ def __init__(self, api_key: str) -> None:
18
+ self.api_key = api_key
19
+
20
+ async def search(self, query: str, max_results: int = 5) -> list[SearchResult]:
21
+ async with httpx.AsyncClient(timeout=30) as client:
22
+ resp = await client.post(
23
+ self.BASE_URL,
24
+ json={
25
+ "api_key": self.api_key,
26
+ "query": query,
27
+ "max_results": min(max_results, 20),
28
+ "include_answer": False,
29
+ },
30
+ )
31
+ resp.raise_for_status()
32
+ data: dict[str, Any] = resp.json()
33
+
34
+ results: list[SearchResult] = []
35
+ for item in (data.get("results") or [])[:max_results]:
36
+ results.append(
37
+ SearchResult(
38
+ title=item.get("title", ""),
39
+ url=item.get("url", ""),
40
+ snippet=item.get("content", ""),
41
+ )
42
+ )
43
+ return results
@@ -0,0 +1,144 @@
1
+ """web_search tool — search the web for real-time information."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from typing import Any
7
+
8
+ from pydantic import BaseModel, Field
9
+
10
+ from pi_agent_core.types import AgentToolResult
11
+
12
+
13
+ class WebSearchParams(BaseModel):
14
+ query: str = Field(description="The search query")
15
+ provider: str | None = Field(
16
+ default=None,
17
+ description="Search provider: 'brave', 'tavily', or 'searxng'. Auto-detected if omitted.",
18
+ )
19
+ max_results: int = Field(
20
+ default=5,
21
+ ge=1,
22
+ le=20,
23
+ description="Maximum number of results to return",
24
+ )
25
+
26
+
27
+ def _detect_provider() -> tuple[str, str]:
28
+ """Auto-detect search provider from environment variables.
29
+
30
+ Returns (provider_name, credential).
31
+ """
32
+ brave_key = os.environ.get("BRAVE_API_KEY")
33
+ if brave_key:
34
+ return "brave", brave_key
35
+
36
+ tavily_key = os.environ.get("TAVILY_API_KEY")
37
+ if tavily_key:
38
+ return "tavily", tavily_key
39
+
40
+ searxng_url = os.environ.get("SEARXNG_URL")
41
+ if searxng_url:
42
+ return "searxng", searxng_url
43
+
44
+ return "", ""
45
+
46
+
47
+ async def web_search_execute(
48
+ tool_call_id: str,
49
+ params: Any,
50
+ signal: Any = None,
51
+ on_update: Any = None,
52
+ ) -> AgentToolResult:
53
+ provider_name = params.provider
54
+ credential = ""
55
+
56
+ if provider_name:
57
+ if provider_name == "brave":
58
+ credential = os.environ.get("BRAVE_API_KEY", "")
59
+ elif provider_name == "tavily":
60
+ credential = os.environ.get("TAVILY_API_KEY", "")
61
+ elif provider_name == "searxng":
62
+ credential = os.environ.get("SEARXNG_URL", "")
63
+ else:
64
+ return AgentToolResult(
65
+ content=[{"type": "text", "text": f"Unknown provider: {provider_name}"}]
66
+ )
67
+ else:
68
+ provider_name, credential = _detect_provider()
69
+
70
+ if not provider_name or not credential:
71
+ return AgentToolResult(
72
+ content=[
73
+ {
74
+ "type": "text",
75
+ "text": (
76
+ "No search provider configured. Set one of these environment variables:\n"
77
+ " BRAVE_API_KEY — Brave Search API key\n"
78
+ " TAVILY_API_KEY — Tavily Search API key\n"
79
+ " SEARXNG_URL — SearXNG instance URL (e.g. http://localhost:8080)"
80
+ ),
81
+ }
82
+ ]
83
+ )
84
+
85
+ try:
86
+ if provider_name == "brave":
87
+ from pi_web_access.providers.brave import BraveProvider
88
+
89
+ provider = BraveProvider(credential)
90
+ elif provider_name == "tavily":
91
+ from pi_web_access.providers.tavily import TavilyProvider
92
+
93
+ provider = TavilyProvider(credential)
94
+ elif provider_name == "searxng":
95
+ from pi_web_access.providers.searxng import SearXNGProvider
96
+
97
+ provider = SearXNGProvider(credential)
98
+ else:
99
+ return AgentToolResult(
100
+ content=[{"type": "text", "text": f"Unknown provider: {provider_name}"}]
101
+ )
102
+
103
+ results = await provider.search(params.query, params.max_results)
104
+
105
+ if not results:
106
+ return AgentToolResult(
107
+ content=[{"type": "text", "text": f"No results found for: {params.query}"}]
108
+ )
109
+
110
+ lines = [f"Search results for: {params.query}\n"]
111
+ for i, r in enumerate(results, 1):
112
+ lines.append(f"{i}. {r.title}")
113
+ lines.append(f" {r.url}")
114
+ if r.snippet:
115
+ lines.append(f" {r.snippet}")
116
+ lines.append("")
117
+
118
+ return AgentToolResult(
119
+ content=[{"type": "text", "text": "\n".join(lines)}],
120
+ details={"provider": provider_name, "resultCount": len(results)},
121
+ )
122
+ except Exception as e:
123
+ return AgentToolResult(
124
+ content=[{"type": "text", "text": f"Search failed ({provider_name}): {e}"}]
125
+ )
126
+
127
+
128
+ def create_web_search_tool() -> dict[str, Any]:
129
+ """Return a ToolDefinition-compatible dict for the web_search tool."""
130
+ from pi_agent_core.extensions.types import ToolDefinition
131
+
132
+ return ToolDefinition(
133
+ name="web_search",
134
+ description="Search the web for real-time information using Brave, Tavily, or SearXNG",
135
+ parameters=WebSearchParams,
136
+ execute=web_search_execute,
137
+ label="Web Search",
138
+ prompt_snippet="Search the web for real-time information",
139
+ prompt_guidelines=[
140
+ "Use web_search when the user asks about current events, recent news, "
141
+ "or information that may not be in your training data.",
142
+ "Prefer web_search over guessing when you are unsure about facts.",
143
+ ],
144
+ )
@@ -0,0 +1,43 @@
1
+ Metadata-Version: 2.5
2
+ Name: pi-web-access-py
3
+ Version: 0.1.0
4
+ Summary: Web search and URL fetching extension for pi-python (port of pi-web-access)
5
+ Project-URL: Homepage, https://github.com/zy1233/pi-python
6
+ Project-URL: Repository, https://github.com/zy1233/pi-python
7
+ Author-email: zy1233 <zy1233@users.noreply.github.com>
8
+ License-Expression: MIT
9
+ Keywords: agent,extension,fetch,pi,web-search
10
+ Requires-Python: >=3.11
11
+ Requires-Dist: httpx>=0.27
12
+ Requires-Dist: pi-agent-core-lc>=0.3.0
13
+ Provides-Extra: brave
14
+ Provides-Extra: readability
15
+ Requires-Dist: trafilatura>=1.12; extra == 'readability'
16
+ Provides-Extra: tavily
17
+ Requires-Dist: tavily-python>=0.5; extra == 'tavily'
18
+ Description-Content-Type: text/markdown
19
+
20
+ # pi-web-access-py
21
+
22
+ Web search and URL fetching extension for [pi-python](https://github.com/zy1233/pi-python).
23
+
24
+ Python port of [pi-web-access](https://www.npmjs.com/package/pi-web-access).
25
+
26
+ ## Install
27
+
28
+ ```bash
29
+ pip install pi-web-access-py
30
+ ```
31
+
32
+ ## Configuration
33
+
34
+ Set one of these environment variables:
35
+
36
+ - `BRAVE_API_KEY` — Brave Search API key
37
+ - `TAVILY_API_KEY` — Tavily Search API key
38
+ - `SEARXNG_URL` — SearXNG instance URL
39
+
40
+ ## Tools
41
+
42
+ - **web_search** — Search the web for real-time information
43
+ - **fetch_url** — Fetch and extract text from a URL
@@ -0,0 +1,11 @@
1
+ pi_web_access/__init__.py,sha256=uiTR0OuIX_FJ5rMyuvyIYEOCb0btw0A_tK_net1Ag5Y,1443
2
+ pi_web_access/fetch_url.py,sha256=kwpoZRUffOKdLUjCHR_LWr3J3zJLBD6hWqYI6HbO4gA,4057
3
+ pi_web_access/web_search.py,sha256=IW5ML_MUOQDA4MyD0nP1s1YnVf5vYawrJyYq9ynDhvg,4743
4
+ pi_web_access/providers/__init__.py,sha256=bJGNrTBmDQsuAB5ru1suNTiBu8vP1thrMLfnVGpmbC0,291
5
+ pi_web_access/providers/brave.py,sha256=sjE6N2VEgiOMmS47zPyr24skPqdi861DCgeY4niokaA,1406
6
+ pi_web_access/providers/searxng.py,sha256=czOelkddMLaXWmbrzD6Vwa6ybfoB-Vw84dFV-1019Io,1108
7
+ pi_web_access/providers/tavily.py,sha256=NgVx2V_zxkxeLoiiw-gOp0bX8Py2lHw8BcDrKGcO_jU,1269
8
+ pi_web_access_py-0.1.0.dist-info/METADATA,sha256=qCT6zmlwZsUo4lZTk8HaVU8wi4KWS0bePf3TAz6258o,1264
9
+ pi_web_access_py-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
10
+ pi_web_access_py-0.1.0.dist-info/entry_points.txt,sha256=i82VtScN0CIfgaXCe8naMfzrtDacLlItxaBFCKnaz4k,61
11
+ pi_web_access_py-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [pi_agent.extensions]
2
+ pi-web-access = pi_web_access:activate