pi-web-access-py 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pi_web_access_py-0.1.0/.gitignore +34 -0
- pi_web_access_py-0.1.0/PKG-INFO +43 -0
- pi_web_access_py-0.1.0/README.md +24 -0
- pi_web_access_py-0.1.0/pi_web_access/__init__.py +45 -0
- pi_web_access_py-0.1.0/pi_web_access/fetch_url.py +122 -0
- pi_web_access_py-0.1.0/pi_web_access/providers/__init__.py +7 -0
- pi_web_access_py-0.1.0/pi_web_access/providers/brave.py +49 -0
- pi_web_access_py-0.1.0/pi_web_access/providers/searxng.py +37 -0
- pi_web_access_py-0.1.0/pi_web_access/providers/tavily.py +43 -0
- pi_web_access_py-0.1.0/pi_web_access/web_search.py +144 -0
- pi_web_access_py-0.1.0/pyproject.toml +37 -0
- pi_web_access_py-0.1.0/tests/test_web_access.py +214 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
*.egg-info/
|
|
4
|
+
.eggs/
|
|
5
|
+
dist/
|
|
6
|
+
build/
|
|
7
|
+
!tui/crates/build/
|
|
8
|
+
.pytest_cache/
|
|
9
|
+
.pytest-audit/
|
|
10
|
+
.mypy_cache/
|
|
11
|
+
.ruff_cache/
|
|
12
|
+
.venv/
|
|
13
|
+
.venv-*/
|
|
14
|
+
venv/
|
|
15
|
+
.env
|
|
16
|
+
# Rust TUI workspace (Apache-2.0 fork under tui/)
|
|
17
|
+
tui/target/
|
|
18
|
+
**/*.rs.bk
|
|
19
|
+
|
|
20
|
+
# Benchmark evaluation cache and trial artifacts
|
|
21
|
+
.cache/
|
|
22
|
+
.pi-eval/
|
|
23
|
+
|
|
24
|
+
# Temporary files, coverage and logs
|
|
25
|
+
*.log
|
|
26
|
+
*.tmp
|
|
27
|
+
*.bak
|
|
28
|
+
*.swp
|
|
29
|
+
*.orig
|
|
30
|
+
.coverage
|
|
31
|
+
.coverage.*
|
|
32
|
+
coverage.xml
|
|
33
|
+
htmlcov/
|
|
34
|
+
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pi-web-access-py
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Web search and URL fetching extension for pi-python (port of pi-web-access)
|
|
5
|
+
Project-URL: Homepage, https://github.com/zy1233/pi-python
|
|
6
|
+
Project-URL: Repository, https://github.com/zy1233/pi-python
|
|
7
|
+
Author-email: zy1233 <zy1233@users.noreply.github.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Keywords: agent,extension,fetch,pi,web-search
|
|
10
|
+
Requires-Python: >=3.11
|
|
11
|
+
Requires-Dist: httpx>=0.27
|
|
12
|
+
Requires-Dist: pi-agent-core-lc>=0.3.0
|
|
13
|
+
Provides-Extra: brave
|
|
14
|
+
Provides-Extra: readability
|
|
15
|
+
Requires-Dist: trafilatura>=1.12; extra == 'readability'
|
|
16
|
+
Provides-Extra: tavily
|
|
17
|
+
Requires-Dist: tavily-python>=0.5; extra == 'tavily'
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# pi-web-access-py
|
|
21
|
+
|
|
22
|
+
Web search and URL fetching extension for [pi-python](https://github.com/zy1233/pi-python).
|
|
23
|
+
|
|
24
|
+
Python port of [pi-web-access](https://www.npmjs.com/package/pi-web-access).
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install pi-web-access-py
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
## Configuration
|
|
33
|
+
|
|
34
|
+
Set one of these environment variables:
|
|
35
|
+
|
|
36
|
+
- `BRAVE_API_KEY` — Brave Search API key
|
|
37
|
+
- `TAVILY_API_KEY` — Tavily Search API key
|
|
38
|
+
- `SEARXNG_URL` — SearXNG instance URL
|
|
39
|
+
|
|
40
|
+
## Tools
|
|
41
|
+
|
|
42
|
+
- **web_search** — Search the web for real-time information
|
|
43
|
+
- **fetch_url** — Fetch and extract text from a URL
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# pi-web-access-py
|
|
2
|
+
|
|
3
|
+
Web search and URL fetching extension for [pi-python](https://github.com/zy1233/pi-python).
|
|
4
|
+
|
|
5
|
+
Python port of [pi-web-access](https://www.npmjs.com/package/pi-web-access).
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install pi-web-access-py
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Configuration
|
|
14
|
+
|
|
15
|
+
Set one of these environment variables:
|
|
16
|
+
|
|
17
|
+
- `BRAVE_API_KEY` — Brave Search API key
|
|
18
|
+
- `TAVILY_API_KEY` — Tavily Search API key
|
|
19
|
+
- `SEARXNG_URL` — SearXNG instance URL
|
|
20
|
+
|
|
21
|
+
## Tools
|
|
22
|
+
|
|
23
|
+
- **web_search** — Search the web for real-time information
|
|
24
|
+
- **fetch_url** — Fetch and extract text from a URL
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""pi-web-access — Web search and URL fetching extension for pi-python.
|
|
2
|
+
|
|
3
|
+
Provides two tools:
|
|
4
|
+
- ``web_search`` — Search the web via Brave, Tavily, or SearXNG
|
|
5
|
+
- ``fetch_url`` — Fetch and extract text from a URL
|
|
6
|
+
|
|
7
|
+
Install: ``pip install pi-web-access-py``
|
|
8
|
+
|
|
9
|
+
Configuration (environment variables):
|
|
10
|
+
- ``BRAVE_API_KEY`` — Brave Search API key
|
|
11
|
+
- ``TAVILY_API_KEY`` — Tavily Search API key
|
|
12
|
+
- ``SEARXNG_URL`` — SearXNG instance URL
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from typing import TYPE_CHECKING
|
|
18
|
+
|
|
19
|
+
if TYPE_CHECKING:
|
|
20
|
+
from pi_agent_core.extensions import ExtensionAPI
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def activate(pi: ExtensionAPI) -> None:
|
|
24
|
+
"""Extension entry point — called by the ExtensionLoader."""
|
|
25
|
+
from pi_web_access.fetch_url import create_fetch_url_tool
|
|
26
|
+
from pi_web_access.web_search import create_web_search_tool
|
|
27
|
+
|
|
28
|
+
pi.register_tool(create_web_search_tool())
|
|
29
|
+
pi.register_tool(create_fetch_url_tool())
|
|
30
|
+
|
|
31
|
+
# Passthrough: advertised for autocomplete but forwarded to LLM as a
|
|
32
|
+
# regular prompt so the model invokes the web_search / fetch_url tool.
|
|
33
|
+
_noop = lambda args: None # noqa: E731
|
|
34
|
+
pi.register_command(
|
|
35
|
+
"web_search",
|
|
36
|
+
description="Search the web for real-time information",
|
|
37
|
+
handler=_noop,
|
|
38
|
+
passthrough=True,
|
|
39
|
+
)
|
|
40
|
+
pi.register_command(
|
|
41
|
+
"fetch_url",
|
|
42
|
+
description="Fetch and read the contents of a URL",
|
|
43
|
+
handler=_noop,
|
|
44
|
+
passthrough=True,
|
|
45
|
+
)
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""fetch_url tool — fetch and read the contents of a URL."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import httpx
|
|
9
|
+
from pydantic import BaseModel, Field
|
|
10
|
+
|
|
11
|
+
from pi_agent_core.coding_tools.truncate import DEFAULT_MAX_BYTES, DEFAULT_MAX_LINES, truncate_head
|
|
12
|
+
from pi_agent_core.types import AgentToolResult
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class FetchUrlParams(BaseModel):
|
|
16
|
+
url: str = Field(description="The URL to fetch")
|
|
17
|
+
extract_text: bool = Field(
|
|
18
|
+
default=True,
|
|
19
|
+
description="Extract readable text from HTML (strip tags). Set false for raw content.",
|
|
20
|
+
)
|
|
21
|
+
max_length: int | None = Field(
|
|
22
|
+
default=None,
|
|
23
|
+
description="Maximum number of lines to return (default: 2000)",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
_TAG_RE = re.compile(r"<script[^>]*>.*?</script>|<style[^>]*>.*?</style>", re.DOTALL | re.I)
|
|
28
|
+
_HTML_TAG_RE = re.compile(r"<[^>]+>")
|
|
29
|
+
_WS_RE = re.compile(r"\n{3,}")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _simple_html_to_text(html: str) -> str:
|
|
33
|
+
"""Lightweight HTML-to-text (no external deps)."""
|
|
34
|
+
text = _TAG_RE.sub("", html)
|
|
35
|
+
text = _HTML_TAG_RE.sub("", text)
|
|
36
|
+
text = text.replace("&", "&").replace("<", "<").replace(">", ">")
|
|
37
|
+
text = text.replace(""", '"').replace("'", "'").replace(" ", " ")
|
|
38
|
+
text = _WS_RE.sub("\n\n", text)
|
|
39
|
+
return text.strip()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _extract_text(html: str) -> str:
|
|
43
|
+
"""Try trafilatura first, fall back to simple tag stripping."""
|
|
44
|
+
try:
|
|
45
|
+
import trafilatura
|
|
46
|
+
|
|
47
|
+
result = trafilatura.extract(html, include_comments=False, include_tables=True)
|
|
48
|
+
if result:
|
|
49
|
+
return result
|
|
50
|
+
except ImportError:
|
|
51
|
+
pass
|
|
52
|
+
return _simple_html_to_text(html)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
async def fetch_url_execute(
|
|
56
|
+
tool_call_id: str,
|
|
57
|
+
params: Any,
|
|
58
|
+
signal: Any = None,
|
|
59
|
+
on_update: Any = None,
|
|
60
|
+
) -> AgentToolResult:
|
|
61
|
+
try:
|
|
62
|
+
async with httpx.AsyncClient(
|
|
63
|
+
timeout=30,
|
|
64
|
+
follow_redirects=True,
|
|
65
|
+
headers={"User-Agent": "pi-python/0.1 (web-access extension)"},
|
|
66
|
+
) as client:
|
|
67
|
+
resp = await client.get(params.url)
|
|
68
|
+
resp.raise_for_status()
|
|
69
|
+
raw = resp.text
|
|
70
|
+
|
|
71
|
+
if params.extract_text and "html" in resp.headers.get("content-type", "").lower():
|
|
72
|
+
text = _extract_text(raw)
|
|
73
|
+
else:
|
|
74
|
+
text = raw
|
|
75
|
+
|
|
76
|
+
max_lines = params.max_length or DEFAULT_MAX_LINES
|
|
77
|
+
result = truncate_head(text, max_lines=max_lines, max_bytes=DEFAULT_MAX_BYTES)
|
|
78
|
+
|
|
79
|
+
notice = ""
|
|
80
|
+
if result.truncated:
|
|
81
|
+
notice = (
|
|
82
|
+
f"\n[Showing {result.outputLines} of {result.totalLines} lines. "
|
|
83
|
+
f"Content truncated at {max_lines} lines.]"
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
return AgentToolResult(
|
|
87
|
+
content=[{"type": "text", "text": result.content + notice}],
|
|
88
|
+
details={
|
|
89
|
+
"url": params.url,
|
|
90
|
+
"statusCode": resp.status_code,
|
|
91
|
+
"contentType": resp.headers.get("content-type", ""),
|
|
92
|
+
"truncated": result.truncated,
|
|
93
|
+
},
|
|
94
|
+
)
|
|
95
|
+
except httpx.HTTPStatusError as e:
|
|
96
|
+
return AgentToolResult(
|
|
97
|
+
content=[
|
|
98
|
+
{"type": "text", "text": f"HTTP {e.response.status_code} fetching {params.url}"}
|
|
99
|
+
]
|
|
100
|
+
)
|
|
101
|
+
except Exception as e:
|
|
102
|
+
return AgentToolResult(
|
|
103
|
+
content=[{"type": "text", "text": f"Failed to fetch {params.url}: {e}"}]
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def create_fetch_url_tool() -> Any:
|
|
108
|
+
"""Return a ToolDefinition for the fetch_url tool."""
|
|
109
|
+
from pi_agent_core.extensions.types import ToolDefinition
|
|
110
|
+
|
|
111
|
+
return ToolDefinition(
|
|
112
|
+
name="fetch_url",
|
|
113
|
+
description="Fetch and read the contents of a URL. Extracts readable text from HTML pages.",
|
|
114
|
+
parameters=FetchUrlParams,
|
|
115
|
+
execute=fetch_url_execute,
|
|
116
|
+
label="Fetch URL",
|
|
117
|
+
prompt_snippet="Fetch and read the contents of a URL",
|
|
118
|
+
prompt_guidelines=[
|
|
119
|
+
"Use fetch_url to read the full content of a web page when you have a specific URL.",
|
|
120
|
+
"Combine with web_search: search first, then fetch relevant URLs for details.",
|
|
121
|
+
],
|
|
122
|
+
)
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Search provider adapters for pi-web-access."""
|
|
2
|
+
|
|
3
|
+
from pi_web_access.providers.brave import BraveProvider
|
|
4
|
+
from pi_web_access.providers.searxng import SearXNGProvider
|
|
5
|
+
from pi_web_access.providers.tavily import TavilyProvider
|
|
6
|
+
|
|
7
|
+
__all__ = ["BraveProvider", "SearXNGProvider", "TavilyProvider"]
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Brave Search API provider."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import httpx
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class SearchResult:
|
|
13
|
+
title: str
|
|
14
|
+
url: str
|
|
15
|
+
snippet: str
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class BraveProvider:
|
|
19
|
+
"""Brave Search API (https://api.search.brave.com)."""
|
|
20
|
+
|
|
21
|
+
BASE_URL = "https://api.search.brave.com/res/v1/web/search"
|
|
22
|
+
|
|
23
|
+
def __init__(self, api_key: str) -> None:
|
|
24
|
+
self.api_key = api_key
|
|
25
|
+
|
|
26
|
+
async def search(self, query: str, max_results: int = 5) -> list[SearchResult]:
|
|
27
|
+
async with httpx.AsyncClient(timeout=30) as client:
|
|
28
|
+
resp = await client.get(
|
|
29
|
+
self.BASE_URL,
|
|
30
|
+
params={"q": query, "count": min(max_results, 20)},
|
|
31
|
+
headers={
|
|
32
|
+
"Accept": "application/json",
|
|
33
|
+
"Accept-Encoding": "gzip",
|
|
34
|
+
"X-Subscription-Token": self.api_key,
|
|
35
|
+
},
|
|
36
|
+
)
|
|
37
|
+
resp.raise_for_status()
|
|
38
|
+
data: dict[str, Any] = resp.json()
|
|
39
|
+
|
|
40
|
+
results: list[SearchResult] = []
|
|
41
|
+
for item in (data.get("web", {}).get("results") or [])[:max_results]:
|
|
42
|
+
results.append(
|
|
43
|
+
SearchResult(
|
|
44
|
+
title=item.get("title", ""),
|
|
45
|
+
url=item.get("url", ""),
|
|
46
|
+
snippet=item.get("description", ""),
|
|
47
|
+
)
|
|
48
|
+
)
|
|
49
|
+
return results
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""SearXNG self-hosted provider."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
import httpx
|
|
8
|
+
|
|
9
|
+
from pi_web_access.providers.brave import SearchResult
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class SearXNGProvider:
|
|
13
|
+
"""SearXNG self-hosted instance (JSON API)."""
|
|
14
|
+
|
|
15
|
+
def __init__(self, base_url: str) -> None:
|
|
16
|
+
self.base_url = base_url.rstrip("/")
|
|
17
|
+
|
|
18
|
+
async def search(self, query: str, max_results: int = 5) -> list[SearchResult]:
|
|
19
|
+
url = f"{self.base_url}/search"
|
|
20
|
+
async with httpx.AsyncClient(timeout=30) as client:
|
|
21
|
+
resp = await client.get(
|
|
22
|
+
url,
|
|
23
|
+
params={"q": query, "format": "json", "pageno": 1},
|
|
24
|
+
)
|
|
25
|
+
resp.raise_for_status()
|
|
26
|
+
data: dict[str, Any] = resp.json()
|
|
27
|
+
|
|
28
|
+
results: list[SearchResult] = []
|
|
29
|
+
for item in (data.get("results") or [])[:max_results]:
|
|
30
|
+
results.append(
|
|
31
|
+
SearchResult(
|
|
32
|
+
title=item.get("title", ""),
|
|
33
|
+
url=item.get("url", ""),
|
|
34
|
+
snippet=item.get("content", ""),
|
|
35
|
+
)
|
|
36
|
+
)
|
|
37
|
+
return results
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Tavily Search API provider."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
import httpx
|
|
8
|
+
|
|
9
|
+
from pi_web_access.providers.brave import SearchResult
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class TavilyProvider:
|
|
13
|
+
"""Tavily Search API (https://api.tavily.com)."""
|
|
14
|
+
|
|
15
|
+
BASE_URL = "https://api.tavily.com/search"
|
|
16
|
+
|
|
17
|
+
def __init__(self, api_key: str) -> None:
|
|
18
|
+
self.api_key = api_key
|
|
19
|
+
|
|
20
|
+
async def search(self, query: str, max_results: int = 5) -> list[SearchResult]:
|
|
21
|
+
async with httpx.AsyncClient(timeout=30) as client:
|
|
22
|
+
resp = await client.post(
|
|
23
|
+
self.BASE_URL,
|
|
24
|
+
json={
|
|
25
|
+
"api_key": self.api_key,
|
|
26
|
+
"query": query,
|
|
27
|
+
"max_results": min(max_results, 20),
|
|
28
|
+
"include_answer": False,
|
|
29
|
+
},
|
|
30
|
+
)
|
|
31
|
+
resp.raise_for_status()
|
|
32
|
+
data: dict[str, Any] = resp.json()
|
|
33
|
+
|
|
34
|
+
results: list[SearchResult] = []
|
|
35
|
+
for item in (data.get("results") or [])[:max_results]:
|
|
36
|
+
results.append(
|
|
37
|
+
SearchResult(
|
|
38
|
+
title=item.get("title", ""),
|
|
39
|
+
url=item.get("url", ""),
|
|
40
|
+
snippet=item.get("content", ""),
|
|
41
|
+
)
|
|
42
|
+
)
|
|
43
|
+
return results
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""web_search tool — search the web for real-time information."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from pydantic import BaseModel, Field
|
|
9
|
+
|
|
10
|
+
from pi_agent_core.types import AgentToolResult
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class WebSearchParams(BaseModel):
|
|
14
|
+
query: str = Field(description="The search query")
|
|
15
|
+
provider: str | None = Field(
|
|
16
|
+
default=None,
|
|
17
|
+
description="Search provider: 'brave', 'tavily', or 'searxng'. Auto-detected if omitted.",
|
|
18
|
+
)
|
|
19
|
+
max_results: int = Field(
|
|
20
|
+
default=5,
|
|
21
|
+
ge=1,
|
|
22
|
+
le=20,
|
|
23
|
+
description="Maximum number of results to return",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _detect_provider() -> tuple[str, str]:
|
|
28
|
+
"""Auto-detect search provider from environment variables.
|
|
29
|
+
|
|
30
|
+
Returns (provider_name, credential).
|
|
31
|
+
"""
|
|
32
|
+
brave_key = os.environ.get("BRAVE_API_KEY")
|
|
33
|
+
if brave_key:
|
|
34
|
+
return "brave", brave_key
|
|
35
|
+
|
|
36
|
+
tavily_key = os.environ.get("TAVILY_API_KEY")
|
|
37
|
+
if tavily_key:
|
|
38
|
+
return "tavily", tavily_key
|
|
39
|
+
|
|
40
|
+
searxng_url = os.environ.get("SEARXNG_URL")
|
|
41
|
+
if searxng_url:
|
|
42
|
+
return "searxng", searxng_url
|
|
43
|
+
|
|
44
|
+
return "", ""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
async def web_search_execute(
|
|
48
|
+
tool_call_id: str,
|
|
49
|
+
params: Any,
|
|
50
|
+
signal: Any = None,
|
|
51
|
+
on_update: Any = None,
|
|
52
|
+
) -> AgentToolResult:
|
|
53
|
+
provider_name = params.provider
|
|
54
|
+
credential = ""
|
|
55
|
+
|
|
56
|
+
if provider_name:
|
|
57
|
+
if provider_name == "brave":
|
|
58
|
+
credential = os.environ.get("BRAVE_API_KEY", "")
|
|
59
|
+
elif provider_name == "tavily":
|
|
60
|
+
credential = os.environ.get("TAVILY_API_KEY", "")
|
|
61
|
+
elif provider_name == "searxng":
|
|
62
|
+
credential = os.environ.get("SEARXNG_URL", "")
|
|
63
|
+
else:
|
|
64
|
+
return AgentToolResult(
|
|
65
|
+
content=[{"type": "text", "text": f"Unknown provider: {provider_name}"}]
|
|
66
|
+
)
|
|
67
|
+
else:
|
|
68
|
+
provider_name, credential = _detect_provider()
|
|
69
|
+
|
|
70
|
+
if not provider_name or not credential:
|
|
71
|
+
return AgentToolResult(
|
|
72
|
+
content=[
|
|
73
|
+
{
|
|
74
|
+
"type": "text",
|
|
75
|
+
"text": (
|
|
76
|
+
"No search provider configured. Set one of these environment variables:\n"
|
|
77
|
+
" BRAVE_API_KEY — Brave Search API key\n"
|
|
78
|
+
" TAVILY_API_KEY — Tavily Search API key\n"
|
|
79
|
+
" SEARXNG_URL — SearXNG instance URL (e.g. http://localhost:8080)"
|
|
80
|
+
),
|
|
81
|
+
}
|
|
82
|
+
]
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
try:
|
|
86
|
+
if provider_name == "brave":
|
|
87
|
+
from pi_web_access.providers.brave import BraveProvider
|
|
88
|
+
|
|
89
|
+
provider = BraveProvider(credential)
|
|
90
|
+
elif provider_name == "tavily":
|
|
91
|
+
from pi_web_access.providers.tavily import TavilyProvider
|
|
92
|
+
|
|
93
|
+
provider = TavilyProvider(credential)
|
|
94
|
+
elif provider_name == "searxng":
|
|
95
|
+
from pi_web_access.providers.searxng import SearXNGProvider
|
|
96
|
+
|
|
97
|
+
provider = SearXNGProvider(credential)
|
|
98
|
+
else:
|
|
99
|
+
return AgentToolResult(
|
|
100
|
+
content=[{"type": "text", "text": f"Unknown provider: {provider_name}"}]
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
results = await provider.search(params.query, params.max_results)
|
|
104
|
+
|
|
105
|
+
if not results:
|
|
106
|
+
return AgentToolResult(
|
|
107
|
+
content=[{"type": "text", "text": f"No results found for: {params.query}"}]
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
lines = [f"Search results for: {params.query}\n"]
|
|
111
|
+
for i, r in enumerate(results, 1):
|
|
112
|
+
lines.append(f"{i}. {r.title}")
|
|
113
|
+
lines.append(f" {r.url}")
|
|
114
|
+
if r.snippet:
|
|
115
|
+
lines.append(f" {r.snippet}")
|
|
116
|
+
lines.append("")
|
|
117
|
+
|
|
118
|
+
return AgentToolResult(
|
|
119
|
+
content=[{"type": "text", "text": "\n".join(lines)}],
|
|
120
|
+
details={"provider": provider_name, "resultCount": len(results)},
|
|
121
|
+
)
|
|
122
|
+
except Exception as e:
|
|
123
|
+
return AgentToolResult(
|
|
124
|
+
content=[{"type": "text", "text": f"Search failed ({provider_name}): {e}"}]
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def create_web_search_tool() -> dict[str, Any]:
|
|
129
|
+
"""Return a ToolDefinition-compatible dict for the web_search tool."""
|
|
130
|
+
from pi_agent_core.extensions.types import ToolDefinition
|
|
131
|
+
|
|
132
|
+
return ToolDefinition(
|
|
133
|
+
name="web_search",
|
|
134
|
+
description="Search the web for real-time information using Brave, Tavily, or SearXNG",
|
|
135
|
+
parameters=WebSearchParams,
|
|
136
|
+
execute=web_search_execute,
|
|
137
|
+
label="Web Search",
|
|
138
|
+
prompt_snippet="Search the web for real-time information",
|
|
139
|
+
prompt_guidelines=[
|
|
140
|
+
"Use web_search when the user asks about current events, recent news, "
|
|
141
|
+
"or information that may not be in your training data.",
|
|
142
|
+
"Prefer web_search over guessing when you are unsure about facts.",
|
|
143
|
+
],
|
|
144
|
+
)
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pi-web-access-py"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Web search and URL fetching extension for pi-python (port of pi-web-access)"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "zy1233", email = "zy1233@users.noreply.github.com" },
|
|
14
|
+
]
|
|
15
|
+
keywords = ["agent", "web-search", "fetch", "pi", "extension"]
|
|
16
|
+
dependencies = [
|
|
17
|
+
"pi-agent-core-lc>=0.3.0",
|
|
18
|
+
"httpx>=0.27",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
[project.optional-dependencies]
|
|
22
|
+
readability = ["trafilatura>=1.12"]
|
|
23
|
+
brave = []
|
|
24
|
+
tavily = ["tavily-python>=0.5"]
|
|
25
|
+
|
|
26
|
+
[project.entry-points."pi_agent.extensions"]
|
|
27
|
+
pi-web-access = "pi_web_access:activate"
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/zy1233/pi-python"
|
|
31
|
+
Repository = "https://github.com/zy1233/pi-python"
|
|
32
|
+
|
|
33
|
+
[tool.hatch.build.targets.wheel]
|
|
34
|
+
packages = ["pi_web_access"]
|
|
35
|
+
|
|
36
|
+
[tool.uv.sources]
|
|
37
|
+
pi-agent-core-lc = { workspace = true }
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""Tests for pi-web-access extension (mock HTTP, no real API keys)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from unittest.mock import AsyncMock, MagicMock, patch
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
from pi_web_access.fetch_url import (
|
|
9
|
+
FetchUrlParams,
|
|
10
|
+
_simple_html_to_text,
|
|
11
|
+
create_fetch_url_tool,
|
|
12
|
+
fetch_url_execute,
|
|
13
|
+
)
|
|
14
|
+
from pi_web_access.web_search import (
|
|
15
|
+
WebSearchParams,
|
|
16
|
+
_detect_provider,
|
|
17
|
+
create_web_search_tool,
|
|
18
|
+
web_search_execute,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
# ---------------------------------------------------------------------------
|
|
22
|
+
# _simple_html_to_text
|
|
23
|
+
# ---------------------------------------------------------------------------
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class TestHtmlToText:
|
|
27
|
+
def test_strips_tags(self) -> None:
|
|
28
|
+
html = "<p>Hello <b>world</b></p>"
|
|
29
|
+
assert "Hello world" in _simple_html_to_text(html)
|
|
30
|
+
|
|
31
|
+
def test_strips_scripts(self) -> None:
|
|
32
|
+
html = "<script>alert('x')</script><p>Content</p>"
|
|
33
|
+
text = _simple_html_to_text(html)
|
|
34
|
+
assert "alert" not in text
|
|
35
|
+
assert "Content" in text
|
|
36
|
+
|
|
37
|
+
def test_strips_styles(self) -> None:
|
|
38
|
+
html = "<style>body{color:red}</style><p>Text</p>"
|
|
39
|
+
text = _simple_html_to_text(html)
|
|
40
|
+
assert "color" not in text
|
|
41
|
+
assert "Text" in text
|
|
42
|
+
|
|
43
|
+
def test_decodes_entities(self) -> None:
|
|
44
|
+
html = "& < > " '"
|
|
45
|
+
text = _simple_html_to_text(html)
|
|
46
|
+
assert "& < > " in text
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# ---------------------------------------------------------------------------
|
|
50
|
+
# web_search — provider detection
|
|
51
|
+
# ---------------------------------------------------------------------------
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class TestProviderDetection:
|
|
55
|
+
def test_brave_detected(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
56
|
+
monkeypatch.setenv("BRAVE_API_KEY", "test-key")
|
|
57
|
+
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
|
|
58
|
+
monkeypatch.delenv("SEARXNG_URL", raising=False)
|
|
59
|
+
name, cred = _detect_provider()
|
|
60
|
+
assert name == "brave"
|
|
61
|
+
assert cred == "test-key"
|
|
62
|
+
|
|
63
|
+
def test_tavily_detected(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
64
|
+
monkeypatch.delenv("BRAVE_API_KEY", raising=False)
|
|
65
|
+
monkeypatch.setenv("TAVILY_API_KEY", "tvly-key")
|
|
66
|
+
monkeypatch.delenv("SEARXNG_URL", raising=False)
|
|
67
|
+
name, _cred = _detect_provider()
|
|
68
|
+
assert name == "tavily"
|
|
69
|
+
|
|
70
|
+
def test_searxng_detected(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
71
|
+
monkeypatch.delenv("BRAVE_API_KEY", raising=False)
|
|
72
|
+
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
|
|
73
|
+
monkeypatch.setenv("SEARXNG_URL", "http://localhost:8080")
|
|
74
|
+
name, _cred = _detect_provider()
|
|
75
|
+
assert name == "searxng"
|
|
76
|
+
|
|
77
|
+
def test_none_detected(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
78
|
+
monkeypatch.delenv("BRAVE_API_KEY", raising=False)
|
|
79
|
+
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
|
|
80
|
+
monkeypatch.delenv("SEARXNG_URL", raising=False)
|
|
81
|
+
name, _cred = _detect_provider()
|
|
82
|
+
assert name == ""
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# ---------------------------------------------------------------------------
|
|
86
|
+
# web_search — execute with mock
|
|
87
|
+
# ---------------------------------------------------------------------------
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class TestWebSearchExecute:
|
|
91
|
+
async def test_no_provider_configured(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
92
|
+
monkeypatch.delenv("BRAVE_API_KEY", raising=False)
|
|
93
|
+
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
|
|
94
|
+
monkeypatch.delenv("SEARXNG_URL", raising=False)
|
|
95
|
+
result = await web_search_execute("tc-1", WebSearchParams(query="test"))
|
|
96
|
+
text = result.content[0]["text"]
|
|
97
|
+
assert "No search provider configured" in text
|
|
98
|
+
|
|
99
|
+
async def test_brave_search_mock(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
100
|
+
monkeypatch.setenv("BRAVE_API_KEY", "fake-key")
|
|
101
|
+
mock_results = [
|
|
102
|
+
{"title": "Result 1", "url": "https://example.com", "description": "Snippet 1"}
|
|
103
|
+
]
|
|
104
|
+
mock_response = MagicMock()
|
|
105
|
+
mock_response.status_code = 200
|
|
106
|
+
mock_response.json.return_value = {"web": {"results": mock_results}}
|
|
107
|
+
mock_response.raise_for_status = MagicMock()
|
|
108
|
+
|
|
109
|
+
mock_client = AsyncMock()
|
|
110
|
+
mock_client.get.return_value = mock_response
|
|
111
|
+
mock_client.__aenter__ = AsyncMock(return_value=mock_client)
|
|
112
|
+
mock_client.__aexit__ = AsyncMock(return_value=False)
|
|
113
|
+
|
|
114
|
+
with patch("pi_web_access.providers.brave.httpx.AsyncClient", return_value=mock_client):
|
|
115
|
+
result = await web_search_execute(
|
|
116
|
+
"tc-1", WebSearchParams(query="test", provider="brave")
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
text = result.content[0]["text"]
|
|
120
|
+
assert "Result 1" in text
|
|
121
|
+
assert "example.com" in text
|
|
122
|
+
|
|
123
|
+
async def test_unknown_provider(self, monkeypatch: pytest.MonkeyPatch) -> None:
|
|
124
|
+
monkeypatch.setenv("BRAVE_API_KEY", "x")
|
|
125
|
+
result = await web_search_execute(
|
|
126
|
+
"tc-1", WebSearchParams(query="test", provider="unknown_provider")
|
|
127
|
+
)
|
|
128
|
+
text = result.content[0]["text"]
|
|
129
|
+
assert "Unknown provider" in text
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# ---------------------------------------------------------------------------
|
|
133
|
+
# fetch_url — execute with mock
|
|
134
|
+
# ---------------------------------------------------------------------------
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class TestFetchUrlExecute:
|
|
138
|
+
async def test_fetch_html_extract(self) -> None:
|
|
139
|
+
html = "<html><body><p>Hello World</p></body></html>"
|
|
140
|
+
mock_response = AsyncMock()
|
|
141
|
+
mock_response.status_code = 200
|
|
142
|
+
mock_response.text = html
|
|
143
|
+
mock_response.headers = {"content-type": "text/html; charset=utf-8"}
|
|
144
|
+
mock_response.raise_for_status = lambda: None
|
|
145
|
+
|
|
146
|
+
mock_client = AsyncMock()
|
|
147
|
+
mock_client.get.return_value = mock_response
|
|
148
|
+
mock_client.__aenter__ = AsyncMock(return_value=mock_client)
|
|
149
|
+
mock_client.__aexit__ = AsyncMock(return_value=False)
|
|
150
|
+
|
|
151
|
+
with patch("pi_web_access.fetch_url.httpx.AsyncClient", return_value=mock_client):
|
|
152
|
+
result = await fetch_url_execute("tc-1", FetchUrlParams(url="https://example.com"))
|
|
153
|
+
|
|
154
|
+
text = result.content[0]["text"]
|
|
155
|
+
assert "Hello World" in text
|
|
156
|
+
assert result.details["statusCode"] == 200
|
|
157
|
+
|
|
158
|
+
async def test_fetch_raw_no_extract(self) -> None:
|
|
159
|
+
raw = '{"key": "value"}'
|
|
160
|
+
mock_response = AsyncMock()
|
|
161
|
+
mock_response.status_code = 200
|
|
162
|
+
mock_response.text = raw
|
|
163
|
+
mock_response.headers = {"content-type": "application/json"}
|
|
164
|
+
mock_response.raise_for_status = lambda: None
|
|
165
|
+
|
|
166
|
+
mock_client = AsyncMock()
|
|
167
|
+
mock_client.get.return_value = mock_response
|
|
168
|
+
mock_client.__aenter__ = AsyncMock(return_value=mock_client)
|
|
169
|
+
mock_client.__aexit__ = AsyncMock(return_value=False)
|
|
170
|
+
|
|
171
|
+
with patch("pi_web_access.fetch_url.httpx.AsyncClient", return_value=mock_client):
|
|
172
|
+
result = await fetch_url_execute(
|
|
173
|
+
"tc-1", FetchUrlParams(url="https://api.example.com/data", extract_text=False)
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
text = result.content[0]["text"]
|
|
177
|
+
assert '"key"' in text
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
# ---------------------------------------------------------------------------
|
|
181
|
+
# ToolDefinition shapes
|
|
182
|
+
# ---------------------------------------------------------------------------
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
class TestToolDefinitions:
|
|
186
|
+
def test_web_search_tool_definition(self) -> None:
|
|
187
|
+
tool = create_web_search_tool()
|
|
188
|
+
assert tool.name == "web_search"
|
|
189
|
+
assert tool.prompt_snippet is not None
|
|
190
|
+
|
|
191
|
+
def test_fetch_url_tool_definition(self) -> None:
|
|
192
|
+
tool = create_fetch_url_tool()
|
|
193
|
+
assert tool.name == "fetch_url"
|
|
194
|
+
assert tool.prompt_snippet is not None
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
# ---------------------------------------------------------------------------
|
|
198
|
+
# activate() entry point
|
|
199
|
+
# ---------------------------------------------------------------------------
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
class TestActivate:
|
|
203
|
+
def test_activate_registers_tools(self) -> None:
|
|
204
|
+
from pi_web_access import activate
|
|
205
|
+
|
|
206
|
+
from pi_agent_core.extensions import ExtensionAPI, ExtensionRegistry
|
|
207
|
+
from pi_agent_core.extensions.types import ExtensionMeta
|
|
208
|
+
|
|
209
|
+
reg = ExtensionRegistry()
|
|
210
|
+
api = ExtensionAPI(registry=reg, meta=ExtensionMeta(name="pi-web-access"))
|
|
211
|
+
activate(api)
|
|
212
|
+
tools = reg.get_tools()
|
|
213
|
+
assert "web_search" in tools
|
|
214
|
+
assert "fetch_url" in tools
|