bub-web-search 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bub_web_search-0.0.1/PKG-INFO +85 -0
- bub_web_search-0.0.1/README.md +73 -0
- bub_web_search-0.0.1/pyproject.toml +26 -0
- bub_web_search-0.0.1/src/bub_web_search/__init__.py +0 -0
- bub_web_search-0.0.1/src/bub_web_search/config.py +74 -0
- bub_web_search-0.0.1/src/bub_web_search/ollama.py +59 -0
- bub_web_search-0.0.1/src/bub_web_search/py.typed +0 -0
- bub_web_search-0.0.1/src/bub_web_search/searxng.py +311 -0
- bub_web_search-0.0.1/src/bub_web_search/tools.py +64 -0
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: bub-web-search
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Web search tools for Bub
|
|
5
|
+
Author: Frost Ming
|
|
6
|
+
Author-email: Frost Ming <me@frostming.com>
|
|
7
|
+
Requires-Dist: aiohttp>=3.13.3
|
|
8
|
+
Requires-Dist: pydantic>=2.0.0
|
|
9
|
+
Requires-Dist: pydantic-settings>=2.10.1
|
|
10
|
+
Requires-Python: >=3.12
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
|
|
13
|
+
# bub-web-search
|
|
14
|
+
|
|
15
|
+
Provider-selectable web search tools for `bub`.
|
|
16
|
+
|
|
17
|
+
## Providers
|
|
18
|
+
|
|
19
|
+
Set `BUB_SEARCH_PROVIDER` to enable exactly one search provider:
|
|
20
|
+
|
|
21
|
+
- `ollama` registers `web.search`
|
|
22
|
+
- `searxng` registers `searxng.search`
|
|
23
|
+
|
|
24
|
+
If the provider is unset or its required configuration is missing, neither tool is
|
|
25
|
+
registered.
|
|
26
|
+
|
|
27
|
+
## Installation
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
bub install bub-web-search
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Ollama
|
|
34
|
+
|
|
35
|
+
Required:
|
|
36
|
+
|
|
37
|
+
- `BUB_SEARCH_PROVIDER=ollama`
|
|
38
|
+
- `BUB_SEARCH_OLLAMA_API_KEY`
|
|
39
|
+
|
|
40
|
+
Optional:
|
|
41
|
+
|
|
42
|
+
- `BUB_SEARCH_OLLAMA_API_BASE`
|
|
43
|
+
- Default: `https://ollama.com/api`
|
|
44
|
+
|
|
45
|
+
The `web.search` tool accepts `query` and `max_results`.
|
|
46
|
+
|
|
47
|
+
## SearXNG
|
|
48
|
+
|
|
49
|
+
Required:
|
|
50
|
+
|
|
51
|
+
- `BUB_SEARCH_PROVIDER=searxng`
|
|
52
|
+
- `BUB_SEARCH_SEARXNG_BASE_URL`
|
|
53
|
+
|
|
54
|
+
Optional:
|
|
55
|
+
|
|
56
|
+
- `BUB_SEARCH_SEARXNG_TIMEOUT_SECONDS`
|
|
57
|
+
- Default: `10`
|
|
58
|
+
- `BUB_SEARCH_SEARXNG_DEFAULT_LANGUAGE`
|
|
59
|
+
- `BUB_SEARCH_SEARXNG_DEFAULT_SAFE_SEARCH`
|
|
60
|
+
- `0` off, `1` moderate, `2` strict
|
|
61
|
+
- Default: `1`
|
|
62
|
+
- `BUB_SEARCH_SEARXNG_USER_AGENT`
|
|
63
|
+
- Default: `bub-web-search/1.0`
|
|
64
|
+
- `BUB_SEARCH_SEARXNG_AUTH_HEADER`
|
|
65
|
+
- `BUB_SEARCH_SEARXNG_AUTH_VALUE`
|
|
66
|
+
|
|
67
|
+
The `searxng.search` tool accepts:
|
|
68
|
+
|
|
69
|
+
- `query`
|
|
70
|
+
- `max_results`
|
|
71
|
+
- `categories`
|
|
72
|
+
- `engines`
|
|
73
|
+
- `language`
|
|
74
|
+
- `time_range`
|
|
75
|
+
- `safe_search`
|
|
76
|
+
|
|
77
|
+
The SearXNG instance must allow JSON responses from its `/search` endpoint.
|
|
78
|
+
|
|
79
|
+
## Migration From bub-searxng-search
|
|
80
|
+
|
|
81
|
+
Replace the package with `bub-web-search`, set
|
|
82
|
+
`BUB_SEARCH_PROVIDER=searxng`, and rename the environment variables:
|
|
83
|
+
|
|
84
|
+
- `BUB_SEARXNG_SEARCH_BASE_URL` to `BUB_SEARCH_SEARXNG_BASE_URL`
|
|
85
|
+
- other `BUB_SEARXNG_SEARCH_*` variables to `BUB_SEARCH_SEARXNG_*`
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# bub-web-search
|
|
2
|
+
|
|
3
|
+
Provider-selectable web search tools for `bub`.
|
|
4
|
+
|
|
5
|
+
## Providers
|
|
6
|
+
|
|
7
|
+
Set `BUB_SEARCH_PROVIDER` to enable exactly one search provider:
|
|
8
|
+
|
|
9
|
+
- `ollama` registers `web.search`
|
|
10
|
+
- `searxng` registers `searxng.search`
|
|
11
|
+
|
|
12
|
+
If the provider is unset or its required configuration is missing, neither tool is
|
|
13
|
+
registered.
|
|
14
|
+
|
|
15
|
+
## Installation
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
bub install bub-web-search
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
## Ollama
|
|
22
|
+
|
|
23
|
+
Required:
|
|
24
|
+
|
|
25
|
+
- `BUB_SEARCH_PROVIDER=ollama`
|
|
26
|
+
- `BUB_SEARCH_OLLAMA_API_KEY`
|
|
27
|
+
|
|
28
|
+
Optional:
|
|
29
|
+
|
|
30
|
+
- `BUB_SEARCH_OLLAMA_API_BASE`
|
|
31
|
+
- Default: `https://ollama.com/api`
|
|
32
|
+
|
|
33
|
+
The `web.search` tool accepts `query` and `max_results`.
|
|
34
|
+
|
|
35
|
+
## SearXNG
|
|
36
|
+
|
|
37
|
+
Required:
|
|
38
|
+
|
|
39
|
+
- `BUB_SEARCH_PROVIDER=searxng`
|
|
40
|
+
- `BUB_SEARCH_SEARXNG_BASE_URL`
|
|
41
|
+
|
|
42
|
+
Optional:
|
|
43
|
+
|
|
44
|
+
- `BUB_SEARCH_SEARXNG_TIMEOUT_SECONDS`
|
|
45
|
+
- Default: `10`
|
|
46
|
+
- `BUB_SEARCH_SEARXNG_DEFAULT_LANGUAGE`
|
|
47
|
+
- `BUB_SEARCH_SEARXNG_DEFAULT_SAFE_SEARCH`
|
|
48
|
+
- `0` off, `1` moderate, `2` strict
|
|
49
|
+
- Default: `1`
|
|
50
|
+
- `BUB_SEARCH_SEARXNG_USER_AGENT`
|
|
51
|
+
- Default: `bub-web-search/1.0`
|
|
52
|
+
- `BUB_SEARCH_SEARXNG_AUTH_HEADER`
|
|
53
|
+
- `BUB_SEARCH_SEARXNG_AUTH_VALUE`
|
|
54
|
+
|
|
55
|
+
The `searxng.search` tool accepts:
|
|
56
|
+
|
|
57
|
+
- `query`
|
|
58
|
+
- `max_results`
|
|
59
|
+
- `categories`
|
|
60
|
+
- `engines`
|
|
61
|
+
- `language`
|
|
62
|
+
- `time_range`
|
|
63
|
+
- `safe_search`
|
|
64
|
+
|
|
65
|
+
The SearXNG instance must allow JSON responses from its `/search` endpoint.
|
|
66
|
+
|
|
67
|
+
## Migration From bub-searxng-search
|
|
68
|
+
|
|
69
|
+
Replace the package with `bub-web-search`, set
|
|
70
|
+
`BUB_SEARCH_PROVIDER=searxng`, and rename the environment variables:
|
|
71
|
+
|
|
72
|
+
- `BUB_SEARXNG_SEARCH_BASE_URL` to `BUB_SEARCH_SEARXNG_BASE_URL`
|
|
73
|
+
- other `BUB_SEARXNG_SEARCH_*` variables to `BUB_SEARCH_SEARXNG_*`
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "bub-web-search"
|
|
3
|
+
version = "0.0.1"
|
|
4
|
+
description = "Web search tools for Bub"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
authors = [
|
|
7
|
+
{ name = "Frost Ming", email = "me@frostming.com" }
|
|
8
|
+
]
|
|
9
|
+
requires-python = ">=3.12"
|
|
10
|
+
dependencies = [
|
|
11
|
+
"aiohttp>=3.13.3",
|
|
12
|
+
"pydantic>=2.0.0",
|
|
13
|
+
"pydantic-settings>=2.10.1",
|
|
14
|
+
]
|
|
15
|
+
|
|
16
|
+
[project.entry-points.bub]
|
|
17
|
+
web-search = "bub_web_search.tools"
|
|
18
|
+
|
|
19
|
+
[build-system]
|
|
20
|
+
requires = ["uv_build>=0.9.7,<0.10.0"]
|
|
21
|
+
build-backend = "uv_build"
|
|
22
|
+
|
|
23
|
+
[dependency-groups]
|
|
24
|
+
dev = [
|
|
25
|
+
"pytest>=9.0.3",
|
|
26
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
from typing import Literal
|
|
2
|
+
|
|
3
|
+
import bub
|
|
4
|
+
from pydantic_settings import SettingsConfigDict
|
|
5
|
+
|
|
6
|
+
DEFAULT_OLLAMA_API_BASE = "https://ollama.com/api"
|
|
7
|
+
DEFAULT_SEARXNG_TIMEOUT_SECONDS = 10
|
|
8
|
+
DEFAULT_SEARXNG_SAFE_SEARCH = 1
|
|
9
|
+
DEFAULT_SEARXNG_USER_AGENT = "bub-web-search/1.0"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@bub.config(name="web-search")
|
|
13
|
+
class WebSearchSettings(bub.Settings):
|
|
14
|
+
model_config = SettingsConfigDict(
|
|
15
|
+
env_prefix="BUB_SEARCH_",
|
|
16
|
+
env_file=".env",
|
|
17
|
+
env_file_encoding="utf-8",
|
|
18
|
+
extra="ignore",
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
provider: Literal["ollama", "searxng"] | None = None
|
|
22
|
+
|
|
23
|
+
ollama_api_key: str | None = None
|
|
24
|
+
ollama_api_base: str = DEFAULT_OLLAMA_API_BASE
|
|
25
|
+
|
|
26
|
+
searxng_base_url: str | None = None
|
|
27
|
+
searxng_timeout_seconds: int = DEFAULT_SEARXNG_TIMEOUT_SECONDS
|
|
28
|
+
searxng_default_language: str | None = None
|
|
29
|
+
searxng_default_safe_search: int = DEFAULT_SEARXNG_SAFE_SEARCH
|
|
30
|
+
searxng_user_agent: str = DEFAULT_SEARXNG_USER_AGENT
|
|
31
|
+
searxng_auth_header: str | None = None
|
|
32
|
+
searxng_auth_value: str | None = None
|
|
33
|
+
|
|
34
|
+
@property
|
|
35
|
+
def resolved_provider(self) -> Literal["ollama", "searxng"] | None:
|
|
36
|
+
if self.provider is not None:
|
|
37
|
+
return self.provider
|
|
38
|
+
if self.ollama_api_key:
|
|
39
|
+
return "ollama"
|
|
40
|
+
if self.searxng_base_url:
|
|
41
|
+
return "searxng"
|
|
42
|
+
return None
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def resolved_searxng_base_url(self) -> str | None:
|
|
46
|
+
if self.searxng_base_url is None:
|
|
47
|
+
return None
|
|
48
|
+
base_url = self.searxng_base_url.strip().rstrip("/")
|
|
49
|
+
return base_url or None
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def resolved_searxng_timeout_seconds(self) -> int:
|
|
53
|
+
return max(1, self.searxng_timeout_seconds)
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def resolved_searxng_default_safe_search(self) -> int:
|
|
57
|
+
if self.searxng_default_safe_search in {0, 1, 2}:
|
|
58
|
+
return self.searxng_default_safe_search
|
|
59
|
+
return DEFAULT_SEARXNG_SAFE_SEARCH
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def resolved_searxng_user_agent(self) -> str:
|
|
63
|
+
user_agent = self.searxng_user_agent.strip()
|
|
64
|
+
return user_agent or DEFAULT_SEARXNG_USER_AGENT
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def resolved_searxng_auth_headers(self) -> dict[str, str]:
|
|
68
|
+
if self.searxng_auth_header is None or self.searxng_auth_value is None:
|
|
69
|
+
return {}
|
|
70
|
+
header_name = self.searxng_auth_header.strip()
|
|
71
|
+
header_value = self.searxng_auth_value.strip()
|
|
72
|
+
if not header_name or not header_value:
|
|
73
|
+
return {}
|
|
74
|
+
return {header_name: header_value}
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
import json
|
|
2
|
+
|
|
3
|
+
from bub_web_search.config import WebSearchSettings
|
|
4
|
+
|
|
5
|
+
WEB_USER_AGENT = "bub-web-search/1.0"
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
async def search(query: str, max_results: int, settings: WebSearchSettings) -> str:
|
|
9
|
+
import aiohttp
|
|
10
|
+
|
|
11
|
+
api_key = settings.ollama_api_key
|
|
12
|
+
if not api_key:
|
|
13
|
+
return "error: ollama api key is not configured"
|
|
14
|
+
|
|
15
|
+
api_base = settings.ollama_api_base.rstrip("/")
|
|
16
|
+
if not api_base:
|
|
17
|
+
return "error: invalid ollama api base url"
|
|
18
|
+
|
|
19
|
+
endpoint = f"{api_base}/web_search"
|
|
20
|
+
payload = {"query": query, "max_results": max_results}
|
|
21
|
+
try:
|
|
22
|
+
async with (
|
|
23
|
+
aiohttp.ClientSession(timeout=aiohttp.ClientTimeout(total=20)) as session,
|
|
24
|
+
session.post(
|
|
25
|
+
endpoint,
|
|
26
|
+
json=payload,
|
|
27
|
+
headers={
|
|
28
|
+
"Content-Type": "application/json",
|
|
29
|
+
"Authorization": f"Bearer {api_key}",
|
|
30
|
+
"User-Agent": WEB_USER_AGENT,
|
|
31
|
+
},
|
|
32
|
+
) as response,
|
|
33
|
+
):
|
|
34
|
+
data = await response.json()
|
|
35
|
+
except aiohttp.ClientError as exc:
|
|
36
|
+
return f"HTTP error: {exc!s}"
|
|
37
|
+
except json.JSONDecodeError as exc:
|
|
38
|
+
return f"error: invalid json response: {exc!s}"
|
|
39
|
+
|
|
40
|
+
results = data.get("results")
|
|
41
|
+
if not isinstance(results, list) or not results:
|
|
42
|
+
return "none"
|
|
43
|
+
return _format_search_results(results)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _format_search_results(results: list[object]) -> str:
|
|
47
|
+
lines: list[str] = []
|
|
48
|
+
for idx, item in enumerate(results, start=1):
|
|
49
|
+
if not isinstance(item, dict):
|
|
50
|
+
continue
|
|
51
|
+
title = str(item.get("title") or "(untitled)")
|
|
52
|
+
url = str(item.get("url") or "")
|
|
53
|
+
content = str(item.get("content") or "")
|
|
54
|
+
lines.append(f"{idx}. {title}")
|
|
55
|
+
if url:
|
|
56
|
+
lines.append(f" {url}")
|
|
57
|
+
if content:
|
|
58
|
+
lines.append(f" {content}")
|
|
59
|
+
return "\n".join(lines) if lines else "none"
|
|
File without changes
|
|
@@ -0,0 +1,311 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from collections.abc import Iterable
|
|
6
|
+
from json import JSONDecodeError
|
|
7
|
+
from typing import Any, Literal
|
|
8
|
+
|
|
9
|
+
from pydantic import BaseModel, Field, field_validator
|
|
10
|
+
|
|
11
|
+
from bub_web_search.config import WebSearchSettings
|
|
12
|
+
|
|
13
|
+
MAX_RESULTS_LIMIT = 10
|
|
14
|
+
MAX_SNIPPET_CHARS = 280
|
|
15
|
+
MAX_TITLE_CHARS = 160
|
|
16
|
+
_WHITESPACE_RE = re.compile(r"\s+")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class SearXNGSearchInput(BaseModel):
|
|
20
|
+
query: str = Field(..., description="The search query string.")
|
|
21
|
+
max_results: int = Field(
|
|
22
|
+
5,
|
|
23
|
+
ge=1,
|
|
24
|
+
le=MAX_RESULTS_LIMIT,
|
|
25
|
+
description="Maximum number of search results to return.",
|
|
26
|
+
)
|
|
27
|
+
categories: list[str] | None = Field(
|
|
28
|
+
None,
|
|
29
|
+
description="Optional list of SearXNG categories, such as general, news, or science.",
|
|
30
|
+
)
|
|
31
|
+
engines: list[str] | None = Field(
|
|
32
|
+
None, description="Optional list of SearXNG engine names to limit the search."
|
|
33
|
+
)
|
|
34
|
+
language: str | None = Field(
|
|
35
|
+
None, description="Optional language code, such as en-US or zh-CN."
|
|
36
|
+
)
|
|
37
|
+
time_range: Literal["day", "month", "year"] | None = Field(
|
|
38
|
+
None, description="Optional SearXNG time filter."
|
|
39
|
+
)
|
|
40
|
+
safe_search: int | None = Field(
|
|
41
|
+
None,
|
|
42
|
+
ge=0,
|
|
43
|
+
le=2,
|
|
44
|
+
description="Optional safe search level: 0 off, 1 moderate, 2 strict.",
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
@field_validator("query")
|
|
48
|
+
@classmethod
|
|
49
|
+
def validate_query(cls, value: str) -> str:
|
|
50
|
+
query = value.strip()
|
|
51
|
+
if not query:
|
|
52
|
+
raise ValueError("query must not be blank")
|
|
53
|
+
return query
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
async def search(*, param: SearXNGSearchInput, settings: WebSearchSettings) -> str:
|
|
57
|
+
import aiohttp
|
|
58
|
+
|
|
59
|
+
base_url = settings.resolved_searxng_base_url
|
|
60
|
+
if base_url is None:
|
|
61
|
+
return "error: searxng base url is not configured"
|
|
62
|
+
|
|
63
|
+
endpoint = f"{base_url}/search"
|
|
64
|
+
params = _build_request_params(param=param, settings=settings)
|
|
65
|
+
headers = {
|
|
66
|
+
"Accept": "application/json",
|
|
67
|
+
"User-Agent": settings.resolved_searxng_user_agent,
|
|
68
|
+
**settings.resolved_searxng_auth_headers,
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
try:
|
|
72
|
+
async with (
|
|
73
|
+
aiohttp.ClientSession(
|
|
74
|
+
headers=headers,
|
|
75
|
+
timeout=aiohttp.ClientTimeout(
|
|
76
|
+
total=settings.resolved_searxng_timeout_seconds
|
|
77
|
+
),
|
|
78
|
+
) as session,
|
|
79
|
+
session.get(endpoint, params=params) as response,
|
|
80
|
+
):
|
|
81
|
+
body = await response.text()
|
|
82
|
+
if response.status >= 400:
|
|
83
|
+
detail = (
|
|
84
|
+
_compact_text(body, limit=MAX_SNIPPET_CHARS) or "request failed"
|
|
85
|
+
)
|
|
86
|
+
return f"HTTP {response.status}: {detail}"
|
|
87
|
+
except aiohttp.ClientError as exc:
|
|
88
|
+
return f"HTTP error: {exc!s}"
|
|
89
|
+
except TimeoutError:
|
|
90
|
+
return (
|
|
91
|
+
"error: request timed out after "
|
|
92
|
+
f"{settings.resolved_searxng_timeout_seconds} seconds"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
payload = json.loads(body)
|
|
97
|
+
except JSONDecodeError as exc:
|
|
98
|
+
return f"error: invalid json response: {exc!s}"
|
|
99
|
+
if not isinstance(payload, dict):
|
|
100
|
+
return "error: invalid json response: expected a top-level object"
|
|
101
|
+
return _format_search_response(payload, max_results=param.max_results)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _build_request_params(
|
|
105
|
+
*, param: SearXNGSearchInput, settings: WebSearchSettings
|
|
106
|
+
) -> dict[str, str | int]:
|
|
107
|
+
request_params: dict[str, str | int] = {
|
|
108
|
+
"q": param.query,
|
|
109
|
+
"format": "json",
|
|
110
|
+
"safesearch": (
|
|
111
|
+
param.safe_search
|
|
112
|
+
if param.safe_search is not None
|
|
113
|
+
else settings.resolved_searxng_default_safe_search
|
|
114
|
+
),
|
|
115
|
+
}
|
|
116
|
+
if language := _clean_value(param.language) or _clean_value(
|
|
117
|
+
settings.searxng_default_language
|
|
118
|
+
):
|
|
119
|
+
request_params["language"] = language
|
|
120
|
+
if categories := _join_csv(param.categories):
|
|
121
|
+
request_params["categories"] = categories
|
|
122
|
+
if engines := _join_csv(param.engines):
|
|
123
|
+
request_params["engines"] = engines
|
|
124
|
+
if param.time_range is not None:
|
|
125
|
+
request_params["time_range"] = param.time_range
|
|
126
|
+
return request_params
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _format_search_response(payload: dict[str, Any], *, max_results: int) -> str:
|
|
130
|
+
lines: list[str] = []
|
|
131
|
+
|
|
132
|
+
answer_lines = _format_answer_lines(payload.get("answers"))
|
|
133
|
+
if answer_lines:
|
|
134
|
+
lines.extend(["Answers:", *answer_lines])
|
|
135
|
+
|
|
136
|
+
suggestion_lines = _format_suggestion_lines(payload.get("suggestions"))
|
|
137
|
+
if suggestion_lines:
|
|
138
|
+
if lines:
|
|
139
|
+
lines.append("")
|
|
140
|
+
lines.extend(["Suggestions:", *suggestion_lines])
|
|
141
|
+
|
|
142
|
+
infobox_lines = _format_infobox_lines(payload.get("infoboxes"))
|
|
143
|
+
if infobox_lines:
|
|
144
|
+
if lines:
|
|
145
|
+
lines.append("")
|
|
146
|
+
lines.extend(["Infoboxes:", *infobox_lines])
|
|
147
|
+
|
|
148
|
+
result_blocks = _format_result_blocks(
|
|
149
|
+
payload.get("results"), max_results=max_results
|
|
150
|
+
)
|
|
151
|
+
if result_blocks:
|
|
152
|
+
if lines:
|
|
153
|
+
lines.append("")
|
|
154
|
+
lines.extend(result_blocks)
|
|
155
|
+
|
|
156
|
+
return "\n".join(lines) if lines else "none"
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _format_answer_lines(raw_answers: object) -> list[str]:
|
|
160
|
+
if not isinstance(raw_answers, list):
|
|
161
|
+
return []
|
|
162
|
+
lines: list[str] = []
|
|
163
|
+
for item in raw_answers:
|
|
164
|
+
text = _stringify_answer(item)
|
|
165
|
+
if text:
|
|
166
|
+
lines.append(f"- {text}")
|
|
167
|
+
return lines
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _stringify_answer(value: object) -> str:
|
|
171
|
+
if isinstance(value, str):
|
|
172
|
+
return _compact_text(value, limit=MAX_SNIPPET_CHARS)
|
|
173
|
+
if isinstance(value, dict):
|
|
174
|
+
text = _first_non_empty(
|
|
175
|
+
value.get("answer"),
|
|
176
|
+
value.get("content"),
|
|
177
|
+
value.get("text"),
|
|
178
|
+
value.get("title"),
|
|
179
|
+
)
|
|
180
|
+
return _compact_text(text, limit=MAX_SNIPPET_CHARS)
|
|
181
|
+
return ""
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _format_suggestion_lines(raw_suggestions: object) -> list[str]:
|
|
185
|
+
if not isinstance(raw_suggestions, list):
|
|
186
|
+
return []
|
|
187
|
+
lines: list[str] = []
|
|
188
|
+
for item in raw_suggestions:
|
|
189
|
+
if isinstance(item, str):
|
|
190
|
+
suggestion = _compact_text(item, limit=MAX_TITLE_CHARS)
|
|
191
|
+
if suggestion:
|
|
192
|
+
lines.append(f"- {suggestion}")
|
|
193
|
+
return lines
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _format_infobox_lines(raw_infoboxes: object) -> list[str]:
|
|
197
|
+
if not isinstance(raw_infoboxes, list):
|
|
198
|
+
return []
|
|
199
|
+
lines: list[str] = []
|
|
200
|
+
for item in raw_infoboxes:
|
|
201
|
+
if not isinstance(item, dict):
|
|
202
|
+
continue
|
|
203
|
+
title = _compact_text(
|
|
204
|
+
_first_non_empty(
|
|
205
|
+
item.get("infobox"), item.get("id"), item.get("title"), "(untitled)"
|
|
206
|
+
),
|
|
207
|
+
limit=MAX_TITLE_CHARS,
|
|
208
|
+
)
|
|
209
|
+
content = _compact_text(
|
|
210
|
+
_first_non_empty(
|
|
211
|
+
item.get("content"), item.get("description"), item.get("title")
|
|
212
|
+
),
|
|
213
|
+
limit=MAX_SNIPPET_CHARS,
|
|
214
|
+
)
|
|
215
|
+
lines.append(f"- {title}")
|
|
216
|
+
if url := _extract_url(item):
|
|
217
|
+
lines.append(f" {url}")
|
|
218
|
+
if content and content != title:
|
|
219
|
+
lines.append(f" {content}")
|
|
220
|
+
return lines
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _format_result_blocks(raw_results: object, *, max_results: int) -> list[str]:
|
|
224
|
+
if not isinstance(raw_results, list):
|
|
225
|
+
return []
|
|
226
|
+
|
|
227
|
+
lines: list[str] = []
|
|
228
|
+
rendered = 0
|
|
229
|
+
for item in raw_results:
|
|
230
|
+
if not isinstance(item, dict):
|
|
231
|
+
continue
|
|
232
|
+
title = _compact_text(
|
|
233
|
+
_first_non_empty(
|
|
234
|
+
item.get("title"), item.get("content"), item.get("url"), "(untitled)"
|
|
235
|
+
),
|
|
236
|
+
limit=MAX_TITLE_CHARS,
|
|
237
|
+
)
|
|
238
|
+
lines.append(f"{rendered + 1}. {title}")
|
|
239
|
+
if url := _extract_url(item):
|
|
240
|
+
lines.append(f" {url}")
|
|
241
|
+
snippet = _compact_text(
|
|
242
|
+
_first_non_empty(
|
|
243
|
+
item.get("content"), item.get("snippet"), item.get("description")
|
|
244
|
+
),
|
|
245
|
+
limit=MAX_SNIPPET_CHARS,
|
|
246
|
+
)
|
|
247
|
+
if snippet and snippet != title:
|
|
248
|
+
lines.append(f" {snippet}")
|
|
249
|
+
|
|
250
|
+
metadata: list[str] = []
|
|
251
|
+
if engine := _clean_value(item.get("engine")):
|
|
252
|
+
metadata.append(engine)
|
|
253
|
+
if category := _clean_value(item.get("category")):
|
|
254
|
+
metadata.append(f"[{category}]")
|
|
255
|
+
if published := _clean_value(item.get("publishedDate")):
|
|
256
|
+
metadata.append(published)
|
|
257
|
+
if metadata:
|
|
258
|
+
lines.append(f" source: {' '.join(metadata)}")
|
|
259
|
+
|
|
260
|
+
rendered += 1
|
|
261
|
+
if rendered >= max_results:
|
|
262
|
+
break
|
|
263
|
+
return lines if rendered else []
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _extract_url(item: dict[str, Any]) -> str:
|
|
267
|
+
if url := _clean_value(item.get("url")):
|
|
268
|
+
return url
|
|
269
|
+
raw_urls = item.get("urls")
|
|
270
|
+
if not isinstance(raw_urls, list):
|
|
271
|
+
return ""
|
|
272
|
+
for candidate in raw_urls:
|
|
273
|
+
if isinstance(candidate, str):
|
|
274
|
+
if url := _clean_value(candidate):
|
|
275
|
+
return url
|
|
276
|
+
elif isinstance(candidate, dict):
|
|
277
|
+
if url := _clean_value(candidate.get("url")):
|
|
278
|
+
return url
|
|
279
|
+
return ""
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _join_csv(values: Iterable[object] | None) -> str:
|
|
283
|
+
if values is None:
|
|
284
|
+
return ""
|
|
285
|
+
parts = [part for value in values if (part := _clean_value(value))]
|
|
286
|
+
return ",".join(parts)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _clean_value(value: object) -> str:
|
|
290
|
+
if value is None:
|
|
291
|
+
return ""
|
|
292
|
+
if not isinstance(value, str):
|
|
293
|
+
value = str(value)
|
|
294
|
+
return value.strip()
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _compact_text(value: object, *, limit: int) -> str:
|
|
298
|
+
text = _clean_value(value)
|
|
299
|
+
if not text:
|
|
300
|
+
return ""
|
|
301
|
+
compact = _WHITESPACE_RE.sub(" ", text)
|
|
302
|
+
if len(compact) <= limit:
|
|
303
|
+
return compact
|
|
304
|
+
return compact[: limit - 3].rstrip() + "..."
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _first_non_empty(*values: object) -> str:
|
|
308
|
+
for value in values:
|
|
309
|
+
if cleaned := _clean_value(value):
|
|
310
|
+
return cleaned
|
|
311
|
+
return ""
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Callable
|
|
4
|
+
from typing import TYPE_CHECKING
|
|
5
|
+
|
|
6
|
+
import bub
|
|
7
|
+
from bub import tool
|
|
8
|
+
from bub.tools import REGISTRY
|
|
9
|
+
|
|
10
|
+
from bub_web_search import ollama, searxng
|
|
11
|
+
from bub_web_search.config import WebSearchSettings
|
|
12
|
+
|
|
13
|
+
if TYPE_CHECKING:
|
|
14
|
+
from republic import Tool
|
|
15
|
+
|
|
16
|
+
SEARCH_TOOL_NAME = "web.search"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def register_tools(
|
|
20
|
+
settings_factory: Callable[[], WebSearchSettings] = lambda: bub.ensure_config(
|
|
21
|
+
WebSearchSettings
|
|
22
|
+
),
|
|
23
|
+
) -> Tool | None:
|
|
24
|
+
REGISTRY.pop(SEARCH_TOOL_NAME, None)
|
|
25
|
+
|
|
26
|
+
settings = settings_factory()
|
|
27
|
+
provider = settings.resolved_provider
|
|
28
|
+
if provider == "ollama":
|
|
29
|
+
return _register_ollama_tool(settings)
|
|
30
|
+
elif provider == "searxng":
|
|
31
|
+
return _register_searxng_tool(settings)
|
|
32
|
+
return None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _register_ollama_tool(settings: WebSearchSettings) -> Tool | None:
|
|
36
|
+
if not settings.ollama_api_key:
|
|
37
|
+
return None
|
|
38
|
+
|
|
39
|
+
@tool(name=SEARCH_TOOL_NAME)
|
|
40
|
+
async def web_search_ollama(query: str, max_results: int = 10) -> str:
|
|
41
|
+
"""Search the web with Ollama and return concise results."""
|
|
42
|
+
return await ollama.search(
|
|
43
|
+
query=query, max_results=max_results, settings=settings
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
return web_search_ollama
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _register_searxng_tool(settings: WebSearchSettings) -> Tool | None:
|
|
50
|
+
if settings.resolved_searxng_base_url is None:
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
@tool(
|
|
54
|
+
name=SEARCH_TOOL_NAME,
|
|
55
|
+
model=searxng.SearXNGSearchInput,
|
|
56
|
+
description="Search a configured SearXNG instance and return concise web results.",
|
|
57
|
+
)
|
|
58
|
+
async def searxng_search(param: searxng.SearXNGSearchInput) -> str:
|
|
59
|
+
return await searxng.search(param=param, settings=settings)
|
|
60
|
+
|
|
61
|
+
return searxng_search
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
register_tools()
|