bathys 0.6.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bathys/__init__.py +8 -0
- bathys/agents/HARNESS-DROPIN.md +50 -0
- bathys/agents/bathys-researcher.md +91 -0
- bathys/agents/skills/bathys-deep-dive/SKILL.md +55 -0
- bathys/agents/skills/bathys-source-audit/SKILL.md +48 -0
- bathys/batch.py +86 -0
- bathys/cache.py +47 -0
- bathys/compose.yaml +22 -0
- bathys/config.py +71 -0
- bathys/core.py +425 -0
- bathys/crawler.py +104 -0
- bathys/distill.py +135 -0
- bathys/doctor.py +139 -0
- bathys/installer.py +424 -0
- bathys/searx.py +149 -0
- bathys/searxng-settings.yml +13 -0
- bathys/server.py +399 -0
- bathys/services.py +263 -0
- bathys-0.6.1.dist-info/METADATA +147 -0
- bathys-0.6.1.dist-info/RECORD +23 -0
- bathys-0.6.1.dist-info/WHEEL +4 -0
- bathys-0.6.1.dist-info/entry_points.txt +4 -0
- bathys-0.6.1.dist-info/licenses/LICENSE +21 -0
bathys/core.py
ADDED
|
@@ -0,0 +1,425 @@
|
|
|
1
|
+
"""Core pipeline: search -> dive -> distill, cached. Framework-free on purpose,
|
|
2
|
+
so scripts and tests can drive it without MCP framing."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import asyncio
|
|
7
|
+
import time
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
|
|
10
|
+
import httpx
|
|
11
|
+
|
|
12
|
+
from . import distill, searx
|
|
13
|
+
from .cache import Cache
|
|
14
|
+
from .config import Config
|
|
15
|
+
from .crawler import Crawler, Page
|
|
16
|
+
from .services import ensure_running, stop_native
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class RobotsRefusal(Exception):
|
|
20
|
+
"""Target host disallows this path for crawlers in its robots.txt."""
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _ch(chars: int) -> str:
|
|
24
|
+
"""Honest character count — one scale for every footer field (contract K1 fix)."""
|
|
25
|
+
return str(chars)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# Fallback engine sets for empty-result retries (F-101). Ordered by independence
|
|
29
|
+
# from big-scraping backends; rotation always happens WITH backoff pauses.
|
|
30
|
+
_RETRY_ENGINE_SETS: list[str | None] = [
|
|
31
|
+
None, # as requested / instance defaults
|
|
32
|
+
"duckduckgo,bing,brave,startpage",
|
|
33
|
+
"wikipedia,duckduckgo,mojeek,bing",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class ReadResult:
|
|
39
|
+
page: Page
|
|
40
|
+
distilled: str
|
|
41
|
+
cache_hit: bool
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class Engine:
|
|
45
|
+
"""Everything the tools need; one instance per server process."""
|
|
46
|
+
|
|
47
|
+
def __init__(self, cfg: Config) -> None:
|
|
48
|
+
self.cfg = cfg
|
|
49
|
+
self.cache = Cache(cfg.cache_dir / "cache.db")
|
|
50
|
+
self.crawler = Crawler(cfg)
|
|
51
|
+
self.http: httpx.AsyncClient | None = None
|
|
52
|
+
self._searx_ok = False
|
|
53
|
+
self._search_ts = 0.0
|
|
54
|
+
self._pace_lock = asyncio.Lock()
|
|
55
|
+
self._engine_fails: dict[str, int] = {}
|
|
56
|
+
self._dive_sem = asyncio.Semaphore(cfg.dive_concurrency)
|
|
57
|
+
self._robots: dict[str, "object | None"] = {} # host -> RobotFileParser | None (None = allow)
|
|
58
|
+
self._metrics_path = cfg.data_dir / "metrics.jsonl"
|
|
59
|
+
try:
|
|
60
|
+
cfg.data_dir.mkdir(parents=True, exist_ok=True)
|
|
61
|
+
except OSError:
|
|
62
|
+
pass
|
|
63
|
+
|
|
64
|
+
# ------------------------------------------------------------ metrics ---
|
|
65
|
+
|
|
66
|
+
def _log_metrics(self, tool: str, *, cache: str | None = None, chars_in: int = 0,
|
|
67
|
+
chars_out: int = 0, secs: float = 0.0, ok: bool = True,
|
|
68
|
+
error: str | None = None, url: str | None = None,
|
|
69
|
+
q: str | None = None) -> None:
|
|
70
|
+
"""F-304: local JSONL journal (schema — docs/operations/metrics.md §3.2).
|
|
71
|
+
Metrics must never break a tool: swallow everything."""
|
|
72
|
+
if not self.cfg.metrics:
|
|
73
|
+
return
|
|
74
|
+
import datetime
|
|
75
|
+
import hashlib
|
|
76
|
+
|
|
77
|
+
def h(v: str | None) -> str | None:
|
|
78
|
+
return hashlib.sha256(v.encode()).hexdigest()[:12] if v else None
|
|
79
|
+
|
|
80
|
+
event = {
|
|
81
|
+
"ts": datetime.datetime.now(datetime.timezone.utc).isoformat(timespec="seconds"),
|
|
82
|
+
"tool": tool,
|
|
83
|
+
"cache": cache,
|
|
84
|
+
"chars_in": chars_in,
|
|
85
|
+
"chars_out": chars_out,
|
|
86
|
+
"secs": secs,
|
|
87
|
+
"ok": ok,
|
|
88
|
+
"error_class": error,
|
|
89
|
+
"url_hash": h(url),
|
|
90
|
+
"q_hash": h(q),
|
|
91
|
+
}
|
|
92
|
+
try:
|
|
93
|
+
import json as _json
|
|
94
|
+
with open(self._metrics_path, "a", encoding="utf-8") as f:
|
|
95
|
+
f.write(_json.dumps(event, ensure_ascii=False, separators=(",", ":")) + "\n")
|
|
96
|
+
except OSError:
|
|
97
|
+
pass
|
|
98
|
+
|
|
99
|
+
# ------------------------------------------------------------- robots ---
|
|
100
|
+
|
|
101
|
+
async def _robots_allowed(self, url: str) -> bool:
|
|
102
|
+
"""F-303: respect robots.txt for our direct page dives; fail-open on
|
|
103
|
+
missing/unreachable robots (standard robots semantics). Search itself
|
|
104
|
+
is not crawling — engines fetch, we only query them."""
|
|
105
|
+
if not self.cfg.respect_robots:
|
|
106
|
+
return True
|
|
107
|
+
from urllib import robotparser
|
|
108
|
+
from urllib.parse import urlsplit
|
|
109
|
+
|
|
110
|
+
parts = urlsplit(url)
|
|
111
|
+
host = (parts.scheme, parts.netloc)
|
|
112
|
+
if host in self._robots:
|
|
113
|
+
rp = self._robots[host]
|
|
114
|
+
else:
|
|
115
|
+
rp = None
|
|
116
|
+
if self.http is not None and parts.netloc:
|
|
117
|
+
try:
|
|
118
|
+
resp = await self.http.get(
|
|
119
|
+
f"{parts.scheme}://{parts.netloc}/robots.txt", timeout=5.0)
|
|
120
|
+
if resp.status_code == 200:
|
|
121
|
+
parser = robotparser.RobotFileParser()
|
|
122
|
+
parser.parse(resp.text.splitlines())
|
|
123
|
+
rp = parser
|
|
124
|
+
except (httpx.HTTPError, ValueError):
|
|
125
|
+
rp = None # unreachable robots -> allow
|
|
126
|
+
self._robots[host] = rp
|
|
127
|
+
if rp is None:
|
|
128
|
+
return True
|
|
129
|
+
return rp.can_fetch("*", parts.path or "/")
|
|
130
|
+
|
|
131
|
+
async def start(self) -> None:
|
|
132
|
+
self.http = httpx.AsyncClient(follow_redirects=True, timeout=self.cfg.search_timeout)
|
|
133
|
+
|
|
134
|
+
async def stop(self) -> None:
|
|
135
|
+
await self.crawler.stop()
|
|
136
|
+
await stop_native()
|
|
137
|
+
if self.http:
|
|
138
|
+
await self.http.aclose()
|
|
139
|
+
|
|
140
|
+
# ------------------------------------------------------------ search ----
|
|
141
|
+
|
|
142
|
+
async def _polite_pace(self) -> None:
|
|
143
|
+
"""F-301: keep a configurable minimum interval between backend searches."""
|
|
144
|
+
async with self._pace_lock:
|
|
145
|
+
wait = self.cfg.search_min_interval - (time.monotonic() - self._search_ts)
|
|
146
|
+
if wait > 0:
|
|
147
|
+
await asyncio.sleep(wait)
|
|
148
|
+
self._search_ts = time.monotonic()
|
|
149
|
+
|
|
150
|
+
def _record_health(self, outcome: searx.SearchOutcome) -> None:
|
|
151
|
+
"""F-102: consecutive-failure streak per engine; any hit resets it."""
|
|
152
|
+
for e in outcome.unresponsive:
|
|
153
|
+
name = e.split(":")[0]
|
|
154
|
+
self._engine_fails[name] = self._engine_fails.get(name, 0) + 1
|
|
155
|
+
for h in outcome.hits:
|
|
156
|
+
for name in h.engines:
|
|
157
|
+
self._engine_fails.pop(name, None)
|
|
158
|
+
|
|
159
|
+
def _healthy_fallback(self, engine_set: str) -> str:
|
|
160
|
+
bad = {e for e, c in self._engine_fails.items() if c >= 3}
|
|
161
|
+
if not bad:
|
|
162
|
+
return engine_set
|
|
163
|
+
kept = [e for e in engine_set.split(",") if e.strip() and e.strip() not in bad]
|
|
164
|
+
return ",".join(kept) if kept else engine_set
|
|
165
|
+
|
|
166
|
+
def bad_engines(self) -> list[str]:
|
|
167
|
+
"""Engines with 3+ consecutive empty/unresponsive streaks (for doctor/diagnostics)."""
|
|
168
|
+
return sorted(e for e, c in self._engine_fails.items() if c >= 3)
|
|
169
|
+
|
|
170
|
+
async def _search_outcome(self, query: str, *, max_results: int, category: str | None,
|
|
171
|
+
engines: str | None, language: str | None, time_range: str | None,
|
|
172
|
+
refresh: bool = False):
|
|
173
|
+
max_results = max(1, min(20, max_results))
|
|
174
|
+
if time_range not in (None, "", "day", "week", "month", "year"):
|
|
175
|
+
time_range = None
|
|
176
|
+
ck = Cache.key("search", query, max_results, category, engines, language, time_range)
|
|
177
|
+
if not refresh:
|
|
178
|
+
got, stored = self.cache.get(ck)
|
|
179
|
+
if got:
|
|
180
|
+
return stored, True
|
|
181
|
+
http = await self._backend()
|
|
182
|
+
# F-101: on an empty outcome retry with independent engine sets + backoff.
|
|
183
|
+
plan: list[str | None] = ([engines] if engines else []) + _RETRY_ENGINE_SETS
|
|
184
|
+
seen_sets: set[str] = set()
|
|
185
|
+
attempts = 0
|
|
186
|
+
outcome: searx.SearchOutcome | None = None
|
|
187
|
+
for engine_set in plan:
|
|
188
|
+
key = engine_set or ""
|
|
189
|
+
if key in seen_sets:
|
|
190
|
+
continue
|
|
191
|
+
seen_sets.add(key)
|
|
192
|
+
if attempts > self.cfg.search_retries:
|
|
193
|
+
break
|
|
194
|
+
if engine_set and attempts:
|
|
195
|
+
engine_set = self._healthy_fallback(engine_set)
|
|
196
|
+
await self._polite_pace()
|
|
197
|
+
try:
|
|
198
|
+
outcome = await searx.search(
|
|
199
|
+
self.cfg, http, query,
|
|
200
|
+
categories=category, engines=engine_set, language=language, time_range=time_range,
|
|
201
|
+
)
|
|
202
|
+
except searx.SearxError:
|
|
203
|
+
self._searx_ok = False
|
|
204
|
+
raise
|
|
205
|
+
self._record_health(outcome)
|
|
206
|
+
attempts += 1
|
|
207
|
+
if outcome.hits or outcome.answers:
|
|
208
|
+
break
|
|
209
|
+
if attempts <= self.cfg.search_retries:
|
|
210
|
+
await asyncio.sleep(1.5 * attempts) # growing pause: rotation never without a delay
|
|
211
|
+
assert outcome is not None
|
|
212
|
+
stored = {
|
|
213
|
+
"hits": [
|
|
214
|
+
{"title": h.title, "url": h.url, "snippet": h.snippet, "engines": h.engines,
|
|
215
|
+
"score": h.score, "published": h.published}
|
|
216
|
+
for h in outcome.hits
|
|
217
|
+
],
|
|
218
|
+
"answers": outcome.answers,
|
|
219
|
+
"suggestions": outcome.suggestions,
|
|
220
|
+
"seconds": outcome.seconds,
|
|
221
|
+
"raw_chars": outcome.raw_chars,
|
|
222
|
+
"retries": attempts - 1,
|
|
223
|
+
"unresponsive": outcome.unresponsive,
|
|
224
|
+
}
|
|
225
|
+
# empty outcomes are real answers too — but cache them briefly so a
|
|
226
|
+
# broken minute doesn't shadow an hour of retries
|
|
227
|
+
ttl = self.cfg.search_ttl if (outcome.hits or outcome.answers) else min(self.cfg.search_ttl, 600)
|
|
228
|
+
self.cache.set(ck, stored, ttl)
|
|
229
|
+
return stored, False
|
|
230
|
+
|
|
231
|
+
async def search(self, query: str, *, max_results: int = 8, category: str | None = None,
|
|
232
|
+
engines: str | None = None, language: str | None = None,
|
|
233
|
+
time_range: str | None = None, refresh: bool = False,
|
|
234
|
+
as_json: bool = False) -> str:
|
|
235
|
+
started = time.monotonic()
|
|
236
|
+
try:
|
|
237
|
+
stored, cached = await self._search_outcome(
|
|
238
|
+
query, max_results=max_results, category=category,
|
|
239
|
+
engines=engines, language=language, time_range=time_range, refresh=refresh,
|
|
240
|
+
)
|
|
241
|
+
except Exception as e:
|
|
242
|
+
self._log_metrics("web_search", secs=round(time.monotonic() - started, 1),
|
|
243
|
+
ok=False, error=e.__class__.__name__, q=query)
|
|
244
|
+
raise
|
|
245
|
+
secs = round(time.monotonic() - started, 1)
|
|
246
|
+
hits = stored["hits"][: max(1, min(20, max_results))]
|
|
247
|
+
lines: list[str] = []
|
|
248
|
+
if stored["answers"]:
|
|
249
|
+
lines.append("Answer: " + stored["answers"][0])
|
|
250
|
+
for i, h in enumerate(hits, 1):
|
|
251
|
+
date = f" [{h['published']}]" if h["published"] else ""
|
|
252
|
+
lines.append(f"{i}. {h['title']}\n {h['url']}\n {h['snippet']}{date}")
|
|
253
|
+
if stored["suggestions"]:
|
|
254
|
+
lines.append("Refine: " + ", ".join(stored["suggestions"]))
|
|
255
|
+
body = "\n".join(lines) if lines else self._empty_body(query, stored)
|
|
256
|
+
if as_json:
|
|
257
|
+
# F-204: machine-readable mode — pure JSON, no footer (metrics go to
|
|
258
|
+
# metrics.jsonl; footer would break strict json.loads consumers).
|
|
259
|
+
import json as _json
|
|
260
|
+
payload = {
|
|
261
|
+
"query": query,
|
|
262
|
+
"count": len(hits),
|
|
263
|
+
"hits": [
|
|
264
|
+
{"title": h["title"], "url": h["url"], "snippet": h["snippet"],
|
|
265
|
+
"engines": h["engines"], "published": h["published"] or None}
|
|
266
|
+
for h in hits
|
|
267
|
+
],
|
|
268
|
+
}
|
|
269
|
+
if stored["answers"]:
|
|
270
|
+
payload["answer"] = stored["answers"][0]
|
|
271
|
+
out = _json.dumps(payload, ensure_ascii=False, separators=(",", ":"))
|
|
272
|
+
self._log_metrics("web_search", cache="HIT" if cached else "MISS",
|
|
273
|
+
chars_in=stored["raw_chars"], chars_out=len(out),
|
|
274
|
+
secs=secs, q=query)
|
|
275
|
+
return out
|
|
276
|
+
footer = (
|
|
277
|
+
f"[bathys: {len(hits)} hits · cache HIT · {secs}s · searx json "
|
|
278
|
+
f"{_ch(stored['raw_chars'])} ch → {_ch(len(body))} ch]"
|
|
279
|
+
if cached else
|
|
280
|
+
f"[bathys: {len(hits)} hits · {secs}s · searx json "
|
|
281
|
+
f"{_ch(stored['raw_chars'])} ch → {_ch(len(body))} ch]"
|
|
282
|
+
)
|
|
283
|
+
self._log_metrics("web_search", cache="HIT" if cached else "MISS",
|
|
284
|
+
chars_in=stored["raw_chars"], chars_out=len(body),
|
|
285
|
+
secs=secs, q=query)
|
|
286
|
+
return body + "\n" + footer
|
|
287
|
+
|
|
288
|
+
@staticmethod
|
|
289
|
+
def _empty_body(query: str, stored: dict) -> str:
|
|
290
|
+
body = f"No results for: {query}"
|
|
291
|
+
notes: list[str] = []
|
|
292
|
+
if stored.get("retries"):
|
|
293
|
+
notes.append(f"tried {stored['retries'] + 1} engine sets")
|
|
294
|
+
if stored.get("unresponsive"):
|
|
295
|
+
notes.append("unresponsive: " + ", ".join(stored["unresponsive"][:5]))
|
|
296
|
+
return body + (" · " + " · ".join(notes) if notes else "")
|
|
297
|
+
|
|
298
|
+
# -------------------------------------------------------------- read ----
|
|
299
|
+
|
|
300
|
+
async def _read(self, url: str, *, query: str | None, max_chars: int,
|
|
301
|
+
refresh: bool = False) -> ReadResult:
|
|
302
|
+
max_chars = max(300, min(50_000, max_chars))
|
|
303
|
+
pk = Cache.key("page", url)
|
|
304
|
+
got, stored = (False, None) if refresh else self.cache.get(pk)
|
|
305
|
+
if got and "error" in stored:
|
|
306
|
+
raise RuntimeError(stored["error"])
|
|
307
|
+
if got:
|
|
308
|
+
page = Page(**stored)
|
|
309
|
+
hit = True
|
|
310
|
+
else:
|
|
311
|
+
if not await self._robots_allowed(url):
|
|
312
|
+
err = f"robots.txt disallows this path: {url}"
|
|
313
|
+
self.cache.set(pk, {"error": err}, ttl=max(3600, self.cfg.page_ttl // 24))
|
|
314
|
+
raise RobotsRefusal(err)
|
|
315
|
+
try:
|
|
316
|
+
page = await self.crawler.fetch(url)
|
|
317
|
+
except Exception as e:
|
|
318
|
+
err = f"{e.__class__.__name__}: {e}"
|
|
319
|
+
self.cache.set(pk, {"error": err}, ttl=max(3600, self.cfg.page_ttl // 24))
|
|
320
|
+
raise RuntimeError(err) from e
|
|
321
|
+
self.cache.set(pk, {"url": page.url, "status": page.status, "title": page.title,
|
|
322
|
+
"text": page.text, "raw_chars": page.raw_chars}, self.cfg.page_ttl)
|
|
323
|
+
hit = False
|
|
324
|
+
slim = distill.slim_markdown(page.text)
|
|
325
|
+
distilled = distill.passages(slim, query, max_chars)
|
|
326
|
+
return ReadResult(page=page, distilled=distilled, cache_hit=hit)
|
|
327
|
+
|
|
328
|
+
async def read(self, url: str, *, query: str | None = None, max_chars: int = 8000,
|
|
329
|
+
refresh: bool = False) -> str:
|
|
330
|
+
started = time.monotonic()
|
|
331
|
+
try:
|
|
332
|
+
res = await self._read(url, query=query, max_chars=max_chars, refresh=refresh)
|
|
333
|
+
except RobotsRefusal as e:
|
|
334
|
+
secs = round(time.monotonic() - started, 1)
|
|
335
|
+
self._log_metrics("read_url", cache="MISS", secs=secs, ok=False,
|
|
336
|
+
error="robots", url=url, q=query)
|
|
337
|
+
body = f"# {url}\n{url}\n\n(not fetched — {e})"
|
|
338
|
+
footer = f"[bathys: page 0 ch → 0 ch · robots-refused · cache MISS · {secs}s]"
|
|
339
|
+
return body + "\n\n" + footer
|
|
340
|
+
except Exception as e:
|
|
341
|
+
self._log_metrics("read_url", cache="MISS", secs=round(time.monotonic() - started, 1),
|
|
342
|
+
ok=False, error=e.__class__.__name__, url=url, q=query)
|
|
343
|
+
raise
|
|
344
|
+
secs = round(time.monotonic() - started, 1)
|
|
345
|
+
p = res.page
|
|
346
|
+
mode = "query-distilled" if query else "head-trimmed"
|
|
347
|
+
footer = (
|
|
348
|
+
f"[bathys: page {_ch(p.raw_chars)} ch → {_ch(len(res.distilled))} ch · "
|
|
349
|
+
f"{mode} · cache HIT · {secs}s]"
|
|
350
|
+
if res.cache_hit else
|
|
351
|
+
f"[bathys: page {_ch(p.raw_chars)} ch → {_ch(len(res.distilled))} ch · "
|
|
352
|
+
f"{mode} · cache MISS · {secs}s]"
|
|
353
|
+
)
|
|
354
|
+
header = f"# {p.title or url}\n{p.url}\n"
|
|
355
|
+
self._log_metrics("read_url", cache="HIT" if res.cache_hit else "MISS",
|
|
356
|
+
chars_in=p.raw_chars, chars_out=len(res.distilled),
|
|
357
|
+
secs=secs, url=url, q=query)
|
|
358
|
+
return header + "\n" + res.distilled + "\n\n" + footer
|
|
359
|
+
|
|
360
|
+
# ----------------------------------------------------------- research ----
|
|
361
|
+
|
|
362
|
+
async def research(self, query: str, *, max_sources: int = 3, max_results: int = 10,
|
|
363
|
+
per_source_chars: int = 3500, category: str | None = None,
|
|
364
|
+
engines: str | None = None, language: str | None = None,
|
|
365
|
+
time_range: str | None = None, refresh: bool = False) -> str:
|
|
366
|
+
started = time.monotonic()
|
|
367
|
+
max_sources = max(1, min(6, max_sources))
|
|
368
|
+
per_source_chars = max(300, min(8000, per_source_chars))
|
|
369
|
+
stored, _ = await self._search_outcome(
|
|
370
|
+
query, max_results=max_results, category=category,
|
|
371
|
+
engines=engines, language=language, time_range=time_range, refresh=refresh,
|
|
372
|
+
)
|
|
373
|
+
raw_total_hits = len(stored["hits"])
|
|
374
|
+
hits = stored["hits"][: max(1, min(20, max_results))]
|
|
375
|
+
top = hits[:max_sources]
|
|
376
|
+
sem = self._dive_sem
|
|
377
|
+
|
|
378
|
+
async def dive(h):
|
|
379
|
+
async with sem:
|
|
380
|
+
try:
|
|
381
|
+
return h, await self._read(h["url"], query=query, max_chars=per_source_chars,
|
|
382
|
+
refresh=refresh), None
|
|
383
|
+
except Exception as e:
|
|
384
|
+
return h, None, f"{e.__class__.__name__}: {e}"
|
|
385
|
+
|
|
386
|
+
results = await asyncio.gather(*(dive(h) for h in top))
|
|
387
|
+
|
|
388
|
+
sections: list[str] = [f"# Bathys research: {query!r}"]
|
|
389
|
+
if stored["answers"]:
|
|
390
|
+
sections.append("Answer: " + stored["answers"][0])
|
|
391
|
+
raw_total = out_total = 0
|
|
392
|
+
for i, (hit, res, err) in enumerate(results, 1):
|
|
393
|
+
meta_bits = [hit["url"]]
|
|
394
|
+
if hit.get("engines"):
|
|
395
|
+
meta_bits.append("engines: " + ", ".join(hit["engines"][:4]))
|
|
396
|
+
if hit.get("published"):
|
|
397
|
+
meta_bits.append(hit["published"])
|
|
398
|
+
head = f"## {i}. {hit['title']}\n" + " · ".join(meta_bits)
|
|
399
|
+
if res is None:
|
|
400
|
+
sections.append(f"{head}\n\n(not fetched — {err}; snippet: {hit['snippet']})")
|
|
401
|
+
continue
|
|
402
|
+
raw_total += res.page.raw_chars
|
|
403
|
+
out_total += len(res.distilled)
|
|
404
|
+
sections.append(f"{head}\n\n{res.distilled}")
|
|
405
|
+
rest = hits[max_sources:max_sources + 5]
|
|
406
|
+
if rest:
|
|
407
|
+
sections.append("More hits (not fetched):\n" + "\n".join(
|
|
408
|
+
f"- {h['title']} — {h['url']}" for h in rest))
|
|
409
|
+
secs = round(time.monotonic() - started, 1)
|
|
410
|
+
sections.append(
|
|
411
|
+
f"[bathys: {raw_total_hits} raw hits, top {len(hits)} considered · "
|
|
412
|
+
f"dove {len(top)} pages · "
|
|
413
|
+
f"{_ch(raw_total)} ch fetched → {_ch(out_total)} ch returned · {secs}s]"
|
|
414
|
+
)
|
|
415
|
+
self._log_metrics("deep_research", chars_in=raw_total, chars_out=out_total,
|
|
416
|
+
secs=secs, q=query)
|
|
417
|
+
return "\n\n".join(sections)
|
|
418
|
+
|
|
419
|
+
# ----------------------------------------------------------- backend ----
|
|
420
|
+
|
|
421
|
+
async def _backend(self) -> httpx.AsyncClient:
|
|
422
|
+
if not self._searx_ok:
|
|
423
|
+
await ensure_running(self.cfg, self.http)
|
|
424
|
+
self._searx_ok = True
|
|
425
|
+
return self.http
|
bathys/crawler.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Shared Crawl4AI browser: one headless instance for all dives."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from .config import Config
|
|
9
|
+
|
|
10
|
+
_UA = (
|
|
11
|
+
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
|
12
|
+
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
_EXCLUDED_TAGS = [
|
|
16
|
+
"nav", "footer", "header", "aside", "form", "noscript", "button",
|
|
17
|
+
"svg", "iframe", "style", "script",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
_EXCLUDED_SELECTOR = (
|
|
21
|
+
"nav,footer,header,aside,.sidebar,.cookie,#cookie-banner,.ads,"
|
|
22
|
+
"[aria-hidden='true']"
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class Page:
|
|
28
|
+
url: str
|
|
29
|
+
status: int
|
|
30
|
+
title: str
|
|
31
|
+
text: str
|
|
32
|
+
raw_chars: int
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _imports():
|
|
36
|
+
from crawl4ai import AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig
|
|
37
|
+
try:
|
|
38
|
+
from crawl4ai import DefaultMarkdownGenerator, PruningContentFilter
|
|
39
|
+
except ImportError: # older layout
|
|
40
|
+
from crawl4ai.content_filter_strategy import PruningContentFilter
|
|
41
|
+
from crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator
|
|
42
|
+
return AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig, DefaultMarkdownGenerator, PruningContentFilter
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Crawler:
|
|
46
|
+
def __init__(self, cfg: Config) -> None:
|
|
47
|
+
self._cfg = cfg
|
|
48
|
+
self._crawler = None
|
|
49
|
+
self._lock = asyncio.Lock()
|
|
50
|
+
|
|
51
|
+
async def _ensure(self):
|
|
52
|
+
if self._crawler is not None:
|
|
53
|
+
return self._crawler
|
|
54
|
+
async with self._lock:
|
|
55
|
+
if self._crawler is None:
|
|
56
|
+
AsyncWebCrawler, BrowserConfig, *_ = _imports()
|
|
57
|
+
browser = BrowserConfig(
|
|
58
|
+
headless=True,
|
|
59
|
+
text_mode=True,
|
|
60
|
+
light_mode=True,
|
|
61
|
+
user_agent=_UA,
|
|
62
|
+
verbose=False,
|
|
63
|
+
)
|
|
64
|
+
self._crawler = AsyncWebCrawler(config=browser)
|
|
65
|
+
await self._crawler.start()
|
|
66
|
+
return self._crawler
|
|
67
|
+
|
|
68
|
+
async def stop(self) -> None:
|
|
69
|
+
if self._crawler is not None:
|
|
70
|
+
try:
|
|
71
|
+
await self._crawler.stop()
|
|
72
|
+
except Exception:
|
|
73
|
+
pass
|
|
74
|
+
self._crawler = None
|
|
75
|
+
|
|
76
|
+
async def fetch(self, url: str) -> Page:
|
|
77
|
+
crawler = await self._ensure()
|
|
78
|
+
_, _, CacheMode, CrawlerRunConfig, DefaultMarkdownGenerator, PruningContentFilter = _imports()
|
|
79
|
+
run = CrawlerRunConfig(
|
|
80
|
+
cache_mode=CacheMode.BYPASS,
|
|
81
|
+
page_timeout=int(self._cfg.crawl_timeout * 1000),
|
|
82
|
+
excluded_tags=_EXCLUDED_TAGS,
|
|
83
|
+
excluded_selector=_EXCLUDED_SELECTOR,
|
|
84
|
+
word_count_threshold=8,
|
|
85
|
+
verbose=False,
|
|
86
|
+
markdown_generator=DefaultMarkdownGenerator(
|
|
87
|
+
content_filter=PruningContentFilter(threshold=0.48, threshold_type="fixed")
|
|
88
|
+
),
|
|
89
|
+
)
|
|
90
|
+
result = await crawler.arun(url=url, config=run)
|
|
91
|
+
if not getattr(result, "success", False):
|
|
92
|
+
reason = getattr(result, "error_message", "") or f"status {getattr(result, 'status_code', '?')}"
|
|
93
|
+
raise RuntimeError(f"crawl failed: {reason}")
|
|
94
|
+
md = result.markdown
|
|
95
|
+
text = getattr(md, "fit_markdown", None) or getattr(md, "raw_markdown", None) or str(md or "")
|
|
96
|
+
meta = getattr(result, "metadata", None) or {}
|
|
97
|
+
raw_len = len(getattr(md, "raw_markdown", "") or "") or len(text)
|
|
98
|
+
return Page(
|
|
99
|
+
url=getattr(result, "url", url) or url,
|
|
100
|
+
status=int(getattr(result, "status_code", 0) or 0),
|
|
101
|
+
title=str(meta.get("title", "") or "").strip(),
|
|
102
|
+
text=text.strip(),
|
|
103
|
+
raw_chars=raw_len,
|
|
104
|
+
)
|
bathys/distill.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Query-focused distillation.
|
|
2
|
+
|
|
3
|
+
Score text chunks against the query (BM25-flavoured), keep the best ones in
|
|
4
|
+
document order, trim to a hard character budget. This is where most of the
|
|
5
|
+
token savings happen: a crawled page is often 50-300k chars, we return a few
|
|
6
|
+
thousand.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import math
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
_MD_LINK = re.compile(r"!?\[([^\]\n]*)\]\([^)]*\)")
|
|
15
|
+
_MD_REF = re.compile(r"\n\[\^?\d+\]:.*")
|
|
16
|
+
_SENT_SPLIT = re.compile(r"(?<=[.!?…])\s+")
|
|
17
|
+
_WS = re.compile(r"[ \t]+")
|
|
18
|
+
|
|
19
|
+
_STOP = frozenset("""
|
|
20
|
+
a about an and are as at be been but by for from has have how in into is it its
|
|
21
|
+
of on or that the their there this to was were what when where which who will
|
|
22
|
+
with you your
|
|
23
|
+
и в во не что он на я с со как а то все она так его но да ты к у же вы за бы по
|
|
24
|
+
только ее мне было вот от меня еще нет о из ему теперь когда даже ну вдруг ли
|
|
25
|
+
если уже или ни быть был него до вас нибудь опять уж вам ведь там потом себя
|
|
26
|
+
ничего ей может они тут где есть надо ней для мы тебя их чем была сам чтоб без
|
|
27
|
+
будто чего раз тоже себе под будет тогда кто этот того потому этого какой совсем
|
|
28
|
+
ним здесь этом один почти мой тем чтобы нее сейчас были куда зачем всех никогда
|
|
29
|
+
сегодня можно при наконец два об другой хоть после над больше тот через эти нас
|
|
30
|
+
про всего них какая много три эту моя впрочем хорошо свою этой перед иногда
|
|
31
|
+
лучше чуть том нельзя такой им более всегда конечно всю между
|
|
32
|
+
""".split())
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def slim_markdown(text: str) -> str:
|
|
36
|
+
"""Token-oriented markdown slimming: inline links/images become their label."""
|
|
37
|
+
text = _MD_REF.sub("", text)
|
|
38
|
+
text = _MD_LINK.sub(lambda m: (m.group(1) or "").strip(), text)
|
|
39
|
+
text = re.sub(r"\n{3,}", "\n\n", text)
|
|
40
|
+
return text.strip()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _tokens(text: str) -> list[str]:
|
|
44
|
+
return [w for w in re.findall(r"[a-zа-яё0-9]+", text.lower()) if len(w) > 1 and w not in _STOP]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _split_chunks(text: str, target: int = 320, hard: int = 500) -> list[str]:
|
|
48
|
+
chunks: list[str] = []
|
|
49
|
+
for para in re.split(r"\n\s*\n", text):
|
|
50
|
+
para = para.strip()
|
|
51
|
+
if not para:
|
|
52
|
+
continue
|
|
53
|
+
if len(para) <= hard:
|
|
54
|
+
chunks.append(para)
|
|
55
|
+
continue
|
|
56
|
+
buf: list[str] = []
|
|
57
|
+
size = 0
|
|
58
|
+
for sent in _SENT_SPLIT.split(_WS.sub(" ", para)):
|
|
59
|
+
if not sent:
|
|
60
|
+
continue
|
|
61
|
+
if buf and size + len(sent) > target:
|
|
62
|
+
chunks.append(" ".join(buf))
|
|
63
|
+
buf, size = [sent], len(sent)
|
|
64
|
+
else:
|
|
65
|
+
buf.append(sent)
|
|
66
|
+
size += len(sent) + 1
|
|
67
|
+
if buf:
|
|
68
|
+
chunks.append(" ".join(buf))
|
|
69
|
+
out: list[str] = []
|
|
70
|
+
for c in chunks:
|
|
71
|
+
while len(c) > 900:
|
|
72
|
+
out.append(c[:900])
|
|
73
|
+
c = c[900:]
|
|
74
|
+
out.append(c)
|
|
75
|
+
return out
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def passages(text: str, query: str | None, max_chars: int) -> str:
|
|
79
|
+
"""Best passages of `text` for `query`, in document order, <= max_chars."""
|
|
80
|
+
text = text.strip()
|
|
81
|
+
if not text:
|
|
82
|
+
return ""
|
|
83
|
+
if len(text) <= max_chars:
|
|
84
|
+
return text
|
|
85
|
+
chunks = _split_chunks(text)
|
|
86
|
+
if not chunks:
|
|
87
|
+
return text[:max_chars]
|
|
88
|
+
keep: set[int] = set()
|
|
89
|
+
q_counts: dict[str, int] = {}
|
|
90
|
+
if query:
|
|
91
|
+
for w in _tokens(query):
|
|
92
|
+
q_counts[w] = q_counts.get(w, 0) + 1
|
|
93
|
+
if q_counts:
|
|
94
|
+
toks = [_tokens(c) for c in chunks]
|
|
95
|
+
n = len(chunks)
|
|
96
|
+
df: dict[str, int] = {}
|
|
97
|
+
for ts in toks:
|
|
98
|
+
for t in set(ts):
|
|
99
|
+
df[t] = df.get(t, 0) + 1
|
|
100
|
+
scored: list[tuple[float, int]] = []
|
|
101
|
+
for i, ts in enumerate(toks):
|
|
102
|
+
counts: dict[str, int] = {}
|
|
103
|
+
for t in ts:
|
|
104
|
+
counts[t] = counts.get(t, 0) + 1
|
|
105
|
+
s = 0.0
|
|
106
|
+
for term, qtf in q_counts.items():
|
|
107
|
+
tf = counts.get(term, 0)
|
|
108
|
+
if tf:
|
|
109
|
+
s += qtf * (1.0 + math.log(tf)) * math.log(1.0 + n / df.get(term, 1))
|
|
110
|
+
scored.append((s, i))
|
|
111
|
+
scored.sort(key=lambda p: -p[0])
|
|
112
|
+
for s, i in scored:
|
|
113
|
+
if s <= 0:
|
|
114
|
+
break
|
|
115
|
+
if sum(len(chunks[j]) + 2 for j in keep) >= max_chars:
|
|
116
|
+
break
|
|
117
|
+
keep.add(i)
|
|
118
|
+
# query terms absent from the page -> fall back to the lead
|
|
119
|
+
if not keep:
|
|
120
|
+
keep = {0}
|
|
121
|
+
else:
|
|
122
|
+
keep = {0} # no query: head-trimmed reading, lead paragraph carries it
|
|
123
|
+
picked = sorted(keep)
|
|
124
|
+
parts: list[str] = []
|
|
125
|
+
used = 0
|
|
126
|
+
for i in picked:
|
|
127
|
+
c = chunks[i]
|
|
128
|
+
if used + len(c) + 2 > max_chars:
|
|
129
|
+
remain = max_chars - used - 1
|
|
130
|
+
if remain > 80:
|
|
131
|
+
parts.append(c[:remain].rsplit(" ", 1)[0] + "…")
|
|
132
|
+
break
|
|
133
|
+
parts.append(c)
|
|
134
|
+
used += len(c) + 2
|
|
135
|
+
return "\n\n".join(parts)
|