bathys 0.6.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
bathys/core.py ADDED
@@ -0,0 +1,425 @@
1
+ """Core pipeline: search -> dive -> distill, cached. Framework-free on purpose,
2
+ so scripts and tests can drive it without MCP framing."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import asyncio
7
+ import time
8
+ from dataclasses import dataclass
9
+
10
+ import httpx
11
+
12
+ from . import distill, searx
13
+ from .cache import Cache
14
+ from .config import Config
15
+ from .crawler import Crawler, Page
16
+ from .services import ensure_running, stop_native
17
+
18
+
19
+ class RobotsRefusal(Exception):
20
+ """Target host disallows this path for crawlers in its robots.txt."""
21
+
22
+
23
+ def _ch(chars: int) -> str:
24
+ """Honest character count — one scale for every footer field (contract K1 fix)."""
25
+ return str(chars)
26
+
27
+
28
+ # Fallback engine sets for empty-result retries (F-101). Ordered by independence
29
+ # from big-scraping backends; rotation always happens WITH backoff pauses.
30
+ _RETRY_ENGINE_SETS: list[str | None] = [
31
+ None, # as requested / instance defaults
32
+ "duckduckgo,bing,brave,startpage",
33
+ "wikipedia,duckduckgo,mojeek,bing",
34
+ ]
35
+
36
+
37
+ @dataclass
38
+ class ReadResult:
39
+ page: Page
40
+ distilled: str
41
+ cache_hit: bool
42
+
43
+
44
+ class Engine:
45
+ """Everything the tools need; one instance per server process."""
46
+
47
+ def __init__(self, cfg: Config) -> None:
48
+ self.cfg = cfg
49
+ self.cache = Cache(cfg.cache_dir / "cache.db")
50
+ self.crawler = Crawler(cfg)
51
+ self.http: httpx.AsyncClient | None = None
52
+ self._searx_ok = False
53
+ self._search_ts = 0.0
54
+ self._pace_lock = asyncio.Lock()
55
+ self._engine_fails: dict[str, int] = {}
56
+ self._dive_sem = asyncio.Semaphore(cfg.dive_concurrency)
57
+ self._robots: dict[str, "object | None"] = {} # host -> RobotFileParser | None (None = allow)
58
+ self._metrics_path = cfg.data_dir / "metrics.jsonl"
59
+ try:
60
+ cfg.data_dir.mkdir(parents=True, exist_ok=True)
61
+ except OSError:
62
+ pass
63
+
64
+ # ------------------------------------------------------------ metrics ---
65
+
66
+ def _log_metrics(self, tool: str, *, cache: str | None = None, chars_in: int = 0,
67
+ chars_out: int = 0, secs: float = 0.0, ok: bool = True,
68
+ error: str | None = None, url: str | None = None,
69
+ q: str | None = None) -> None:
70
+ """F-304: local JSONL journal (schema — docs/operations/metrics.md §3.2).
71
+ Metrics must never break a tool: swallow everything."""
72
+ if not self.cfg.metrics:
73
+ return
74
+ import datetime
75
+ import hashlib
76
+
77
+ def h(v: str | None) -> str | None:
78
+ return hashlib.sha256(v.encode()).hexdigest()[:12] if v else None
79
+
80
+ event = {
81
+ "ts": datetime.datetime.now(datetime.timezone.utc).isoformat(timespec="seconds"),
82
+ "tool": tool,
83
+ "cache": cache,
84
+ "chars_in": chars_in,
85
+ "chars_out": chars_out,
86
+ "secs": secs,
87
+ "ok": ok,
88
+ "error_class": error,
89
+ "url_hash": h(url),
90
+ "q_hash": h(q),
91
+ }
92
+ try:
93
+ import json as _json
94
+ with open(self._metrics_path, "a", encoding="utf-8") as f:
95
+ f.write(_json.dumps(event, ensure_ascii=False, separators=(",", ":")) + "\n")
96
+ except OSError:
97
+ pass
98
+
99
+ # ------------------------------------------------------------- robots ---
100
+
101
+ async def _robots_allowed(self, url: str) -> bool:
102
+ """F-303: respect robots.txt for our direct page dives; fail-open on
103
+ missing/unreachable robots (standard robots semantics). Search itself
104
+ is not crawling — engines fetch, we only query them."""
105
+ if not self.cfg.respect_robots:
106
+ return True
107
+ from urllib import robotparser
108
+ from urllib.parse import urlsplit
109
+
110
+ parts = urlsplit(url)
111
+ host = (parts.scheme, parts.netloc)
112
+ if host in self._robots:
113
+ rp = self._robots[host]
114
+ else:
115
+ rp = None
116
+ if self.http is not None and parts.netloc:
117
+ try:
118
+ resp = await self.http.get(
119
+ f"{parts.scheme}://{parts.netloc}/robots.txt", timeout=5.0)
120
+ if resp.status_code == 200:
121
+ parser = robotparser.RobotFileParser()
122
+ parser.parse(resp.text.splitlines())
123
+ rp = parser
124
+ except (httpx.HTTPError, ValueError):
125
+ rp = None # unreachable robots -> allow
126
+ self._robots[host] = rp
127
+ if rp is None:
128
+ return True
129
+ return rp.can_fetch("*", parts.path or "/")
130
+
131
+ async def start(self) -> None:
132
+ self.http = httpx.AsyncClient(follow_redirects=True, timeout=self.cfg.search_timeout)
133
+
134
+ async def stop(self) -> None:
135
+ await self.crawler.stop()
136
+ await stop_native()
137
+ if self.http:
138
+ await self.http.aclose()
139
+
140
+ # ------------------------------------------------------------ search ----
141
+
142
+ async def _polite_pace(self) -> None:
143
+ """F-301: keep a configurable minimum interval between backend searches."""
144
+ async with self._pace_lock:
145
+ wait = self.cfg.search_min_interval - (time.monotonic() - self._search_ts)
146
+ if wait > 0:
147
+ await asyncio.sleep(wait)
148
+ self._search_ts = time.monotonic()
149
+
150
+ def _record_health(self, outcome: searx.SearchOutcome) -> None:
151
+ """F-102: consecutive-failure streak per engine; any hit resets it."""
152
+ for e in outcome.unresponsive:
153
+ name = e.split(":")[0]
154
+ self._engine_fails[name] = self._engine_fails.get(name, 0) + 1
155
+ for h in outcome.hits:
156
+ for name in h.engines:
157
+ self._engine_fails.pop(name, None)
158
+
159
+ def _healthy_fallback(self, engine_set: str) -> str:
160
+ bad = {e for e, c in self._engine_fails.items() if c >= 3}
161
+ if not bad:
162
+ return engine_set
163
+ kept = [e for e in engine_set.split(",") if e.strip() and e.strip() not in bad]
164
+ return ",".join(kept) if kept else engine_set
165
+
166
+ def bad_engines(self) -> list[str]:
167
+ """Engines with 3+ consecutive empty/unresponsive streaks (for doctor/diagnostics)."""
168
+ return sorted(e for e, c in self._engine_fails.items() if c >= 3)
169
+
170
+ async def _search_outcome(self, query: str, *, max_results: int, category: str | None,
171
+ engines: str | None, language: str | None, time_range: str | None,
172
+ refresh: bool = False):
173
+ max_results = max(1, min(20, max_results))
174
+ if time_range not in (None, "", "day", "week", "month", "year"):
175
+ time_range = None
176
+ ck = Cache.key("search", query, max_results, category, engines, language, time_range)
177
+ if not refresh:
178
+ got, stored = self.cache.get(ck)
179
+ if got:
180
+ return stored, True
181
+ http = await self._backend()
182
+ # F-101: on an empty outcome retry with independent engine sets + backoff.
183
+ plan: list[str | None] = ([engines] if engines else []) + _RETRY_ENGINE_SETS
184
+ seen_sets: set[str] = set()
185
+ attempts = 0
186
+ outcome: searx.SearchOutcome | None = None
187
+ for engine_set in plan:
188
+ key = engine_set or ""
189
+ if key in seen_sets:
190
+ continue
191
+ seen_sets.add(key)
192
+ if attempts > self.cfg.search_retries:
193
+ break
194
+ if engine_set and attempts:
195
+ engine_set = self._healthy_fallback(engine_set)
196
+ await self._polite_pace()
197
+ try:
198
+ outcome = await searx.search(
199
+ self.cfg, http, query,
200
+ categories=category, engines=engine_set, language=language, time_range=time_range,
201
+ )
202
+ except searx.SearxError:
203
+ self._searx_ok = False
204
+ raise
205
+ self._record_health(outcome)
206
+ attempts += 1
207
+ if outcome.hits or outcome.answers:
208
+ break
209
+ if attempts <= self.cfg.search_retries:
210
+ await asyncio.sleep(1.5 * attempts) # growing pause: rotation never without a delay
211
+ assert outcome is not None
212
+ stored = {
213
+ "hits": [
214
+ {"title": h.title, "url": h.url, "snippet": h.snippet, "engines": h.engines,
215
+ "score": h.score, "published": h.published}
216
+ for h in outcome.hits
217
+ ],
218
+ "answers": outcome.answers,
219
+ "suggestions": outcome.suggestions,
220
+ "seconds": outcome.seconds,
221
+ "raw_chars": outcome.raw_chars,
222
+ "retries": attempts - 1,
223
+ "unresponsive": outcome.unresponsive,
224
+ }
225
+ # empty outcomes are real answers too — but cache them briefly so a
226
+ # broken minute doesn't shadow an hour of retries
227
+ ttl = self.cfg.search_ttl if (outcome.hits or outcome.answers) else min(self.cfg.search_ttl, 600)
228
+ self.cache.set(ck, stored, ttl)
229
+ return stored, False
230
+
231
+ async def search(self, query: str, *, max_results: int = 8, category: str | None = None,
232
+ engines: str | None = None, language: str | None = None,
233
+ time_range: str | None = None, refresh: bool = False,
234
+ as_json: bool = False) -> str:
235
+ started = time.monotonic()
236
+ try:
237
+ stored, cached = await self._search_outcome(
238
+ query, max_results=max_results, category=category,
239
+ engines=engines, language=language, time_range=time_range, refresh=refresh,
240
+ )
241
+ except Exception as e:
242
+ self._log_metrics("web_search", secs=round(time.monotonic() - started, 1),
243
+ ok=False, error=e.__class__.__name__, q=query)
244
+ raise
245
+ secs = round(time.monotonic() - started, 1)
246
+ hits = stored["hits"][: max(1, min(20, max_results))]
247
+ lines: list[str] = []
248
+ if stored["answers"]:
249
+ lines.append("Answer: " + stored["answers"][0])
250
+ for i, h in enumerate(hits, 1):
251
+ date = f" [{h['published']}]" if h["published"] else ""
252
+ lines.append(f"{i}. {h['title']}\n {h['url']}\n {h['snippet']}{date}")
253
+ if stored["suggestions"]:
254
+ lines.append("Refine: " + ", ".join(stored["suggestions"]))
255
+ body = "\n".join(lines) if lines else self._empty_body(query, stored)
256
+ if as_json:
257
+ # F-204: machine-readable mode — pure JSON, no footer (metrics go to
258
+ # metrics.jsonl; footer would break strict json.loads consumers).
259
+ import json as _json
260
+ payload = {
261
+ "query": query,
262
+ "count": len(hits),
263
+ "hits": [
264
+ {"title": h["title"], "url": h["url"], "snippet": h["snippet"],
265
+ "engines": h["engines"], "published": h["published"] or None}
266
+ for h in hits
267
+ ],
268
+ }
269
+ if stored["answers"]:
270
+ payload["answer"] = stored["answers"][0]
271
+ out = _json.dumps(payload, ensure_ascii=False, separators=(",", ":"))
272
+ self._log_metrics("web_search", cache="HIT" if cached else "MISS",
273
+ chars_in=stored["raw_chars"], chars_out=len(out),
274
+ secs=secs, q=query)
275
+ return out
276
+ footer = (
277
+ f"[bathys: {len(hits)} hits · cache HIT · {secs}s · searx json "
278
+ f"{_ch(stored['raw_chars'])} ch → {_ch(len(body))} ch]"
279
+ if cached else
280
+ f"[bathys: {len(hits)} hits · {secs}s · searx json "
281
+ f"{_ch(stored['raw_chars'])} ch → {_ch(len(body))} ch]"
282
+ )
283
+ self._log_metrics("web_search", cache="HIT" if cached else "MISS",
284
+ chars_in=stored["raw_chars"], chars_out=len(body),
285
+ secs=secs, q=query)
286
+ return body + "\n" + footer
287
+
288
+ @staticmethod
289
+ def _empty_body(query: str, stored: dict) -> str:
290
+ body = f"No results for: {query}"
291
+ notes: list[str] = []
292
+ if stored.get("retries"):
293
+ notes.append(f"tried {stored['retries'] + 1} engine sets")
294
+ if stored.get("unresponsive"):
295
+ notes.append("unresponsive: " + ", ".join(stored["unresponsive"][:5]))
296
+ return body + (" · " + " · ".join(notes) if notes else "")
297
+
298
+ # -------------------------------------------------------------- read ----
299
+
300
+ async def _read(self, url: str, *, query: str | None, max_chars: int,
301
+ refresh: bool = False) -> ReadResult:
302
+ max_chars = max(300, min(50_000, max_chars))
303
+ pk = Cache.key("page", url)
304
+ got, stored = (False, None) if refresh else self.cache.get(pk)
305
+ if got and "error" in stored:
306
+ raise RuntimeError(stored["error"])
307
+ if got:
308
+ page = Page(**stored)
309
+ hit = True
310
+ else:
311
+ if not await self._robots_allowed(url):
312
+ err = f"robots.txt disallows this path: {url}"
313
+ self.cache.set(pk, {"error": err}, ttl=max(3600, self.cfg.page_ttl // 24))
314
+ raise RobotsRefusal(err)
315
+ try:
316
+ page = await self.crawler.fetch(url)
317
+ except Exception as e:
318
+ err = f"{e.__class__.__name__}: {e}"
319
+ self.cache.set(pk, {"error": err}, ttl=max(3600, self.cfg.page_ttl // 24))
320
+ raise RuntimeError(err) from e
321
+ self.cache.set(pk, {"url": page.url, "status": page.status, "title": page.title,
322
+ "text": page.text, "raw_chars": page.raw_chars}, self.cfg.page_ttl)
323
+ hit = False
324
+ slim = distill.slim_markdown(page.text)
325
+ distilled = distill.passages(slim, query, max_chars)
326
+ return ReadResult(page=page, distilled=distilled, cache_hit=hit)
327
+
328
+ async def read(self, url: str, *, query: str | None = None, max_chars: int = 8000,
329
+ refresh: bool = False) -> str:
330
+ started = time.monotonic()
331
+ try:
332
+ res = await self._read(url, query=query, max_chars=max_chars, refresh=refresh)
333
+ except RobotsRefusal as e:
334
+ secs = round(time.monotonic() - started, 1)
335
+ self._log_metrics("read_url", cache="MISS", secs=secs, ok=False,
336
+ error="robots", url=url, q=query)
337
+ body = f"# {url}\n{url}\n\n(not fetched — {e})"
338
+ footer = f"[bathys: page 0 ch → 0 ch · robots-refused · cache MISS · {secs}s]"
339
+ return body + "\n\n" + footer
340
+ except Exception as e:
341
+ self._log_metrics("read_url", cache="MISS", secs=round(time.monotonic() - started, 1),
342
+ ok=False, error=e.__class__.__name__, url=url, q=query)
343
+ raise
344
+ secs = round(time.monotonic() - started, 1)
345
+ p = res.page
346
+ mode = "query-distilled" if query else "head-trimmed"
347
+ footer = (
348
+ f"[bathys: page {_ch(p.raw_chars)} ch → {_ch(len(res.distilled))} ch · "
349
+ f"{mode} · cache HIT · {secs}s]"
350
+ if res.cache_hit else
351
+ f"[bathys: page {_ch(p.raw_chars)} ch → {_ch(len(res.distilled))} ch · "
352
+ f"{mode} · cache MISS · {secs}s]"
353
+ )
354
+ header = f"# {p.title or url}\n{p.url}\n"
355
+ self._log_metrics("read_url", cache="HIT" if res.cache_hit else "MISS",
356
+ chars_in=p.raw_chars, chars_out=len(res.distilled),
357
+ secs=secs, url=url, q=query)
358
+ return header + "\n" + res.distilled + "\n\n" + footer
359
+
360
+ # ----------------------------------------------------------- research ----
361
+
362
+ async def research(self, query: str, *, max_sources: int = 3, max_results: int = 10,
363
+ per_source_chars: int = 3500, category: str | None = None,
364
+ engines: str | None = None, language: str | None = None,
365
+ time_range: str | None = None, refresh: bool = False) -> str:
366
+ started = time.monotonic()
367
+ max_sources = max(1, min(6, max_sources))
368
+ per_source_chars = max(300, min(8000, per_source_chars))
369
+ stored, _ = await self._search_outcome(
370
+ query, max_results=max_results, category=category,
371
+ engines=engines, language=language, time_range=time_range, refresh=refresh,
372
+ )
373
+ raw_total_hits = len(stored["hits"])
374
+ hits = stored["hits"][: max(1, min(20, max_results))]
375
+ top = hits[:max_sources]
376
+ sem = self._dive_sem
377
+
378
+ async def dive(h):
379
+ async with sem:
380
+ try:
381
+ return h, await self._read(h["url"], query=query, max_chars=per_source_chars,
382
+ refresh=refresh), None
383
+ except Exception as e:
384
+ return h, None, f"{e.__class__.__name__}: {e}"
385
+
386
+ results = await asyncio.gather(*(dive(h) for h in top))
387
+
388
+ sections: list[str] = [f"# Bathys research: {query!r}"]
389
+ if stored["answers"]:
390
+ sections.append("Answer: " + stored["answers"][0])
391
+ raw_total = out_total = 0
392
+ for i, (hit, res, err) in enumerate(results, 1):
393
+ meta_bits = [hit["url"]]
394
+ if hit.get("engines"):
395
+ meta_bits.append("engines: " + ", ".join(hit["engines"][:4]))
396
+ if hit.get("published"):
397
+ meta_bits.append(hit["published"])
398
+ head = f"## {i}. {hit['title']}\n" + " · ".join(meta_bits)
399
+ if res is None:
400
+ sections.append(f"{head}\n\n(not fetched — {err}; snippet: {hit['snippet']})")
401
+ continue
402
+ raw_total += res.page.raw_chars
403
+ out_total += len(res.distilled)
404
+ sections.append(f"{head}\n\n{res.distilled}")
405
+ rest = hits[max_sources:max_sources + 5]
406
+ if rest:
407
+ sections.append("More hits (not fetched):\n" + "\n".join(
408
+ f"- {h['title']} — {h['url']}" for h in rest))
409
+ secs = round(time.monotonic() - started, 1)
410
+ sections.append(
411
+ f"[bathys: {raw_total_hits} raw hits, top {len(hits)} considered · "
412
+ f"dove {len(top)} pages · "
413
+ f"{_ch(raw_total)} ch fetched → {_ch(out_total)} ch returned · {secs}s]"
414
+ )
415
+ self._log_metrics("deep_research", chars_in=raw_total, chars_out=out_total,
416
+ secs=secs, q=query)
417
+ return "\n\n".join(sections)
418
+
419
+ # ----------------------------------------------------------- backend ----
420
+
421
+ async def _backend(self) -> httpx.AsyncClient:
422
+ if not self._searx_ok:
423
+ await ensure_running(self.cfg, self.http)
424
+ self._searx_ok = True
425
+ return self.http
bathys/crawler.py ADDED
@@ -0,0 +1,104 @@
1
+ """Shared Crawl4AI browser: one headless instance for all dives."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ from dataclasses import dataclass
7
+
8
+ from .config import Config
9
+
10
+ _UA = (
11
+ "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
12
+ "(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
13
+ )
14
+
15
+ _EXCLUDED_TAGS = [
16
+ "nav", "footer", "header", "aside", "form", "noscript", "button",
17
+ "svg", "iframe", "style", "script",
18
+ ]
19
+
20
+ _EXCLUDED_SELECTOR = (
21
+ "nav,footer,header,aside,.sidebar,.cookie,#cookie-banner,.ads,"
22
+ "[aria-hidden='true']"
23
+ )
24
+
25
+
26
+ @dataclass
27
+ class Page:
28
+ url: str
29
+ status: int
30
+ title: str
31
+ text: str
32
+ raw_chars: int
33
+
34
+
35
+ def _imports():
36
+ from crawl4ai import AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig
37
+ try:
38
+ from crawl4ai import DefaultMarkdownGenerator, PruningContentFilter
39
+ except ImportError: # older layout
40
+ from crawl4ai.content_filter_strategy import PruningContentFilter
41
+ from crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator
42
+ return AsyncWebCrawler, BrowserConfig, CacheMode, CrawlerRunConfig, DefaultMarkdownGenerator, PruningContentFilter
43
+
44
+
45
+ class Crawler:
46
+ def __init__(self, cfg: Config) -> None:
47
+ self._cfg = cfg
48
+ self._crawler = None
49
+ self._lock = asyncio.Lock()
50
+
51
+ async def _ensure(self):
52
+ if self._crawler is not None:
53
+ return self._crawler
54
+ async with self._lock:
55
+ if self._crawler is None:
56
+ AsyncWebCrawler, BrowserConfig, *_ = _imports()
57
+ browser = BrowserConfig(
58
+ headless=True,
59
+ text_mode=True,
60
+ light_mode=True,
61
+ user_agent=_UA,
62
+ verbose=False,
63
+ )
64
+ self._crawler = AsyncWebCrawler(config=browser)
65
+ await self._crawler.start()
66
+ return self._crawler
67
+
68
+ async def stop(self) -> None:
69
+ if self._crawler is not None:
70
+ try:
71
+ await self._crawler.stop()
72
+ except Exception:
73
+ pass
74
+ self._crawler = None
75
+
76
+ async def fetch(self, url: str) -> Page:
77
+ crawler = await self._ensure()
78
+ _, _, CacheMode, CrawlerRunConfig, DefaultMarkdownGenerator, PruningContentFilter = _imports()
79
+ run = CrawlerRunConfig(
80
+ cache_mode=CacheMode.BYPASS,
81
+ page_timeout=int(self._cfg.crawl_timeout * 1000),
82
+ excluded_tags=_EXCLUDED_TAGS,
83
+ excluded_selector=_EXCLUDED_SELECTOR,
84
+ word_count_threshold=8,
85
+ verbose=False,
86
+ markdown_generator=DefaultMarkdownGenerator(
87
+ content_filter=PruningContentFilter(threshold=0.48, threshold_type="fixed")
88
+ ),
89
+ )
90
+ result = await crawler.arun(url=url, config=run)
91
+ if not getattr(result, "success", False):
92
+ reason = getattr(result, "error_message", "") or f"status {getattr(result, 'status_code', '?')}"
93
+ raise RuntimeError(f"crawl failed: {reason}")
94
+ md = result.markdown
95
+ text = getattr(md, "fit_markdown", None) or getattr(md, "raw_markdown", None) or str(md or "")
96
+ meta = getattr(result, "metadata", None) or {}
97
+ raw_len = len(getattr(md, "raw_markdown", "") or "") or len(text)
98
+ return Page(
99
+ url=getattr(result, "url", url) or url,
100
+ status=int(getattr(result, "status_code", 0) or 0),
101
+ title=str(meta.get("title", "") or "").strip(),
102
+ text=text.strip(),
103
+ raw_chars=raw_len,
104
+ )
bathys/distill.py ADDED
@@ -0,0 +1,135 @@
1
+ """Query-focused distillation.
2
+
3
+ Score text chunks against the query (BM25-flavoured), keep the best ones in
4
+ document order, trim to a hard character budget. This is where most of the
5
+ token savings happen: a crawled page is often 50-300k chars, we return a few
6
+ thousand.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import math
12
+ import re
13
+
14
+ _MD_LINK = re.compile(r"!?\[([^\]\n]*)\]\([^)]*\)")
15
+ _MD_REF = re.compile(r"\n\[\^?\d+\]:.*")
16
+ _SENT_SPLIT = re.compile(r"(?<=[.!?…])\s+")
17
+ _WS = re.compile(r"[ \t]+")
18
+
19
+ _STOP = frozenset("""
20
+ a about an and are as at be been but by for from has have how in into is it its
21
+ of on or that the their there this to was were what when where which who will
22
+ with you your
23
+ и в во не что он на я с со как а то все она так его но да ты к у же вы за бы по
24
+ только ее мне было вот от меня еще нет о из ему теперь когда даже ну вдруг ли
25
+ если уже или ни быть был него до вас нибудь опять уж вам ведь там потом себя
26
+ ничего ей может они тут где есть надо ней для мы тебя их чем была сам чтоб без
27
+ будто чего раз тоже себе под будет тогда кто этот того потому этого какой совсем
28
+ ним здесь этом один почти мой тем чтобы нее сейчас были куда зачем всех никогда
29
+ сегодня можно при наконец два об другой хоть после над больше тот через эти нас
30
+ про всего них какая много три эту моя впрочем хорошо свою этой перед иногда
31
+ лучше чуть том нельзя такой им более всегда конечно всю между
32
+ """.split())
33
+
34
+
35
+ def slim_markdown(text: str) -> str:
36
+ """Token-oriented markdown slimming: inline links/images become their label."""
37
+ text = _MD_REF.sub("", text)
38
+ text = _MD_LINK.sub(lambda m: (m.group(1) or "").strip(), text)
39
+ text = re.sub(r"\n{3,}", "\n\n", text)
40
+ return text.strip()
41
+
42
+
43
+ def _tokens(text: str) -> list[str]:
44
+ return [w for w in re.findall(r"[a-zа-яё0-9]+", text.lower()) if len(w) > 1 and w not in _STOP]
45
+
46
+
47
+ def _split_chunks(text: str, target: int = 320, hard: int = 500) -> list[str]:
48
+ chunks: list[str] = []
49
+ for para in re.split(r"\n\s*\n", text):
50
+ para = para.strip()
51
+ if not para:
52
+ continue
53
+ if len(para) <= hard:
54
+ chunks.append(para)
55
+ continue
56
+ buf: list[str] = []
57
+ size = 0
58
+ for sent in _SENT_SPLIT.split(_WS.sub(" ", para)):
59
+ if not sent:
60
+ continue
61
+ if buf and size + len(sent) > target:
62
+ chunks.append(" ".join(buf))
63
+ buf, size = [sent], len(sent)
64
+ else:
65
+ buf.append(sent)
66
+ size += len(sent) + 1
67
+ if buf:
68
+ chunks.append(" ".join(buf))
69
+ out: list[str] = []
70
+ for c in chunks:
71
+ while len(c) > 900:
72
+ out.append(c[:900])
73
+ c = c[900:]
74
+ out.append(c)
75
+ return out
76
+
77
+
78
+ def passages(text: str, query: str | None, max_chars: int) -> str:
79
+ """Best passages of `text` for `query`, in document order, <= max_chars."""
80
+ text = text.strip()
81
+ if not text:
82
+ return ""
83
+ if len(text) <= max_chars:
84
+ return text
85
+ chunks = _split_chunks(text)
86
+ if not chunks:
87
+ return text[:max_chars]
88
+ keep: set[int] = set()
89
+ q_counts: dict[str, int] = {}
90
+ if query:
91
+ for w in _tokens(query):
92
+ q_counts[w] = q_counts.get(w, 0) + 1
93
+ if q_counts:
94
+ toks = [_tokens(c) for c in chunks]
95
+ n = len(chunks)
96
+ df: dict[str, int] = {}
97
+ for ts in toks:
98
+ for t in set(ts):
99
+ df[t] = df.get(t, 0) + 1
100
+ scored: list[tuple[float, int]] = []
101
+ for i, ts in enumerate(toks):
102
+ counts: dict[str, int] = {}
103
+ for t in ts:
104
+ counts[t] = counts.get(t, 0) + 1
105
+ s = 0.0
106
+ for term, qtf in q_counts.items():
107
+ tf = counts.get(term, 0)
108
+ if tf:
109
+ s += qtf * (1.0 + math.log(tf)) * math.log(1.0 + n / df.get(term, 1))
110
+ scored.append((s, i))
111
+ scored.sort(key=lambda p: -p[0])
112
+ for s, i in scored:
113
+ if s <= 0:
114
+ break
115
+ if sum(len(chunks[j]) + 2 for j in keep) >= max_chars:
116
+ break
117
+ keep.add(i)
118
+ # query terms absent from the page -> fall back to the lead
119
+ if not keep:
120
+ keep = {0}
121
+ else:
122
+ keep = {0} # no query: head-trimmed reading, lead paragraph carries it
123
+ picked = sorted(keep)
124
+ parts: list[str] = []
125
+ used = 0
126
+ for i in picked:
127
+ c = chunks[i]
128
+ if used + len(c) + 2 > max_chars:
129
+ remain = max_chars - used - 1
130
+ if remain > 80:
131
+ parts.append(c[:remain].rsplit(" ", 1)[0] + "…")
132
+ break
133
+ parts.append(c)
134
+ used += len(c) + 2
135
+ return "\n\n".join(parts)