harvester-kit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
harvester/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """harvester — polite, reproducible web harvesting.
2
+
3
+ Fetch once, parse forever: every response is kept in a content-addressed
4
+ store, so parsers can be improved and re-run offline without touching the
5
+ network again.
6
+ """
7
+
8
+ from harvester.models import CachePolicy, DataLicense, Record, Request
9
+ from harvester.page import Page
10
+ from harvester.source import Source
11
+
12
+ __version__ = "0.1.0"
13
+
14
+ __all__ = [
15
+ "CachePolicy",
16
+ "DataLicense",
17
+ "Page",
18
+ "Record",
19
+ "Request",
20
+ "Source",
21
+ "__version__",
22
+ ]
harvester/cli.py ADDED
@@ -0,0 +1,437 @@
1
+ """Command-line interface."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import json
7
+ import signal
8
+ import sys
9
+ import time
10
+ from collections.abc import Awaitable, Callable
11
+ from pathlib import Path
12
+ from types import FrameType
13
+ from typing import Annotated, Any, Optional
14
+
15
+ import typer
16
+ from rich.console import Console, Group
17
+ from rich.live import Live
18
+ from rich.panel import Panel
19
+ from rich.table import Table
20
+ from rich.text import Text
21
+
22
+ from harvester import __version__
23
+ from harvester.engine import Harvester, RunReport, Settings
24
+ from harvester.fetch import DEFAULT_USER_AGENT, ScraplingFetcher
25
+ from harvester.frontier import Frontier
26
+ from harvester.models import url_fingerprint
27
+ from harvester.politeness import RobotsPolicy
28
+ from harvester.registry import SourceNotFound, installed_sources, load_source
29
+ from harvester.source import Source
30
+ from harvester.store import RawStore
31
+
32
+ app = typer.Typer(
33
+ name="harvester",
34
+ help="Polite, reproducible web harvesting: fetch once, parse forever.",
35
+ no_args_is_help=True,
36
+ rich_markup_mode="rich",
37
+ pretty_exceptions_show_locals=False,
38
+ )
39
+ console = Console()
40
+
41
+ DataDir = Annotated[
42
+ Path, typer.Option("--data-dir", "-d", help="Where stores, queues and runs live.")
43
+ ]
44
+ Options = Annotated[
45
+ Optional[list[str]], # noqa: UP045 — typer needs Optional on 3.10
46
+ typer.Option("--option", "-o", help="Source option as key=value (repeatable)."),
47
+ ]
48
+
49
+
50
+ def _version(value: bool) -> None:
51
+ if value:
52
+ console.print(f"harvester {__version__}")
53
+ raise typer.Exit()
54
+
55
+
56
+ @app.callback()
57
+ def main(
58
+ version: Annotated[
59
+ bool, typer.Option("--version", callback=_version, is_eager=True, help="Show version.")
60
+ ] = False,
61
+ ) -> None:
62
+ """Polite, reproducible web harvesting: fetch once, parse forever."""
63
+
64
+
65
+ # ── Helpers ─────────────────────────────────────────────────────────────
66
+
67
+
68
+ def _make_source(spec: str, options: list[str] | None) -> Source:
69
+ try:
70
+ cls = load_source(spec)
71
+ except SourceNotFound as exc:
72
+ console.print(f"[red]error:[/] {exc}")
73
+ raise typer.Exit(2) from None
74
+ parsed: dict[str, Any] = {}
75
+ for item in options or []:
76
+ key, sep, value = item.partition("=")
77
+ if not sep:
78
+ console.print(f"[red]error:[/] option {item!r} is not key=value")
79
+ raise typer.Exit(2)
80
+ parsed[key.strip()] = value
81
+ try:
82
+ return cls(**parsed)
83
+ except ValueError as exc:
84
+ console.print(f"[red]error:[/] {exc}")
85
+ raise typer.Exit(2) from None
86
+
87
+
88
+ def _human_bytes(n: float) -> str:
89
+ for unit in ("B", "KB", "MB", "GB"):
90
+ if n < 1024:
91
+ return f"{n:.0f} {unit}" if unit == "B" else f"{n:.1f} {unit}"
92
+ n /= 1024
93
+ return f"{n:.1f} TB"
94
+
95
+
96
+ def _dashboard(engine: Harvester, started: float) -> Panel:
97
+ s = engine.stats
98
+ frontier = engine.frontier.counts() if engine.frontier else {}
99
+ elapsed = max(time.monotonic() - started, 1e-6)
100
+
101
+ table = Table.grid(padding=(0, 3))
102
+ for _ in range(4):
103
+ table.add_column(justify="right", style="bold")
104
+ table.add_column(style="dim")
105
+ table.add_row(
106
+ str(s.pages), "pages", str(s.records), "records",
107
+ str(s.fetched), "fetched", str(s.from_cache + s.not_modified), "from cache",
108
+ ) # fmt: skip
109
+ table.add_row(
110
+ str(frontier.get("pending", 0)), "queued", str(s.retries), "retries",
111
+ str(s.failed), "failed", str(s.robots_blocked), "blocked",
112
+ ) # fmt: skip
113
+ table.add_row(
114
+ _human_bytes(s.bytes), "downloaded", f"{s.pages / elapsed * 60:.0f}", "pages/min",
115
+ f"{elapsed:.0f}s", "elapsed", str(s.offsite + s.missing), "skipped",
116
+ ) # fmt: skip
117
+
118
+ parts: list[Any] = [table]
119
+ if s.recent_errors:
120
+ errors = Text("\n".join(list(s.recent_errors)[-4:]), style="yellow", overflow="ellipsis")
121
+ parts += [Text(""), errors]
122
+ title = f"[bold]{engine.source.name}[/] | {engine._mode}"
123
+ return Panel(Group(*parts), title=title, border_style="cyan", expand=False)
124
+
125
+
126
+ def _run_with_dashboard(
127
+ engine: Harvester, coro_factory: Callable[[], Awaitable[RunReport]]
128
+ ) -> RunReport:
129
+ """Run the engine with a live view; first Ctrl+C stops gracefully, second aborts."""
130
+ interrupts = 0
131
+ loop: asyncio.AbstractEventLoop | None = None
132
+
133
+ def on_sigint(signum: int, frame: FrameType | None) -> None:
134
+ nonlocal interrupts
135
+ interrupts += 1
136
+ if interrupts == 1 and loop is not None:
137
+ console.print("[yellow]stopping after in-flight requests... (Ctrl+C again to abort)[/]")
138
+ loop.call_soon_threadsafe(engine.stop)
139
+ else:
140
+ raise KeyboardInterrupt
141
+
142
+ async def runner() -> RunReport:
143
+ nonlocal loop
144
+ loop = asyncio.get_running_loop()
145
+ started = time.monotonic()
146
+ with Live(_dashboard(engine, started), console=console, refresh_per_second=4) as live:
147
+ task: asyncio.Future[RunReport] = asyncio.ensure_future(coro_factory())
148
+ while not task.done():
149
+ live.update(_dashboard(engine, started))
150
+ await asyncio.sleep(0.25)
151
+ live.update(_dashboard(engine, started))
152
+ return await task
153
+
154
+ previous = signal.signal(signal.SIGINT, on_sigint)
155
+ try:
156
+ return asyncio.run(runner())
157
+ finally:
158
+ signal.signal(signal.SIGINT, previous)
159
+
160
+
161
+ def _print_report(report: RunReport) -> None:
162
+ colour = {"complete": "green", "limit": "cyan", "interrupted": "yellow"}[report.outcome]
163
+ console.print(f"[{colour}]{report.outcome}[/] run [bold]{report.run_id}[/]")
164
+ kinds = ", ".join(f"{n} {k}" for k, n in sorted(report.records_by_kind.items())) or "none"
165
+ console.print(f" records: {kinds}")
166
+ console.print(f" output: {report.records_path}")
167
+ if report.frontier.get("failed") or report.frontier.get("skipped"):
168
+ console.print(
169
+ f" [yellow]{report.frontier.get('failed', 0)} failed, "
170
+ f"{report.frontier.get('skipped', 0)} skipped[/] - see `harvester status`"
171
+ )
172
+
173
+
174
+ # ── Commands ────────────────────────────────────────────────────────────
175
+
176
+
177
+ @app.command("list")
178
+ def list_sources() -> None:
179
+ """Show installed sources."""
180
+ table = Table(title="Installed sources", title_justify="left")
181
+ table.add_column("name", style="bold cyan")
182
+ table.add_column("description")
183
+ table.add_column("data licence", style="dim")
184
+ for name, cls in installed_sources().items():
185
+ table.add_row(name, cls.description, cls.license.name if cls.license else "-")
186
+ console.print(table)
187
+
188
+
189
+ @app.command()
190
+ def run(
191
+ source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
192
+ option: Options = None,
193
+ data_dir: DataDir = Path(".harvester"),
194
+ limit: Annotated[Optional[int], typer.Option(help="Stop after this many pages.")] = None, # noqa: UP045
195
+ concurrency: Annotated[int, typer.Option(help="Concurrent workers in total.")] = 4,
196
+ delay: Annotated[float, typer.Option(help="Seconds between requests to one host.")] = 1.0,
197
+ refresh: Annotated[
198
+ bool, typer.Option(help="Revalidate cached responses (conditional requests).")
199
+ ] = False,
200
+ restart: Annotated[
201
+ bool, typer.Option(help="Discard unfinished progress and start a new pass.")
202
+ ] = False,
203
+ user_agent: Annotated[str, typer.Option(help="User-Agent; keep a contact URL in it.")] = (
204
+ DEFAULT_USER_AGENT
205
+ ),
206
+ robots_unavailable: Annotated[
207
+ Optional[str], # noqa: UP045
208
+ typer.Option(
209
+ help="Override the source's policy when robots.txt is unreachable: allow|disallow."
210
+ ),
211
+ ] = None,
212
+ ) -> None:
213
+ """Crawl a source. Resumes automatically if the last run was interrupted."""
214
+ src = _make_source(source, option)
215
+ if robots_unavailable not in (None, "allow", "disallow"):
216
+ console.print("[red]error:[/] --robots-unavailable must be allow or disallow")
217
+ raise typer.Exit(2)
218
+ settings = Settings(
219
+ data_dir=data_dir,
220
+ concurrency=concurrency,
221
+ delay=delay,
222
+ limit=limit,
223
+ refresh=refresh,
224
+ restart=restart,
225
+ user_agent=user_agent,
226
+ robots_unavailable=robots_unavailable,
227
+ )
228
+ engine = Harvester(src, settings)
229
+ try:
230
+ report = _run_with_dashboard(engine, engine.run)
231
+ finally:
232
+ engine.close()
233
+ _print_report(report)
234
+
235
+
236
+ @app.command()
237
+ def reparse(
238
+ source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
239
+ option: Options = None,
240
+ data_dir: DataDir = Path(".harvester"),
241
+ ) -> None:
242
+ """Re-run parsers over the raw store. Makes no network requests."""
243
+ src = _make_source(source, option)
244
+ engine = Harvester(src, Settings(data_dir=data_dir, concurrency=1))
245
+ try:
246
+ report = _run_with_dashboard(engine, engine.reparse)
247
+ finally:
248
+ engine.close()
249
+ _print_report(report)
250
+ if report.stats.get("missing"):
251
+ console.print(
252
+ f" [yellow]{report.stats['missing']} requests were never fetched; "
253
+ "run the crawl to fill them in[/]"
254
+ )
255
+
256
+
257
+ @app.command()
258
+ def status(
259
+ source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
260
+ data_dir: DataDir = Path(".harvester"),
261
+ ) -> None:
262
+ """Show queue, store and recent runs for a source."""
263
+ cls = load_source(source)
264
+ root = data_dir / cls.name
265
+ if not root.exists():
266
+ console.print(f"no data for {cls.name} in {data_dir}")
267
+ raise typer.Exit(1)
268
+
269
+ store = RawStore(root / "store")
270
+ stats = store.stats()
271
+ store.close()
272
+ console.print(
273
+ f"[bold]{cls.name}[/] raw store: {stats['responses']} responses, "
274
+ f"{stats['unique_bodies']} unique bodies, {_human_bytes(stats['bytes'])}"
275
+ )
276
+
277
+ frontier_path = root / "frontier.sqlite"
278
+ if frontier_path.exists():
279
+ frontier = Frontier(frontier_path)
280
+ counts = frontier.counts()
281
+ console.print("queue: " + ", ".join(f"{n} {s}" for s, n in sorted(counts.items())))
282
+ problems = frontier.problems()
283
+ frontier.close()
284
+ if problems:
285
+ table = Table(title="Failed / skipped", title_justify="left")
286
+ table.add_column("state")
287
+ table.add_column("url", overflow="fold")
288
+ table.add_column("reason", style="yellow", overflow="fold")
289
+ for state, url, reason in problems:
290
+ table.add_row(state, url, reason)
291
+ console.print(table)
292
+
293
+ runs = sorted((root / "runs").glob("*/manifest.json"), reverse=True)[:5]
294
+ if runs:
295
+ table = Table(title="Recent runs", title_justify="left")
296
+ for col in ("run", "outcome", "pages", "records", "fetched", "duration"):
297
+ table.add_column(col)
298
+ for path in runs:
299
+ m = json.loads(path.read_text(encoding="utf-8"))
300
+ table.add_row(
301
+ m["run_id"],
302
+ m["outcome"],
303
+ str(m["stats"]["pages"]),
304
+ str(m["stats"]["records"]),
305
+ str(m["stats"]["fetched"]),
306
+ f"{m['duration_s']:.0f}s",
307
+ )
308
+ console.print(table)
309
+
310
+
311
+ @app.command()
312
+ def changes(
313
+ source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
314
+ data_dir: DataDir = Path(".harvester"),
315
+ ) -> None:
316
+ """List URLs whose content changed between fetches."""
317
+ cls = load_source(source)
318
+ store = RawStore(data_dir / cls.name / "store")
319
+ rows = store.changed()
320
+ store.close()
321
+ if not rows:
322
+ console.print("no content changes recorded")
323
+ return
324
+ for url, versions in rows:
325
+ console.print(f"{versions:>3} versions {url}")
326
+
327
+
328
+ @app.command()
329
+ def show(
330
+ url: Annotated[str, typer.Argument(help="A URL that was fetched.")],
331
+ source: Annotated[str, typer.Option("--source", "-s", help="Source it was fetched by.")],
332
+ data_dir: DataDir = Path(".harvester"),
333
+ save: Annotated[
334
+ Optional[Path], # noqa: UP045
335
+ typer.Option(help="Write the stored body to this file."),
336
+ ] = None,
337
+ ) -> None:
338
+ """Inspect a stored response (metadata, and optionally the body)."""
339
+ cls = load_source(source)
340
+ store = RawStore(data_dir / cls.name / "store")
341
+ snapshot = store.get(url_fingerprint(url))
342
+ if snapshot is None:
343
+ store.close()
344
+ console.print("[red]not in the raw store[/]")
345
+ raise typer.Exit(1)
346
+ console.print_json(
347
+ json.dumps(
348
+ {
349
+ "url": snapshot.url,
350
+ "final_url": snapshot.final_url,
351
+ "status": snapshot.status,
352
+ "sha256": snapshot.sha256,
353
+ "size": snapshot.size,
354
+ "fetched_at": snapshot.fetched_at.isoformat(),
355
+ "validated_at": snapshot.validated_at.isoformat(),
356
+ "headers": snapshot.headers,
357
+ }
358
+ )
359
+ )
360
+ if save:
361
+ save.write_bytes(store.read_blob(snapshot.sha256))
362
+ console.print(f"saved {snapshot.size} bytes to {save}")
363
+ store.close()
364
+
365
+
366
+ @app.command()
367
+ def robots(
368
+ url: Annotated[str, typer.Argument(help="URL to check.")],
369
+ user_agent: Annotated[str, typer.Option(help="User-Agent to evaluate.")] = DEFAULT_USER_AGENT,
370
+ ) -> None:
371
+ """Explain whether harvester may fetch a URL, per robots.txt (RFC 9309)."""
372
+
373
+ async def check() -> None:
374
+ fetcher = ScraplingFetcher(user_agent)
375
+ policy = RobotsPolicy(lambda u: fetcher.fetch(u, {}), user_agent)
376
+ try:
377
+ decision = await policy.check(url)
378
+ finally:
379
+ await fetcher.close()
380
+ verdict = "[green]allowed[/]" if decision.allowed else "[red]disallowed[/]"
381
+ console.print(f"{verdict} {decision.reason}")
382
+ if decision.crawl_delay:
383
+ console.print(f"crawl-delay: {decision.crawl_delay}s")
384
+
385
+ asyncio.run(check())
386
+
387
+
388
+ _TEMPLATE = '''"""{title} — a harvester source."""
389
+
390
+ from harvester import CachePolicy, DataLicense, Page, Request, Source
391
+
392
+
393
+ class {cls}(Source):
394
+ name = "{name}"
395
+ description = "TODO: one line about what this harvests"
396
+ start_urls = ("https://example.org/",)
397
+ allowed_domains = ("example.org",)
398
+ license = DataLicense(name="TODO: the data's licence", commercial_use=None)
399
+
400
+ def start(self):
401
+ # Listings change, so revalidate them; detail pages are cached forever.
402
+ for url in self.start_urls:
403
+ yield Request(url=url, cache=CachePolicy.REVALIDATE)
404
+
405
+ def parse(self, page: Page):
406
+ for link in page.css("a.item::attr(href)").getall():
407
+ yield page.follow(link, callback="parse_item")
408
+ if next_page := page.css("a.next::attr(href)").get():
409
+ yield page.follow(next_page, cache=CachePolicy.REVALIDATE)
410
+
411
+ def parse_item(self, page: Page):
412
+ yield page.record(
413
+ "item",
414
+ id=page.url,
415
+ title=page.css("h1::text").get(),
416
+ )
417
+ '''
418
+
419
+
420
+ @app.command()
421
+ def new(
422
+ name: Annotated[str, typer.Argument(help="Source name, e.g. my-site.")],
423
+ directory: Annotated[Path, typer.Option(help="Where to write the file.")] = Path(),
424
+ ) -> None:
425
+ """Scaffold a new source file."""
426
+ stem = name.replace("-", "_")
427
+ cls = "".join(part.capitalize() for part in stem.split("_")) + "Source"
428
+ path = directory / f"{stem}.py"
429
+ if path.exists():
430
+ console.print(f"[red]error:[/] {path} already exists")
431
+ raise typer.Exit(1)
432
+ path.write_text(_TEMPLATE.format(title=name, cls=cls, name=name), encoding="utf-8")
433
+ console.print(f"created {path}\nrun it with: [bold]harvester run {path} --limit 5[/]")
434
+
435
+
436
+ if __name__ == "__main__": # pragma: no cover
437
+ sys.exit(app())