harvester-kit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- harvester/__init__.py +22 -0
- harvester/cli.py +437 -0
- harvester/engine.py +421 -0
- harvester/fetch.py +113 -0
- harvester/frontier.py +142 -0
- harvester/models.py +89 -0
- harvester/page.py +97 -0
- harvester/politeness.py +181 -0
- harvester/py.typed +0 -0
- harvester/registry.py +78 -0
- harvester/sinks.py +40 -0
- harvester/source.py +78 -0
- harvester/sources/__init__.py +1 -0
- harvester/sources/dspace.py +153 -0
- harvester/sources/india_code.py +96 -0
- harvester/sources/oai_pmh.py +99 -0
- harvester/store.py +178 -0
- harvester_kit-0.1.0.dist-info/METADATA +231 -0
- harvester_kit-0.1.0.dist-info/RECORD +22 -0
- harvester_kit-0.1.0.dist-info/WHEEL +4 -0
- harvester_kit-0.1.0.dist-info/entry_points.txt +7 -0
- harvester_kit-0.1.0.dist-info/licenses/LICENSE +21 -0
harvester/__init__.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""harvester — polite, reproducible web harvesting.
|
|
2
|
+
|
|
3
|
+
Fetch once, parse forever: every response is kept in a content-addressed
|
|
4
|
+
store, so parsers can be improved and re-run offline without touching the
|
|
5
|
+
network again.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from harvester.models import CachePolicy, DataLicense, Record, Request
|
|
9
|
+
from harvester.page import Page
|
|
10
|
+
from harvester.source import Source
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.0"
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"CachePolicy",
|
|
16
|
+
"DataLicense",
|
|
17
|
+
"Page",
|
|
18
|
+
"Record",
|
|
19
|
+
"Request",
|
|
20
|
+
"Source",
|
|
21
|
+
"__version__",
|
|
22
|
+
]
|
harvester/cli.py
ADDED
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
"""Command-line interface."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
import signal
|
|
8
|
+
import sys
|
|
9
|
+
import time
|
|
10
|
+
from collections.abc import Awaitable, Callable
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from types import FrameType
|
|
13
|
+
from typing import Annotated, Any, Optional
|
|
14
|
+
|
|
15
|
+
import typer
|
|
16
|
+
from rich.console import Console, Group
|
|
17
|
+
from rich.live import Live
|
|
18
|
+
from rich.panel import Panel
|
|
19
|
+
from rich.table import Table
|
|
20
|
+
from rich.text import Text
|
|
21
|
+
|
|
22
|
+
from harvester import __version__
|
|
23
|
+
from harvester.engine import Harvester, RunReport, Settings
|
|
24
|
+
from harvester.fetch import DEFAULT_USER_AGENT, ScraplingFetcher
|
|
25
|
+
from harvester.frontier import Frontier
|
|
26
|
+
from harvester.models import url_fingerprint
|
|
27
|
+
from harvester.politeness import RobotsPolicy
|
|
28
|
+
from harvester.registry import SourceNotFound, installed_sources, load_source
|
|
29
|
+
from harvester.source import Source
|
|
30
|
+
from harvester.store import RawStore
|
|
31
|
+
|
|
32
|
+
app = typer.Typer(
|
|
33
|
+
name="harvester",
|
|
34
|
+
help="Polite, reproducible web harvesting: fetch once, parse forever.",
|
|
35
|
+
no_args_is_help=True,
|
|
36
|
+
rich_markup_mode="rich",
|
|
37
|
+
pretty_exceptions_show_locals=False,
|
|
38
|
+
)
|
|
39
|
+
console = Console()
|
|
40
|
+
|
|
41
|
+
DataDir = Annotated[
|
|
42
|
+
Path, typer.Option("--data-dir", "-d", help="Where stores, queues and runs live.")
|
|
43
|
+
]
|
|
44
|
+
Options = Annotated[
|
|
45
|
+
Optional[list[str]], # noqa: UP045 — typer needs Optional on 3.10
|
|
46
|
+
typer.Option("--option", "-o", help="Source option as key=value (repeatable)."),
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _version(value: bool) -> None:
|
|
51
|
+
if value:
|
|
52
|
+
console.print(f"harvester {__version__}")
|
|
53
|
+
raise typer.Exit()
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@app.callback()
|
|
57
|
+
def main(
|
|
58
|
+
version: Annotated[
|
|
59
|
+
bool, typer.Option("--version", callback=_version, is_eager=True, help="Show version.")
|
|
60
|
+
] = False,
|
|
61
|
+
) -> None:
|
|
62
|
+
"""Polite, reproducible web harvesting: fetch once, parse forever."""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# ── Helpers ─────────────────────────────────────────────────────────────
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _make_source(spec: str, options: list[str] | None) -> Source:
|
|
69
|
+
try:
|
|
70
|
+
cls = load_source(spec)
|
|
71
|
+
except SourceNotFound as exc:
|
|
72
|
+
console.print(f"[red]error:[/] {exc}")
|
|
73
|
+
raise typer.Exit(2) from None
|
|
74
|
+
parsed: dict[str, Any] = {}
|
|
75
|
+
for item in options or []:
|
|
76
|
+
key, sep, value = item.partition("=")
|
|
77
|
+
if not sep:
|
|
78
|
+
console.print(f"[red]error:[/] option {item!r} is not key=value")
|
|
79
|
+
raise typer.Exit(2)
|
|
80
|
+
parsed[key.strip()] = value
|
|
81
|
+
try:
|
|
82
|
+
return cls(**parsed)
|
|
83
|
+
except ValueError as exc:
|
|
84
|
+
console.print(f"[red]error:[/] {exc}")
|
|
85
|
+
raise typer.Exit(2) from None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _human_bytes(n: float) -> str:
|
|
89
|
+
for unit in ("B", "KB", "MB", "GB"):
|
|
90
|
+
if n < 1024:
|
|
91
|
+
return f"{n:.0f} {unit}" if unit == "B" else f"{n:.1f} {unit}"
|
|
92
|
+
n /= 1024
|
|
93
|
+
return f"{n:.1f} TB"
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _dashboard(engine: Harvester, started: float) -> Panel:
|
|
97
|
+
s = engine.stats
|
|
98
|
+
frontier = engine.frontier.counts() if engine.frontier else {}
|
|
99
|
+
elapsed = max(time.monotonic() - started, 1e-6)
|
|
100
|
+
|
|
101
|
+
table = Table.grid(padding=(0, 3))
|
|
102
|
+
for _ in range(4):
|
|
103
|
+
table.add_column(justify="right", style="bold")
|
|
104
|
+
table.add_column(style="dim")
|
|
105
|
+
table.add_row(
|
|
106
|
+
str(s.pages), "pages", str(s.records), "records",
|
|
107
|
+
str(s.fetched), "fetched", str(s.from_cache + s.not_modified), "from cache",
|
|
108
|
+
) # fmt: skip
|
|
109
|
+
table.add_row(
|
|
110
|
+
str(frontier.get("pending", 0)), "queued", str(s.retries), "retries",
|
|
111
|
+
str(s.failed), "failed", str(s.robots_blocked), "blocked",
|
|
112
|
+
) # fmt: skip
|
|
113
|
+
table.add_row(
|
|
114
|
+
_human_bytes(s.bytes), "downloaded", f"{s.pages / elapsed * 60:.0f}", "pages/min",
|
|
115
|
+
f"{elapsed:.0f}s", "elapsed", str(s.offsite + s.missing), "skipped",
|
|
116
|
+
) # fmt: skip
|
|
117
|
+
|
|
118
|
+
parts: list[Any] = [table]
|
|
119
|
+
if s.recent_errors:
|
|
120
|
+
errors = Text("\n".join(list(s.recent_errors)[-4:]), style="yellow", overflow="ellipsis")
|
|
121
|
+
parts += [Text(""), errors]
|
|
122
|
+
title = f"[bold]{engine.source.name}[/] | {engine._mode}"
|
|
123
|
+
return Panel(Group(*parts), title=title, border_style="cyan", expand=False)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _run_with_dashboard(
|
|
127
|
+
engine: Harvester, coro_factory: Callable[[], Awaitable[RunReport]]
|
|
128
|
+
) -> RunReport:
|
|
129
|
+
"""Run the engine with a live view; first Ctrl+C stops gracefully, second aborts."""
|
|
130
|
+
interrupts = 0
|
|
131
|
+
loop: asyncio.AbstractEventLoop | None = None
|
|
132
|
+
|
|
133
|
+
def on_sigint(signum: int, frame: FrameType | None) -> None:
|
|
134
|
+
nonlocal interrupts
|
|
135
|
+
interrupts += 1
|
|
136
|
+
if interrupts == 1 and loop is not None:
|
|
137
|
+
console.print("[yellow]stopping after in-flight requests... (Ctrl+C again to abort)[/]")
|
|
138
|
+
loop.call_soon_threadsafe(engine.stop)
|
|
139
|
+
else:
|
|
140
|
+
raise KeyboardInterrupt
|
|
141
|
+
|
|
142
|
+
async def runner() -> RunReport:
|
|
143
|
+
nonlocal loop
|
|
144
|
+
loop = asyncio.get_running_loop()
|
|
145
|
+
started = time.monotonic()
|
|
146
|
+
with Live(_dashboard(engine, started), console=console, refresh_per_second=4) as live:
|
|
147
|
+
task: asyncio.Future[RunReport] = asyncio.ensure_future(coro_factory())
|
|
148
|
+
while not task.done():
|
|
149
|
+
live.update(_dashboard(engine, started))
|
|
150
|
+
await asyncio.sleep(0.25)
|
|
151
|
+
live.update(_dashboard(engine, started))
|
|
152
|
+
return await task
|
|
153
|
+
|
|
154
|
+
previous = signal.signal(signal.SIGINT, on_sigint)
|
|
155
|
+
try:
|
|
156
|
+
return asyncio.run(runner())
|
|
157
|
+
finally:
|
|
158
|
+
signal.signal(signal.SIGINT, previous)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _print_report(report: RunReport) -> None:
|
|
162
|
+
colour = {"complete": "green", "limit": "cyan", "interrupted": "yellow"}[report.outcome]
|
|
163
|
+
console.print(f"[{colour}]{report.outcome}[/] run [bold]{report.run_id}[/]")
|
|
164
|
+
kinds = ", ".join(f"{n} {k}" for k, n in sorted(report.records_by_kind.items())) or "none"
|
|
165
|
+
console.print(f" records: {kinds}")
|
|
166
|
+
console.print(f" output: {report.records_path}")
|
|
167
|
+
if report.frontier.get("failed") or report.frontier.get("skipped"):
|
|
168
|
+
console.print(
|
|
169
|
+
f" [yellow]{report.frontier.get('failed', 0)} failed, "
|
|
170
|
+
f"{report.frontier.get('skipped', 0)} skipped[/] - see `harvester status`"
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# ── Commands ────────────────────────────────────────────────────────────
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@app.command("list")
|
|
178
|
+
def list_sources() -> None:
|
|
179
|
+
"""Show installed sources."""
|
|
180
|
+
table = Table(title="Installed sources", title_justify="left")
|
|
181
|
+
table.add_column("name", style="bold cyan")
|
|
182
|
+
table.add_column("description")
|
|
183
|
+
table.add_column("data licence", style="dim")
|
|
184
|
+
for name, cls in installed_sources().items():
|
|
185
|
+
table.add_row(name, cls.description, cls.license.name if cls.license else "-")
|
|
186
|
+
console.print(table)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
@app.command()
|
|
190
|
+
def run(
|
|
191
|
+
source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
|
|
192
|
+
option: Options = None,
|
|
193
|
+
data_dir: DataDir = Path(".harvester"),
|
|
194
|
+
limit: Annotated[Optional[int], typer.Option(help="Stop after this many pages.")] = None, # noqa: UP045
|
|
195
|
+
concurrency: Annotated[int, typer.Option(help="Concurrent workers in total.")] = 4,
|
|
196
|
+
delay: Annotated[float, typer.Option(help="Seconds between requests to one host.")] = 1.0,
|
|
197
|
+
refresh: Annotated[
|
|
198
|
+
bool, typer.Option(help="Revalidate cached responses (conditional requests).")
|
|
199
|
+
] = False,
|
|
200
|
+
restart: Annotated[
|
|
201
|
+
bool, typer.Option(help="Discard unfinished progress and start a new pass.")
|
|
202
|
+
] = False,
|
|
203
|
+
user_agent: Annotated[str, typer.Option(help="User-Agent; keep a contact URL in it.")] = (
|
|
204
|
+
DEFAULT_USER_AGENT
|
|
205
|
+
),
|
|
206
|
+
robots_unavailable: Annotated[
|
|
207
|
+
Optional[str], # noqa: UP045
|
|
208
|
+
typer.Option(
|
|
209
|
+
help="Override the source's policy when robots.txt is unreachable: allow|disallow."
|
|
210
|
+
),
|
|
211
|
+
] = None,
|
|
212
|
+
) -> None:
|
|
213
|
+
"""Crawl a source. Resumes automatically if the last run was interrupted."""
|
|
214
|
+
src = _make_source(source, option)
|
|
215
|
+
if robots_unavailable not in (None, "allow", "disallow"):
|
|
216
|
+
console.print("[red]error:[/] --robots-unavailable must be allow or disallow")
|
|
217
|
+
raise typer.Exit(2)
|
|
218
|
+
settings = Settings(
|
|
219
|
+
data_dir=data_dir,
|
|
220
|
+
concurrency=concurrency,
|
|
221
|
+
delay=delay,
|
|
222
|
+
limit=limit,
|
|
223
|
+
refresh=refresh,
|
|
224
|
+
restart=restart,
|
|
225
|
+
user_agent=user_agent,
|
|
226
|
+
robots_unavailable=robots_unavailable,
|
|
227
|
+
)
|
|
228
|
+
engine = Harvester(src, settings)
|
|
229
|
+
try:
|
|
230
|
+
report = _run_with_dashboard(engine, engine.run)
|
|
231
|
+
finally:
|
|
232
|
+
engine.close()
|
|
233
|
+
_print_report(report)
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
@app.command()
|
|
237
|
+
def reparse(
|
|
238
|
+
source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
|
|
239
|
+
option: Options = None,
|
|
240
|
+
data_dir: DataDir = Path(".harvester"),
|
|
241
|
+
) -> None:
|
|
242
|
+
"""Re-run parsers over the raw store. Makes no network requests."""
|
|
243
|
+
src = _make_source(source, option)
|
|
244
|
+
engine = Harvester(src, Settings(data_dir=data_dir, concurrency=1))
|
|
245
|
+
try:
|
|
246
|
+
report = _run_with_dashboard(engine, engine.reparse)
|
|
247
|
+
finally:
|
|
248
|
+
engine.close()
|
|
249
|
+
_print_report(report)
|
|
250
|
+
if report.stats.get("missing"):
|
|
251
|
+
console.print(
|
|
252
|
+
f" [yellow]{report.stats['missing']} requests were never fetched; "
|
|
253
|
+
"run the crawl to fill them in[/]"
|
|
254
|
+
)
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
@app.command()
|
|
258
|
+
def status(
|
|
259
|
+
source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
|
|
260
|
+
data_dir: DataDir = Path(".harvester"),
|
|
261
|
+
) -> None:
|
|
262
|
+
"""Show queue, store and recent runs for a source."""
|
|
263
|
+
cls = load_source(source)
|
|
264
|
+
root = data_dir / cls.name
|
|
265
|
+
if not root.exists():
|
|
266
|
+
console.print(f"no data for {cls.name} in {data_dir}")
|
|
267
|
+
raise typer.Exit(1)
|
|
268
|
+
|
|
269
|
+
store = RawStore(root / "store")
|
|
270
|
+
stats = store.stats()
|
|
271
|
+
store.close()
|
|
272
|
+
console.print(
|
|
273
|
+
f"[bold]{cls.name}[/] raw store: {stats['responses']} responses, "
|
|
274
|
+
f"{stats['unique_bodies']} unique bodies, {_human_bytes(stats['bytes'])}"
|
|
275
|
+
)
|
|
276
|
+
|
|
277
|
+
frontier_path = root / "frontier.sqlite"
|
|
278
|
+
if frontier_path.exists():
|
|
279
|
+
frontier = Frontier(frontier_path)
|
|
280
|
+
counts = frontier.counts()
|
|
281
|
+
console.print("queue: " + ", ".join(f"{n} {s}" for s, n in sorted(counts.items())))
|
|
282
|
+
problems = frontier.problems()
|
|
283
|
+
frontier.close()
|
|
284
|
+
if problems:
|
|
285
|
+
table = Table(title="Failed / skipped", title_justify="left")
|
|
286
|
+
table.add_column("state")
|
|
287
|
+
table.add_column("url", overflow="fold")
|
|
288
|
+
table.add_column("reason", style="yellow", overflow="fold")
|
|
289
|
+
for state, url, reason in problems:
|
|
290
|
+
table.add_row(state, url, reason)
|
|
291
|
+
console.print(table)
|
|
292
|
+
|
|
293
|
+
runs = sorted((root / "runs").glob("*/manifest.json"), reverse=True)[:5]
|
|
294
|
+
if runs:
|
|
295
|
+
table = Table(title="Recent runs", title_justify="left")
|
|
296
|
+
for col in ("run", "outcome", "pages", "records", "fetched", "duration"):
|
|
297
|
+
table.add_column(col)
|
|
298
|
+
for path in runs:
|
|
299
|
+
m = json.loads(path.read_text(encoding="utf-8"))
|
|
300
|
+
table.add_row(
|
|
301
|
+
m["run_id"],
|
|
302
|
+
m["outcome"],
|
|
303
|
+
str(m["stats"]["pages"]),
|
|
304
|
+
str(m["stats"]["records"]),
|
|
305
|
+
str(m["stats"]["fetched"]),
|
|
306
|
+
f"{m['duration_s']:.0f}s",
|
|
307
|
+
)
|
|
308
|
+
console.print(table)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
@app.command()
|
|
312
|
+
def changes(
|
|
313
|
+
source: Annotated[str, typer.Argument(help="Source name, or path/to/source.py[:Class].")],
|
|
314
|
+
data_dir: DataDir = Path(".harvester"),
|
|
315
|
+
) -> None:
|
|
316
|
+
"""List URLs whose content changed between fetches."""
|
|
317
|
+
cls = load_source(source)
|
|
318
|
+
store = RawStore(data_dir / cls.name / "store")
|
|
319
|
+
rows = store.changed()
|
|
320
|
+
store.close()
|
|
321
|
+
if not rows:
|
|
322
|
+
console.print("no content changes recorded")
|
|
323
|
+
return
|
|
324
|
+
for url, versions in rows:
|
|
325
|
+
console.print(f"{versions:>3} versions {url}")
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
@app.command()
|
|
329
|
+
def show(
|
|
330
|
+
url: Annotated[str, typer.Argument(help="A URL that was fetched.")],
|
|
331
|
+
source: Annotated[str, typer.Option("--source", "-s", help="Source it was fetched by.")],
|
|
332
|
+
data_dir: DataDir = Path(".harvester"),
|
|
333
|
+
save: Annotated[
|
|
334
|
+
Optional[Path], # noqa: UP045
|
|
335
|
+
typer.Option(help="Write the stored body to this file."),
|
|
336
|
+
] = None,
|
|
337
|
+
) -> None:
|
|
338
|
+
"""Inspect a stored response (metadata, and optionally the body)."""
|
|
339
|
+
cls = load_source(source)
|
|
340
|
+
store = RawStore(data_dir / cls.name / "store")
|
|
341
|
+
snapshot = store.get(url_fingerprint(url))
|
|
342
|
+
if snapshot is None:
|
|
343
|
+
store.close()
|
|
344
|
+
console.print("[red]not in the raw store[/]")
|
|
345
|
+
raise typer.Exit(1)
|
|
346
|
+
console.print_json(
|
|
347
|
+
json.dumps(
|
|
348
|
+
{
|
|
349
|
+
"url": snapshot.url,
|
|
350
|
+
"final_url": snapshot.final_url,
|
|
351
|
+
"status": snapshot.status,
|
|
352
|
+
"sha256": snapshot.sha256,
|
|
353
|
+
"size": snapshot.size,
|
|
354
|
+
"fetched_at": snapshot.fetched_at.isoformat(),
|
|
355
|
+
"validated_at": snapshot.validated_at.isoformat(),
|
|
356
|
+
"headers": snapshot.headers,
|
|
357
|
+
}
|
|
358
|
+
)
|
|
359
|
+
)
|
|
360
|
+
if save:
|
|
361
|
+
save.write_bytes(store.read_blob(snapshot.sha256))
|
|
362
|
+
console.print(f"saved {snapshot.size} bytes to {save}")
|
|
363
|
+
store.close()
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
@app.command()
|
|
367
|
+
def robots(
|
|
368
|
+
url: Annotated[str, typer.Argument(help="URL to check.")],
|
|
369
|
+
user_agent: Annotated[str, typer.Option(help="User-Agent to evaluate.")] = DEFAULT_USER_AGENT,
|
|
370
|
+
) -> None:
|
|
371
|
+
"""Explain whether harvester may fetch a URL, per robots.txt (RFC 9309)."""
|
|
372
|
+
|
|
373
|
+
async def check() -> None:
|
|
374
|
+
fetcher = ScraplingFetcher(user_agent)
|
|
375
|
+
policy = RobotsPolicy(lambda u: fetcher.fetch(u, {}), user_agent)
|
|
376
|
+
try:
|
|
377
|
+
decision = await policy.check(url)
|
|
378
|
+
finally:
|
|
379
|
+
await fetcher.close()
|
|
380
|
+
verdict = "[green]allowed[/]" if decision.allowed else "[red]disallowed[/]"
|
|
381
|
+
console.print(f"{verdict} {decision.reason}")
|
|
382
|
+
if decision.crawl_delay:
|
|
383
|
+
console.print(f"crawl-delay: {decision.crawl_delay}s")
|
|
384
|
+
|
|
385
|
+
asyncio.run(check())
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
_TEMPLATE = '''"""{title} — a harvester source."""
|
|
389
|
+
|
|
390
|
+
from harvester import CachePolicy, DataLicense, Page, Request, Source
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
class {cls}(Source):
|
|
394
|
+
name = "{name}"
|
|
395
|
+
description = "TODO: one line about what this harvests"
|
|
396
|
+
start_urls = ("https://example.org/",)
|
|
397
|
+
allowed_domains = ("example.org",)
|
|
398
|
+
license = DataLicense(name="TODO: the data's licence", commercial_use=None)
|
|
399
|
+
|
|
400
|
+
def start(self):
|
|
401
|
+
# Listings change, so revalidate them; detail pages are cached forever.
|
|
402
|
+
for url in self.start_urls:
|
|
403
|
+
yield Request(url=url, cache=CachePolicy.REVALIDATE)
|
|
404
|
+
|
|
405
|
+
def parse(self, page: Page):
|
|
406
|
+
for link in page.css("a.item::attr(href)").getall():
|
|
407
|
+
yield page.follow(link, callback="parse_item")
|
|
408
|
+
if next_page := page.css("a.next::attr(href)").get():
|
|
409
|
+
yield page.follow(next_page, cache=CachePolicy.REVALIDATE)
|
|
410
|
+
|
|
411
|
+
def parse_item(self, page: Page):
|
|
412
|
+
yield page.record(
|
|
413
|
+
"item",
|
|
414
|
+
id=page.url,
|
|
415
|
+
title=page.css("h1::text").get(),
|
|
416
|
+
)
|
|
417
|
+
'''
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
@app.command()
|
|
421
|
+
def new(
|
|
422
|
+
name: Annotated[str, typer.Argument(help="Source name, e.g. my-site.")],
|
|
423
|
+
directory: Annotated[Path, typer.Option(help="Where to write the file.")] = Path(),
|
|
424
|
+
) -> None:
|
|
425
|
+
"""Scaffold a new source file."""
|
|
426
|
+
stem = name.replace("-", "_")
|
|
427
|
+
cls = "".join(part.capitalize() for part in stem.split("_")) + "Source"
|
|
428
|
+
path = directory / f"{stem}.py"
|
|
429
|
+
if path.exists():
|
|
430
|
+
console.print(f"[red]error:[/] {path} already exists")
|
|
431
|
+
raise typer.Exit(1)
|
|
432
|
+
path.write_text(_TEMPLATE.format(title=name, cls=cls, name=name), encoding="utf-8")
|
|
433
|
+
console.print(f"created {path}\nrun it with: [bold]harvester run {path} --limit 5[/]")
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
if __name__ == "__main__": # pragma: no cover
|
|
437
|
+
sys.exit(app())
|