harvester-kit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- harvester_kit-0.1.0/.github/workflows/ci.yml +35 -0
- harvester_kit-0.1.0/.github/workflows/release.yml +39 -0
- harvester_kit-0.1.0/.gitignore +14 -0
- harvester_kit-0.1.0/CHANGELOG.md +14 -0
- harvester_kit-0.1.0/CONTRIBUTING.md +10 -0
- harvester_kit-0.1.0/LICENSE +21 -0
- harvester_kit-0.1.0/PKG-INFO +231 -0
- harvester_kit-0.1.0/README.md +189 -0
- harvester_kit-0.1.0/docs/writing-sources.md +96 -0
- harvester_kit-0.1.0/examples/quotes.py +45 -0
- harvester_kit-0.1.0/pyproject.toml +96 -0
- harvester_kit-0.1.0/src/harvester/__init__.py +22 -0
- harvester_kit-0.1.0/src/harvester/cli.py +437 -0
- harvester_kit-0.1.0/src/harvester/engine.py +421 -0
- harvester_kit-0.1.0/src/harvester/fetch.py +113 -0
- harvester_kit-0.1.0/src/harvester/frontier.py +142 -0
- harvester_kit-0.1.0/src/harvester/models.py +89 -0
- harvester_kit-0.1.0/src/harvester/page.py +97 -0
- harvester_kit-0.1.0/src/harvester/politeness.py +181 -0
- harvester_kit-0.1.0/src/harvester/py.typed +0 -0
- harvester_kit-0.1.0/src/harvester/registry.py +78 -0
- harvester_kit-0.1.0/src/harvester/sinks.py +40 -0
- harvester_kit-0.1.0/src/harvester/source.py +78 -0
- harvester_kit-0.1.0/src/harvester/sources/__init__.py +1 -0
- harvester_kit-0.1.0/src/harvester/sources/dspace.py +153 -0
- harvester_kit-0.1.0/src/harvester/sources/india_code.py +96 -0
- harvester_kit-0.1.0/src/harvester/sources/oai_pmh.py +99 -0
- harvester_kit-0.1.0/src/harvester/store.py +178 -0
- harvester_kit-0.1.0/tests/__init__.py +0 -0
- harvester_kit-0.1.0/tests/conftest.py +86 -0
- harvester_kit-0.1.0/tests/fixtures/india_code/bns_text_head.txt +348 -0
- harvester_kit-0.1.0/tests/fixtures/india_code/bundles_bns.json +468 -0
- harvester_kit-0.1.0/tests/fixtures/india_code/search_page.json +1221 -0
- harvester_kit-0.1.0/tests/test_cli.py +91 -0
- harvester_kit-0.1.0/tests/test_engine.py +246 -0
- harvester_kit-0.1.0/tests/test_sources.py +202 -0
- harvester_kit-0.1.0/tests/test_units.py +171 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
lint:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v7
|
|
13
|
+
- uses: actions/setup-python@v7
|
|
14
|
+
with:
|
|
15
|
+
python-version: "3.13"
|
|
16
|
+
- run: pip install -e ".[dev]"
|
|
17
|
+
- run: ruff check .
|
|
18
|
+
- run: ruff format --check .
|
|
19
|
+
- run: mypy src
|
|
20
|
+
|
|
21
|
+
test:
|
|
22
|
+
strategy:
|
|
23
|
+
fail-fast: false
|
|
24
|
+
matrix:
|
|
25
|
+
os: [ubuntu-latest, windows-latest, macos-latest]
|
|
26
|
+
python: ["3.10", "3.11", "3.12", "3.13", "3.14"]
|
|
27
|
+
runs-on: ${{ matrix.os }}
|
|
28
|
+
steps:
|
|
29
|
+
- uses: actions/checkout@v7
|
|
30
|
+
- uses: actions/setup-python@v7
|
|
31
|
+
with:
|
|
32
|
+
python-version: ${{ matrix.python }}
|
|
33
|
+
allow-prereleases: true
|
|
34
|
+
- run: pip install -e ".[dev]"
|
|
35
|
+
- run: pytest -q
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
name: Release
|
|
2
|
+
|
|
3
|
+
# Publishing a GitHub release builds the package and uploads it to PyPI using
|
|
4
|
+
# Trusted Publishing (OIDC): no API token is stored anywhere.
|
|
5
|
+
|
|
6
|
+
on:
|
|
7
|
+
release:
|
|
8
|
+
types: [published]
|
|
9
|
+
|
|
10
|
+
jobs:
|
|
11
|
+
build:
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v7
|
|
15
|
+
- uses: actions/setup-python@v7
|
|
16
|
+
with:
|
|
17
|
+
python-version: "3.13"
|
|
18
|
+
- run: pip install build twine
|
|
19
|
+
- run: python -m build
|
|
20
|
+
- run: twine check --strict dist/*
|
|
21
|
+
- uses: actions/upload-artifact@v7
|
|
22
|
+
with:
|
|
23
|
+
name: dist
|
|
24
|
+
path: dist/
|
|
25
|
+
|
|
26
|
+
publish:
|
|
27
|
+
needs: build
|
|
28
|
+
runs-on: ubuntu-latest
|
|
29
|
+
environment:
|
|
30
|
+
name: pypi
|
|
31
|
+
url: https://pypi.org/project/harvester-kit/
|
|
32
|
+
permissions:
|
|
33
|
+
id-token: write
|
|
34
|
+
steps:
|
|
35
|
+
- uses: actions/download-artifact@v8
|
|
36
|
+
with:
|
|
37
|
+
name: dist
|
|
38
|
+
path: dist/
|
|
39
|
+
- uses: pypa/gh-action-pypi-publish@v1.14.2
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0 — 2026-10-03
|
|
4
|
+
|
|
5
|
+
First release.
|
|
6
|
+
|
|
7
|
+
- Crawl engine with a persistent SQLite frontier: resumable, de-duplicated, prioritised.
|
|
8
|
+
- Content-addressed raw store with fetch history; offline `reparse`.
|
|
9
|
+
- Per-request cache policies (prefer / revalidate / bypass) with conditional requests.
|
|
10
|
+
- robots.txt per RFC 9309, Crawl-delay, per-host throttling with adaptive back-off and Retry-After.
|
|
11
|
+
- Honest Scrapling-based fetcher (no impersonation, real User-Agent).
|
|
12
|
+
- JSONL output with per-record provenance and data licence; run manifests.
|
|
13
|
+
- Built-in sources: OAI-PMH, DSpace 7+, India Code.
|
|
14
|
+
- CLI: run, reparse, status, changes, show, robots, list, new.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Thanks for helping. A few ground rules keep harvester trustworthy:
|
|
4
|
+
|
|
5
|
+
1. **Politeness is not optional.** Changes must not add ways to evade blocks:
|
|
6
|
+
no fingerprint spoofing, CAPTCHA solving or block-dodging proxy rotation.
|
|
7
|
+
2. **Callbacks stay pure.** Anything that would let a parse callback do I/O breaks offline re-parsing.
|
|
8
|
+
3. **Tests come with changes.** `pytest`, `ruff check .`, `ruff format --check .` and `mypy src` must pass.
|
|
9
|
+
|
|
10
|
+
New sources for public bulk-data protocols (CKAN, Zenodo, sitemaps …) are especially welcome.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sarthak Pahwa
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: harvester-kit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Polite, reproducible web harvesting: fetch once, parse forever.
|
|
5
|
+
Project-URL: Homepage, https://github.com/sarthak213/harvester
|
|
6
|
+
Project-URL: Issues, https://github.com/sarthak213/harvester/issues
|
|
7
|
+
Project-URL: Changelog, https://github.com/sarthak213/harvester/blob/main/CHANGELOG.md
|
|
8
|
+
Author: Sarthak Pahwa
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: crawling,data-pipeline,dspace,harvesting,oai-pmh,robots.txt,scraping
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Framework :: AsyncIO
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Intended Audience :: Science/Research
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
25
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
|
|
26
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
27
|
+
Classifier: Typing :: Typed
|
|
28
|
+
Requires-Python: >=3.10
|
|
29
|
+
Requires-Dist: orjson>=3.10
|
|
30
|
+
Requires-Dist: protego>=0.4
|
|
31
|
+
Requires-Dist: pydantic>=2.7
|
|
32
|
+
Requires-Dist: rich>=13.7
|
|
33
|
+
Requires-Dist: scrapling[fetchers]>=0.4.15
|
|
34
|
+
Requires-Dist: typer>=0.12
|
|
35
|
+
Requires-Dist: w3lib>=2.1
|
|
36
|
+
Provides-Extra: dev
|
|
37
|
+
Requires-Dist: mypy>=1.11; extra == 'dev'
|
|
38
|
+
Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
|
|
39
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
40
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
41
|
+
Description-Content-Type: text/markdown
|
|
42
|
+
|
|
43
|
+
<div align="center">
|
|
44
|
+
|
|
45
|
+
# harvester
|
|
46
|
+
|
|
47
|
+
**Polite, reproducible web harvesting. Fetch once, parse forever.**
|
|
48
|
+
|
|
49
|
+
[](https://github.com/sarthak213/harvester/actions/workflows/ci.yml)
|
|
50
|
+
[](https://pypi.org/project/harvester-kit/)
|
|
51
|
+
[](https://www.python.org)
|
|
52
|
+
[](LICENSE)
|
|
53
|
+
[](https://mypy-lang.org)
|
|
54
|
+
|
|
55
|
+
</div>
|
|
56
|
+
|
|
57
|
+
Most scrapers are throwaway scripts. They re-download everything whenever a parser changes, start over after a crash, hammer servers until they get blocked, and produce data nobody can trace back to its source.
|
|
58
|
+
|
|
59
|
+
**harvester** treats harvesting as a data pipeline. Every response goes into a content-addressed store, so parsers can be fixed and re-run **offline**. The crawl queue lives on disk, so interrupted runs **resume** where they stopped. Politeness is built into the engine, not left as an afterthought. Every record carries its **provenance and licence**.
|
|
60
|
+
|
|
61
|
+
It uses [Scrapling](https://github.com/D4Vinci/Scrapling) for fetching and parsing, and adds the engine around it.
|
|
62
|
+
|
|
63
|
+
```text
|
|
64
|
+
┌──────────────────────────── quotes | crawl ─────────────────────────────┐
|
|
65
|
+
│ 60 pages 150 records 60 fetched 0 from cache │
|
|
66
|
+
│ 0 queued 0 retries 0 failed 0 blocked │
|
|
67
|
+
│ 272.5 KB downloaded 93 pages/min 39s elapsed 0 skipped │
|
|
68
|
+
└─────────────────────────────────────────────────────────────────────────┘
|
|
69
|
+
complete run 20261003T044028Z
|
|
70
|
+
records: 50 author, 100 quote
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## Highlights
|
|
74
|
+
|
|
75
|
+
| | |
|
|
76
|
+
|---|---|
|
|
77
|
+
| **Fetch once, parse forever** | Raw responses are stored gzip-compressed and keyed by SHA-256. `harvester reparse` replays the whole crawl graph from disk with **zero network requests**, so fixing a parser never costs another crawl. |
|
|
78
|
+
| **Resumable by design** | The frontier is a SQLite queue. Ctrl+C (or a crash, or a reboot) loses nothing; the next `run` continues where it stopped. |
|
|
79
|
+
| **Polite by default** | robots.txt per RFC 9309, including its 4xx/5xx rules and `Crawl-delay`. Per-host rate limits and concurrency caps. `Retry-After` honoured. The delay adapts, backing off on 429/503 and recovering slowly. |
|
|
80
|
+
| **Honest** | Requests carry a real User-Agent with a contact URL. Scrapling's browser impersonation and header forging are **turned off**. |
|
|
81
|
+
| **Smart caching** | Each request picks a cache policy: documents are fetched once, listings are revalidated with `If-None-Match` / `If-Modified-Since`, and a `304` reuses the stored body. |
|
|
82
|
+
| **Provenance and licensing** | Every record states its source URL, fetch time, content hash, parser version and the data's licence. Every run writes a manifest. |
|
|
83
|
+
| **Change tracking** | The store keeps a history of every fetch. `harvester changes` lists pages whose content changed between crawls. |
|
|
84
|
+
| **Bulk-data connectors** | Built-in **OAI-PMH** and **DSpace 7+** sources cover thousands of libraries, archives and government repositories. Downloads are verified against published MD5s. |
|
|
85
|
+
| **Small API** | A source is a class with a start URL and a parse method. Load one from a file, or publish it as a plugin through entry points. |
|
|
86
|
+
|
|
87
|
+
## Quick start
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
pip install harvester-kit # installs the `harvester` command and package
|
|
91
|
+
|
|
92
|
+
harvester run examples/quotes.py # crawl the scraping sandbox
|
|
93
|
+
harvester reparse examples/quotes.py # rebuild every record offline
|
|
94
|
+
harvester status examples/quotes.py # queue, store and run history
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Output lands in `.harvester/<source>/runs/<run-id>/records.jsonl`, one record per line:
|
|
98
|
+
|
|
99
|
+
```json
|
|
100
|
+
{
|
|
101
|
+
"kind": "author",
|
|
102
|
+
"id": "Albert-Einstein",
|
|
103
|
+
"data": {"name": "Albert Einstein", "born": "March 14, 1879", "born_in": "in Ulm, Germany"},
|
|
104
|
+
"provenance": {
|
|
105
|
+
"source": "quotes",
|
|
106
|
+
"url": "https://quotes.toscrape.com/author/Albert-Einstein/",
|
|
107
|
+
"fetched_at": "2026-10-03T04:41:36Z",
|
|
108
|
+
"sha256": "9d1c…",
|
|
109
|
+
"status": 200,
|
|
110
|
+
"parser_version": "1",
|
|
111
|
+
"harvester_version": "0.1.0",
|
|
112
|
+
"license": {"name": "Scraping sandbox by Zyte; sample data"}
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
## Writing a source
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from harvester import CachePolicy, Page, Request, Source
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class Quotes(Source):
|
|
124
|
+
name = "quotes"
|
|
125
|
+
start_urls = ("https://quotes.toscrape.com/",)
|
|
126
|
+
allowed_domains = ("quotes.toscrape.com",)
|
|
127
|
+
|
|
128
|
+
def start(self):
|
|
129
|
+
# Listings change, so revalidate them on each pass.
|
|
130
|
+
yield Request(url=self.start_urls[0], cache=CachePolicy.REVALIDATE)
|
|
131
|
+
|
|
132
|
+
def parse(self, page: Page):
|
|
133
|
+
for quote in page.css("div.quote"):
|
|
134
|
+
yield page.record(
|
|
135
|
+
"quote",
|
|
136
|
+
id=quote.css("span.text::text").get(),
|
|
137
|
+
author=quote.css("small.author::text").get(),
|
|
138
|
+
tags=quote.css("a.tag::text").getall(),
|
|
139
|
+
)
|
|
140
|
+
if next_href := page.css("li.next a::attr(href)").get():
|
|
141
|
+
yield page.follow(next_href, cache=CachePolicy.REVALIDATE)
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
`harvester new my-site` scaffolds one. Callbacks receive a `Page`, which offers CSS/XPath via Scrapling, `.json()` and `.text`. They yield `Request`s to follow and `Record`s to keep. The one rule: **callbacks must not do their own I/O**. Keeping them pure is what makes offline re-parsing exact.
|
|
145
|
+
|
|
146
|
+
See [docs/writing-sources.md](docs/writing-sources.md) for options, multiple callbacks, file downloads and packaging a source as a plugin.
|
|
147
|
+
|
|
148
|
+
## How it works
|
|
149
|
+
|
|
150
|
+
```mermaid
|
|
151
|
+
flowchart LR
|
|
152
|
+
S[Source.start] --> F[(Frontier<br/>SQLite queue)]
|
|
153
|
+
F -->|claim| R[Scope + robots.txt check]
|
|
154
|
+
R --> C{In raw store?}
|
|
155
|
+
C -->|PREFER| PG[Page]
|
|
156
|
+
C -->|REVALIDATE / miss| T[Host throttle] --> H[Scrapling fetch]
|
|
157
|
+
H -->|200| ST[(Raw store<br/>SHA-256 blobs)] --> PG
|
|
158
|
+
H -->|304| PG
|
|
159
|
+
H -->|429 / 503 / 5xx| B[Back off + retry] --> F
|
|
160
|
+
PG --> CB[Parse callback]
|
|
161
|
+
CB -->|Request| F
|
|
162
|
+
CB -->|Record + provenance| O[(records.jsonl<br/>+ manifest)]
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
```text
|
|
166
|
+
.harvester/<source>/
|
|
167
|
+
├── frontier.sqlite # queue: pending / done / failed / skipped, retries, priorities
|
|
168
|
+
├── store/
|
|
169
|
+
│ ├── index.sqlite # latest response per URL + full fetch history
|
|
170
|
+
│ └── blobs/ab/cd/…gz # bodies, content-addressed and deduplicated
|
|
171
|
+
└── runs/<run-id>/
|
|
172
|
+
├── records.jsonl # output
|
|
173
|
+
└── manifest.json # settings, stats, licence, outcome
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
## Built-in sources
|
|
177
|
+
|
|
178
|
+
| Source | What it harvests |
|
|
179
|
+
|---|---|
|
|
180
|
+
| `oai-pmh` | Any [OAI-PMH 2.0](https://www.openarchives.org/OAI/openarchivesprotocol.html) repository: `ListRecords` with resumption tokens, deleted records, any metadata format. `-o base_url=… -o set=… -o metadata_prefix=…` |
|
|
181
|
+
| `dspace` | Any DSpace 7+ repository via its REST API: items, then files from chosen bundles, MD5-verified, with text files inlined. `-o base_url=… -o scope=<uuid> -o bundles=ORIGINAL` |
|
|
182
|
+
| `india-code` | India's central Acts from [India Code](https://indiacode.gov.in): clean act records (number, year, ministry, enforcement date, repeal status) plus the official text extraction. `-o in_force_only=true -o pdf=true` |
|
|
183
|
+
|
|
184
|
+
## CLI
|
|
185
|
+
|
|
186
|
+
| Command | |
|
|
187
|
+
|---|---|
|
|
188
|
+
| `harvester run SOURCE [-o k=v] [--limit N] [--delay S] [--refresh] [--restart]` | Crawl. Resumes interrupted work; otherwise starts a new pass that reuses the cache. |
|
|
189
|
+
| `harvester reparse SOURCE` | Re-run parsers over the raw store. No network. |
|
|
190
|
+
| `harvester status SOURCE` | Queue counts, store size, failures with reasons, recent runs. |
|
|
191
|
+
| `harvester changes SOURCE` | URLs whose content changed between fetches. |
|
|
192
|
+
| `harvester show URL -s SOURCE [--save FILE]` | Inspect or extract a stored response. |
|
|
193
|
+
| `harvester robots URL` | Explain whether a URL may be fetched, and why. |
|
|
194
|
+
| `harvester list` / `harvester new NAME` | Installed sources / scaffold a new one. |
|
|
195
|
+
|
|
196
|
+
`SOURCE` is an installed source name, `path/to/file.py`, or `path/to/file.py:ClassName`.
|
|
197
|
+
|
|
198
|
+
## Politeness, precisely
|
|
199
|
+
|
|
200
|
+
harvester is meant for collecting data you are entitled to collect, without being a burden on the sites that host it.
|
|
201
|
+
|
|
202
|
+
- **robots.txt (RFC 9309).**
|
|
203
|
+
- `2xx`: rules and `Crawl-delay` are obeyed.
|
|
204
|
+
- `4xx`: no restrictions.
|
|
205
|
+
- `5xx` or network error: **complete disallow**.
|
|
206
|
+
- A source may opt out of that last rule only by declaring `robots_unavailable = "allow"` *with a written reason*. The choice is recorded in every run manifest.
|
|
207
|
+
- **Rate limits.** Per-host minimum delay (default 1 s) and concurrency (default 1). On `429`/`503` the delay doubles, honouring `Retry-After`. It recovers by 10% per healthy response.
|
|
208
|
+
- **Identity.** `harvester/<version> (+https://github.com/sarthak213/harvester)`. Change it with `--user-agent`, but keep a contact URL in it.
|
|
209
|
+
- **No evasion.** No browser fingerprint spoofing, no CAPTCHA solving, no proxy rotation to get around blocks. If a site says no, harvester listens.
|
|
210
|
+
|
|
211
|
+
## Development
|
|
212
|
+
|
|
213
|
+
```bash
|
|
214
|
+
git clone https://github.com/sarthak213/harvester && cd harvester
|
|
215
|
+
python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
216
|
+
pip install -e ".[dev]"
|
|
217
|
+
pytest # 41 tests, including a real HTTP server for end-to-end runs
|
|
218
|
+
ruff check . && ruff format --check . && mypy src
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
## Roadmap
|
|
222
|
+
|
|
223
|
+
- Browser rendering for JavaScript-only pages (Scrapling `DynamicFetcher`), opt-in per request
|
|
224
|
+
- Sitemap and RSS/Atom discovery sources
|
|
225
|
+
- Parquet / SQLite export with record de-duplication across runs
|
|
226
|
+
- Scheduled incremental runs and change notifications
|
|
227
|
+
- More open-data connectors: CKAN, Zenodo, S3 open-data buckets
|
|
228
|
+
|
|
229
|
+
## License
|
|
230
|
+
|
|
231
|
+
MIT. See [LICENSE](LICENSE). The licence covers harvester's code. **The data you harvest has its own terms**; sources declare them, and harvester records them in every output.
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
|
|
3
|
+
# harvester
|
|
4
|
+
|
|
5
|
+
**Polite, reproducible web harvesting. Fetch once, parse forever.**
|
|
6
|
+
|
|
7
|
+
[](https://github.com/sarthak213/harvester/actions/workflows/ci.yml)
|
|
8
|
+
[](https://pypi.org/project/harvester-kit/)
|
|
9
|
+
[](https://www.python.org)
|
|
10
|
+
[](LICENSE)
|
|
11
|
+
[](https://mypy-lang.org)
|
|
12
|
+
|
|
13
|
+
</div>
|
|
14
|
+
|
|
15
|
+
Most scrapers are throwaway scripts. They re-download everything whenever a parser changes, start over after a crash, hammer servers until they get blocked, and produce data nobody can trace back to its source.
|
|
16
|
+
|
|
17
|
+
**harvester** treats harvesting as a data pipeline. Every response goes into a content-addressed store, so parsers can be fixed and re-run **offline**. The crawl queue lives on disk, so interrupted runs **resume** where they stopped. Politeness is built into the engine, not left as an afterthought. Every record carries its **provenance and licence**.
|
|
18
|
+
|
|
19
|
+
It uses [Scrapling](https://github.com/D4Vinci/Scrapling) for fetching and parsing, and adds the engine around it.
|
|
20
|
+
|
|
21
|
+
```text
|
|
22
|
+
┌──────────────────────────── quotes | crawl ─────────────────────────────┐
|
|
23
|
+
│ 60 pages 150 records 60 fetched 0 from cache │
|
|
24
|
+
│ 0 queued 0 retries 0 failed 0 blocked │
|
|
25
|
+
│ 272.5 KB downloaded 93 pages/min 39s elapsed 0 skipped │
|
|
26
|
+
└─────────────────────────────────────────────────────────────────────────┘
|
|
27
|
+
complete run 20261003T044028Z
|
|
28
|
+
records: 50 author, 100 quote
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Highlights
|
|
32
|
+
|
|
33
|
+
| | |
|
|
34
|
+
|---|---|
|
|
35
|
+
| **Fetch once, parse forever** | Raw responses are stored gzip-compressed and keyed by SHA-256. `harvester reparse` replays the whole crawl graph from disk with **zero network requests**, so fixing a parser never costs another crawl. |
|
|
36
|
+
| **Resumable by design** | The frontier is a SQLite queue. Ctrl+C (or a crash, or a reboot) loses nothing; the next `run` continues where it stopped. |
|
|
37
|
+
| **Polite by default** | robots.txt per RFC 9309, including its 4xx/5xx rules and `Crawl-delay`. Per-host rate limits and concurrency caps. `Retry-After` honoured. The delay adapts, backing off on 429/503 and recovering slowly. |
|
|
38
|
+
| **Honest** | Requests carry a real User-Agent with a contact URL. Scrapling's browser impersonation and header forging are **turned off**. |
|
|
39
|
+
| **Smart caching** | Each request picks a cache policy: documents are fetched once, listings are revalidated with `If-None-Match` / `If-Modified-Since`, and a `304` reuses the stored body. |
|
|
40
|
+
| **Provenance and licensing** | Every record states its source URL, fetch time, content hash, parser version and the data's licence. Every run writes a manifest. |
|
|
41
|
+
| **Change tracking** | The store keeps a history of every fetch. `harvester changes` lists pages whose content changed between crawls. |
|
|
42
|
+
| **Bulk-data connectors** | Built-in **OAI-PMH** and **DSpace 7+** sources cover thousands of libraries, archives and government repositories. Downloads are verified against published MD5s. |
|
|
43
|
+
| **Small API** | A source is a class with a start URL and a parse method. Load one from a file, or publish it as a plugin through entry points. |
|
|
44
|
+
|
|
45
|
+
## Quick start
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install harvester-kit # installs the `harvester` command and package
|
|
49
|
+
|
|
50
|
+
harvester run examples/quotes.py # crawl the scraping sandbox
|
|
51
|
+
harvester reparse examples/quotes.py # rebuild every record offline
|
|
52
|
+
harvester status examples/quotes.py # queue, store and run history
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Output lands in `.harvester/<source>/runs/<run-id>/records.jsonl`, one record per line:
|
|
56
|
+
|
|
57
|
+
```json
|
|
58
|
+
{
|
|
59
|
+
"kind": "author",
|
|
60
|
+
"id": "Albert-Einstein",
|
|
61
|
+
"data": {"name": "Albert Einstein", "born": "March 14, 1879", "born_in": "in Ulm, Germany"},
|
|
62
|
+
"provenance": {
|
|
63
|
+
"source": "quotes",
|
|
64
|
+
"url": "https://quotes.toscrape.com/author/Albert-Einstein/",
|
|
65
|
+
"fetched_at": "2026-10-03T04:41:36Z",
|
|
66
|
+
"sha256": "9d1c…",
|
|
67
|
+
"status": 200,
|
|
68
|
+
"parser_version": "1",
|
|
69
|
+
"harvester_version": "0.1.0",
|
|
70
|
+
"license": {"name": "Scraping sandbox by Zyte; sample data"}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## Writing a source
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
from harvester import CachePolicy, Page, Request, Source
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class Quotes(Source):
|
|
82
|
+
name = "quotes"
|
|
83
|
+
start_urls = ("https://quotes.toscrape.com/",)
|
|
84
|
+
allowed_domains = ("quotes.toscrape.com",)
|
|
85
|
+
|
|
86
|
+
def start(self):
|
|
87
|
+
# Listings change, so revalidate them on each pass.
|
|
88
|
+
yield Request(url=self.start_urls[0], cache=CachePolicy.REVALIDATE)
|
|
89
|
+
|
|
90
|
+
def parse(self, page: Page):
|
|
91
|
+
for quote in page.css("div.quote"):
|
|
92
|
+
yield page.record(
|
|
93
|
+
"quote",
|
|
94
|
+
id=quote.css("span.text::text").get(),
|
|
95
|
+
author=quote.css("small.author::text").get(),
|
|
96
|
+
tags=quote.css("a.tag::text").getall(),
|
|
97
|
+
)
|
|
98
|
+
if next_href := page.css("li.next a::attr(href)").get():
|
|
99
|
+
yield page.follow(next_href, cache=CachePolicy.REVALIDATE)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
`harvester new my-site` scaffolds one. Callbacks receive a `Page`, which offers CSS/XPath via Scrapling, `.json()` and `.text`. They yield `Request`s to follow and `Record`s to keep. The one rule: **callbacks must not do their own I/O**. Keeping them pure is what makes offline re-parsing exact.
|
|
103
|
+
|
|
104
|
+
See [docs/writing-sources.md](docs/writing-sources.md) for options, multiple callbacks, file downloads and packaging a source as a plugin.
|
|
105
|
+
|
|
106
|
+
## How it works
|
|
107
|
+
|
|
108
|
+
```mermaid
|
|
109
|
+
flowchart LR
|
|
110
|
+
S[Source.start] --> F[(Frontier<br/>SQLite queue)]
|
|
111
|
+
F -->|claim| R[Scope + robots.txt check]
|
|
112
|
+
R --> C{In raw store?}
|
|
113
|
+
C -->|PREFER| PG[Page]
|
|
114
|
+
C -->|REVALIDATE / miss| T[Host throttle] --> H[Scrapling fetch]
|
|
115
|
+
H -->|200| ST[(Raw store<br/>SHA-256 blobs)] --> PG
|
|
116
|
+
H -->|304| PG
|
|
117
|
+
H -->|429 / 503 / 5xx| B[Back off + retry] --> F
|
|
118
|
+
PG --> CB[Parse callback]
|
|
119
|
+
CB -->|Request| F
|
|
120
|
+
CB -->|Record + provenance| O[(records.jsonl<br/>+ manifest)]
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
```text
|
|
124
|
+
.harvester/<source>/
|
|
125
|
+
├── frontier.sqlite # queue: pending / done / failed / skipped, retries, priorities
|
|
126
|
+
├── store/
|
|
127
|
+
│ ├── index.sqlite # latest response per URL + full fetch history
|
|
128
|
+
│ └── blobs/ab/cd/…gz # bodies, content-addressed and deduplicated
|
|
129
|
+
└── runs/<run-id>/
|
|
130
|
+
├── records.jsonl # output
|
|
131
|
+
└── manifest.json # settings, stats, licence, outcome
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
## Built-in sources
|
|
135
|
+
|
|
136
|
+
| Source | What it harvests |
|
|
137
|
+
|---|---|
|
|
138
|
+
| `oai-pmh` | Any [OAI-PMH 2.0](https://www.openarchives.org/OAI/openarchivesprotocol.html) repository: `ListRecords` with resumption tokens, deleted records, any metadata format. `-o base_url=… -o set=… -o metadata_prefix=…` |
|
|
139
|
+
| `dspace` | Any DSpace 7+ repository via its REST API: items, then files from chosen bundles, MD5-verified, with text files inlined. `-o base_url=… -o scope=<uuid> -o bundles=ORIGINAL` |
|
|
140
|
+
| `india-code` | India's central Acts from [India Code](https://indiacode.gov.in): clean act records (number, year, ministry, enforcement date, repeal status) plus the official text extraction. `-o in_force_only=true -o pdf=true` |
|
|
141
|
+
|
|
142
|
+
## CLI
|
|
143
|
+
|
|
144
|
+
| Command | |
|
|
145
|
+
|---|---|
|
|
146
|
+
| `harvester run SOURCE [-o k=v] [--limit N] [--delay S] [--refresh] [--restart]` | Crawl. Resumes interrupted work; otherwise starts a new pass that reuses the cache. |
|
|
147
|
+
| `harvester reparse SOURCE` | Re-run parsers over the raw store. No network. |
|
|
148
|
+
| `harvester status SOURCE` | Queue counts, store size, failures with reasons, recent runs. |
|
|
149
|
+
| `harvester changes SOURCE` | URLs whose content changed between fetches. |
|
|
150
|
+
| `harvester show URL -s SOURCE [--save FILE]` | Inspect or extract a stored response. |
|
|
151
|
+
| `harvester robots URL` | Explain whether a URL may be fetched, and why. |
|
|
152
|
+
| `harvester list` / `harvester new NAME` | Installed sources / scaffold a new one. |
|
|
153
|
+
|
|
154
|
+
`SOURCE` is an installed source name, `path/to/file.py`, or `path/to/file.py:ClassName`.
|
|
155
|
+
|
|
156
|
+
## Politeness, precisely
|
|
157
|
+
|
|
158
|
+
harvester is meant for collecting data you are entitled to collect, without being a burden on the sites that host it.
|
|
159
|
+
|
|
160
|
+
- **robots.txt (RFC 9309).**
|
|
161
|
+
- `2xx`: rules and `Crawl-delay` are obeyed.
|
|
162
|
+
- `4xx`: no restrictions.
|
|
163
|
+
- `5xx` or network error: **complete disallow**.
|
|
164
|
+
- A source may opt out of that last rule only by declaring `robots_unavailable = "allow"` *with a written reason*. The choice is recorded in every run manifest.
|
|
165
|
+
- **Rate limits.** Per-host minimum delay (default 1 s) and concurrency (default 1). On `429`/`503` the delay doubles, honouring `Retry-After`. It recovers by 10% per healthy response.
|
|
166
|
+
- **Identity.** `harvester/<version> (+https://github.com/sarthak213/harvester)`. Change it with `--user-agent`, but keep a contact URL in it.
|
|
167
|
+
- **No evasion.** No browser fingerprint spoofing, no CAPTCHA solving, no proxy rotation to get around blocks. If a site says no, harvester listens.
|
|
168
|
+
|
|
169
|
+
## Development
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
git clone https://github.com/sarthak213/harvester && cd harvester
|
|
173
|
+
python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
174
|
+
pip install -e ".[dev]"
|
|
175
|
+
pytest # 41 tests, including a real HTTP server for end-to-end runs
|
|
176
|
+
ruff check . && ruff format --check . && mypy src
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
## Roadmap
|
|
180
|
+
|
|
181
|
+
- Browser rendering for JavaScript-only pages (Scrapling `DynamicFetcher`), opt-in per request
|
|
182
|
+
- Sitemap and RSS/Atom discovery sources
|
|
183
|
+
- Parquet / SQLite export with record de-duplication across runs
|
|
184
|
+
- Scheduled incremental runs and change notifications
|
|
185
|
+
- More open-data connectors: CKAN, Zenodo, S3 open-data buckets
|
|
186
|
+
|
|
187
|
+
## License
|
|
188
|
+
|
|
189
|
+
MIT. See [LICENSE](LICENSE). The licence covers harvester's code. **The data you harvest has its own terms**; sources declare them, and harvester records them in every output.
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
# Writing sources
|
|
2
|
+
|
|
3
|
+
A source tells harvester where to start and how to read what comes back. Everything else (queueing, caching, politeness, retries, output) is the engine's job.
|
|
4
|
+
|
|
5
|
+
## Anatomy
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
from harvester import CachePolicy, DataLicense, Page, Request, Source
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class Gazette(Source):
|
|
12
|
+
name = "gazette" # CLI name and data-directory name
|
|
13
|
+
description = "Notifications from the Example Gazette"
|
|
14
|
+
allowed_domains = ("gazette.example.org",) # anything else is skipped
|
|
15
|
+
license = DataLicense(
|
|
16
|
+
name="Government open data licence",
|
|
17
|
+
url="https://gazette.example.org/terms",
|
|
18
|
+
attribution="Source: Example Gazette",
|
|
19
|
+
commercial_use=True,
|
|
20
|
+
)
|
|
21
|
+
parser_version = "2" # bump when output changes
|
|
22
|
+
delay = 2.0 # seconds between requests to this host
|
|
23
|
+
concurrency = 1 # simultaneous requests to this host
|
|
24
|
+
|
|
25
|
+
@classmethod
|
|
26
|
+
def option_defaults(cls):
|
|
27
|
+
return {"year": "2026"} # harvester run gazette -o year=2025
|
|
28
|
+
|
|
29
|
+
def start(self):
|
|
30
|
+
url = f"https://gazette.example.org/{self.options['year']}/"
|
|
31
|
+
yield Request(url=url, callback="parse_index", cache=CachePolicy.REVALIDATE)
|
|
32
|
+
|
|
33
|
+
def parse_index(self, page: Page):
|
|
34
|
+
for href in page.css("a.notice::attr(href)").getall():
|
|
35
|
+
yield page.follow(href, callback="parse_notice", meta={"year": self.options["year"]})
|
|
36
|
+
if nxt := page.css("a[rel=next]::attr(href)").get():
|
|
37
|
+
yield page.follow(nxt, cache=CachePolicy.REVALIDATE)
|
|
38
|
+
|
|
39
|
+
def parse_notice(self, page: Page):
|
|
40
|
+
yield page.record(
|
|
41
|
+
"notice",
|
|
42
|
+
id=page.css("meta[name=notice-id]::attr(content)").get(),
|
|
43
|
+
title=page.css("h1::text").get(),
|
|
44
|
+
year=page.meta["year"],
|
|
45
|
+
)
|
|
46
|
+
if pdf := page.css("a.pdf::attr(href)").get():
|
|
47
|
+
yield page.follow(pdf, callback="parse_pdf")
|
|
48
|
+
|
|
49
|
+
def parse_pdf(self, page: Page):
|
|
50
|
+
# The PDF body is already in the raw store; record where to find it.
|
|
51
|
+
yield page.record("file", id=page.sha256, url=page.url, bytes=len(page.body))
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Callbacks
|
|
55
|
+
|
|
56
|
+
- A callback is any method named in `Request.callback` (default `parse`).
|
|
57
|
+
- It receives a `Page` and yields `Request` and `Record` objects. Use `page.follow()` and `page.record()` for convenience.
|
|
58
|
+
- **Callbacks must not perform I/O.** Everything they need is in the page: `body`, `headers`, `text`, `json()`, `css()`, `xpath()` and `meta`. Purity is what lets `harvester reparse` reproduce a crawl exactly from the raw store.
|
|
59
|
+
- An exception inside a callback fails only that request. The reason is visible in `harvester status`, and the crawl continues.
|
|
60
|
+
|
|
61
|
+
## Cache policies
|
|
62
|
+
|
|
63
|
+
| Policy | Use for | Behaviour on later passes |
|
|
64
|
+
|---|---|---|
|
|
65
|
+
| `PREFER` (default) | documents, files, anything immutable | served from the raw store; no request |
|
|
66
|
+
| `REVALIDATE` | listings, feeds, search results | conditional request; `304` reuses the stored body |
|
|
67
|
+
| `BYPASS` | volatile endpoints | always re-downloaded |
|
|
68
|
+
|
|
69
|
+
`harvester run --refresh` upgrades every `PREFER` request to `REVALIDATE` for one pass.
|
|
70
|
+
|
|
71
|
+
## Passing data between callbacks
|
|
72
|
+
|
|
73
|
+
Put it in `meta`. It is stored with the request in the frontier, so it survives restarts. `page.follow()` merges the current page's meta with any you add.
|
|
74
|
+
|
|
75
|
+
## robots.txt that cannot be fetched
|
|
76
|
+
|
|
77
|
+
If a site's robots.txt returns `5xx` or times out, RFC 9309 says crawlers must assume everything is disallowed, and harvester does. If you have good reason to proceed (for example, the operator has confirmed it, or the endpoint is a published bulk API), declare it:
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
robots_unavailable = "allow"
|
|
81
|
+
robots_unavailable_reason = "Operator confirmed by email on 2026-10-01; bulk API is public."
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
The reason is written into every run manifest.
|
|
85
|
+
|
|
86
|
+
## Packaging a source as a plugin
|
|
87
|
+
|
|
88
|
+
Any installed package can register sources:
|
|
89
|
+
|
|
90
|
+
```toml
|
|
91
|
+
# pyproject.toml of your package
|
|
92
|
+
[project.entry-points."harvester.sources"]
|
|
93
|
+
gazette = "my_package.sources:Gazette"
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
After `pip install`, `harvester list` shows it and `harvester run gazette` works.
|