harvester-kit 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. harvester_kit-0.1.0/.github/workflows/ci.yml +35 -0
  2. harvester_kit-0.1.0/.github/workflows/release.yml +39 -0
  3. harvester_kit-0.1.0/.gitignore +14 -0
  4. harvester_kit-0.1.0/CHANGELOG.md +14 -0
  5. harvester_kit-0.1.0/CONTRIBUTING.md +10 -0
  6. harvester_kit-0.1.0/LICENSE +21 -0
  7. harvester_kit-0.1.0/PKG-INFO +231 -0
  8. harvester_kit-0.1.0/README.md +189 -0
  9. harvester_kit-0.1.0/docs/writing-sources.md +96 -0
  10. harvester_kit-0.1.0/examples/quotes.py +45 -0
  11. harvester_kit-0.1.0/pyproject.toml +96 -0
  12. harvester_kit-0.1.0/src/harvester/__init__.py +22 -0
  13. harvester_kit-0.1.0/src/harvester/cli.py +437 -0
  14. harvester_kit-0.1.0/src/harvester/engine.py +421 -0
  15. harvester_kit-0.1.0/src/harvester/fetch.py +113 -0
  16. harvester_kit-0.1.0/src/harvester/frontier.py +142 -0
  17. harvester_kit-0.1.0/src/harvester/models.py +89 -0
  18. harvester_kit-0.1.0/src/harvester/page.py +97 -0
  19. harvester_kit-0.1.0/src/harvester/politeness.py +181 -0
  20. harvester_kit-0.1.0/src/harvester/py.typed +0 -0
  21. harvester_kit-0.1.0/src/harvester/registry.py +78 -0
  22. harvester_kit-0.1.0/src/harvester/sinks.py +40 -0
  23. harvester_kit-0.1.0/src/harvester/source.py +78 -0
  24. harvester_kit-0.1.0/src/harvester/sources/__init__.py +1 -0
  25. harvester_kit-0.1.0/src/harvester/sources/dspace.py +153 -0
  26. harvester_kit-0.1.0/src/harvester/sources/india_code.py +96 -0
  27. harvester_kit-0.1.0/src/harvester/sources/oai_pmh.py +99 -0
  28. harvester_kit-0.1.0/src/harvester/store.py +178 -0
  29. harvester_kit-0.1.0/tests/__init__.py +0 -0
  30. harvester_kit-0.1.0/tests/conftest.py +86 -0
  31. harvester_kit-0.1.0/tests/fixtures/india_code/bns_text_head.txt +348 -0
  32. harvester_kit-0.1.0/tests/fixtures/india_code/bundles_bns.json +468 -0
  33. harvester_kit-0.1.0/tests/fixtures/india_code/search_page.json +1221 -0
  34. harvester_kit-0.1.0/tests/test_cli.py +91 -0
  35. harvester_kit-0.1.0/tests/test_engine.py +246 -0
  36. harvester_kit-0.1.0/tests/test_sources.py +202 -0
  37. harvester_kit-0.1.0/tests/test_units.py +171 -0
@@ -0,0 +1,35 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ lint:
10
+ runs-on: ubuntu-latest
11
+ steps:
12
+ - uses: actions/checkout@v7
13
+ - uses: actions/setup-python@v7
14
+ with:
15
+ python-version: "3.13"
16
+ - run: pip install -e ".[dev]"
17
+ - run: ruff check .
18
+ - run: ruff format --check .
19
+ - run: mypy src
20
+
21
+ test:
22
+ strategy:
23
+ fail-fast: false
24
+ matrix:
25
+ os: [ubuntu-latest, windows-latest, macos-latest]
26
+ python: ["3.10", "3.11", "3.12", "3.13", "3.14"]
27
+ runs-on: ${{ matrix.os }}
28
+ steps:
29
+ - uses: actions/checkout@v7
30
+ - uses: actions/setup-python@v7
31
+ with:
32
+ python-version: ${{ matrix.python }}
33
+ allow-prereleases: true
34
+ - run: pip install -e ".[dev]"
35
+ - run: pytest -q
@@ -0,0 +1,39 @@
1
+ name: Release
2
+
3
+ # Publishing a GitHub release builds the package and uploads it to PyPI using
4
+ # Trusted Publishing (OIDC): no API token is stored anywhere.
5
+
6
+ on:
7
+ release:
8
+ types: [published]
9
+
10
+ jobs:
11
+ build:
12
+ runs-on: ubuntu-latest
13
+ steps:
14
+ - uses: actions/checkout@v7
15
+ - uses: actions/setup-python@v7
16
+ with:
17
+ python-version: "3.13"
18
+ - run: pip install build twine
19
+ - run: python -m build
20
+ - run: twine check --strict dist/*
21
+ - uses: actions/upload-artifact@v7
22
+ with:
23
+ name: dist
24
+ path: dist/
25
+
26
+ publish:
27
+ needs: build
28
+ runs-on: ubuntu-latest
29
+ environment:
30
+ name: pypi
31
+ url: https://pypi.org/project/harvester-kit/
32
+ permissions:
33
+ id-token: write
34
+ steps:
35
+ - uses: actions/download-artifact@v8
36
+ with:
37
+ name: dist
38
+ path: dist/
39
+ - uses: pypa/gh-action-pypi-publish@v1.14.2
@@ -0,0 +1,14 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ venv/
5
+ dist/
6
+ build/
7
+ *.egg-info/
8
+ .pytest_cache/
9
+ .mypy_cache/
10
+ .ruff_cache/
11
+ .coverage
12
+ htmlcov/
13
+ # harvester's own data directory (stores, queues, runs)
14
+ .harvester/
@@ -0,0 +1,14 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 — 2026-10-03
4
+
5
+ First release.
6
+
7
+ - Crawl engine with a persistent SQLite frontier: resumable, de-duplicated, prioritised.
8
+ - Content-addressed raw store with fetch history; offline `reparse`.
9
+ - Per-request cache policies (prefer / revalidate / bypass) with conditional requests.
10
+ - robots.txt per RFC 9309, Crawl-delay, per-host throttling with adaptive back-off and Retry-After.
11
+ - Honest Scrapling-based fetcher (no impersonation, real User-Agent).
12
+ - JSONL output with per-record provenance and data licence; run manifests.
13
+ - Built-in sources: OAI-PMH, DSpace 7+, India Code.
14
+ - CLI: run, reparse, status, changes, show, robots, list, new.
@@ -0,0 +1,10 @@
1
+ # Contributing
2
+
3
+ Thanks for helping. A few ground rules keep harvester trustworthy:
4
+
5
+ 1. **Politeness is not optional.** Changes must not add ways to evade blocks:
6
+ no fingerprint spoofing, CAPTCHA solving or block-dodging proxy rotation.
7
+ 2. **Callbacks stay pure.** Anything that would let a parse callback do I/O breaks offline re-parsing.
8
+ 3. **Tests come with changes.** `pytest`, `ruff check .`, `ruff format --check .` and `mypy src` must pass.
9
+
10
+ New sources for public bulk-data protocols (CKAN, Zenodo, sitemaps …) are especially welcome.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sarthak Pahwa
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,231 @@
1
+ Metadata-Version: 2.5
2
+ Name: harvester-kit
3
+ Version: 0.1.0
4
+ Summary: Polite, reproducible web harvesting: fetch once, parse forever.
5
+ Project-URL: Homepage, https://github.com/sarthak213/harvester
6
+ Project-URL: Issues, https://github.com/sarthak213/harvester/issues
7
+ Project-URL: Changelog, https://github.com/sarthak213/harvester/blob/main/CHANGELOG.md
8
+ Author: Sarthak Pahwa
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: crawling,data-pipeline,dspace,harvesting,oai-pmh,robots.txt,scraping
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Environment :: Console
14
+ Classifier: Framework :: AsyncIO
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Intended Audience :: Science/Research
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3 :: Only
20
+ Classifier: Programming Language :: Python :: 3.10
21
+ Classifier: Programming Language :: Python :: 3.11
22
+ Classifier: Programming Language :: Python :: 3.12
23
+ Classifier: Programming Language :: Python :: 3.13
24
+ Classifier: Programming Language :: Python :: 3.14
25
+ Classifier: Topic :: Internet :: WWW/HTTP :: Indexing/Search
26
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
27
+ Classifier: Typing :: Typed
28
+ Requires-Python: >=3.10
29
+ Requires-Dist: orjson>=3.10
30
+ Requires-Dist: protego>=0.4
31
+ Requires-Dist: pydantic>=2.7
32
+ Requires-Dist: rich>=13.7
33
+ Requires-Dist: scrapling[fetchers]>=0.4.15
34
+ Requires-Dist: typer>=0.12
35
+ Requires-Dist: w3lib>=2.1
36
+ Provides-Extra: dev
37
+ Requires-Dist: mypy>=1.11; extra == 'dev'
38
+ Requires-Dist: pytest-asyncio>=0.24; extra == 'dev'
39
+ Requires-Dist: pytest>=8; extra == 'dev'
40
+ Requires-Dist: ruff>=0.6; extra == 'dev'
41
+ Description-Content-Type: text/markdown
42
+
43
+ <div align="center">
44
+
45
+ # harvester
46
+
47
+ **Polite, reproducible web harvesting. Fetch once, parse forever.**
48
+
49
+ [![CI](https://github.com/sarthak213/harvester/actions/workflows/ci.yml/badge.svg)](https://github.com/sarthak213/harvester/actions/workflows/ci.yml)
50
+ [![PyPI](https://img.shields.io/pypi/v/harvester-kit)](https://pypi.org/project/harvester-kit/)
51
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10%E2%80%933.14-blue)](https://www.python.org)
52
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
53
+ [![Typed: mypy strict](https://img.shields.io/badge/typed-mypy%20strict-informational)](https://mypy-lang.org)
54
+
55
+ </div>
56
+
57
+ Most scrapers are throwaway scripts. They re-download everything whenever a parser changes, start over after a crash, hammer servers until they get blocked, and produce data nobody can trace back to its source.
58
+
59
+ **harvester** treats harvesting as a data pipeline. Every response goes into a content-addressed store, so parsers can be fixed and re-run **offline**. The crawl queue lives on disk, so interrupted runs **resume** where they stopped. Politeness is built into the engine, not left as an afterthought. Every record carries its **provenance and licence**.
60
+
61
+ It uses [Scrapling](https://github.com/D4Vinci/Scrapling) for fetching and parsing, and adds the engine around it.
62
+
63
+ ```text
64
+ ┌──────────────────────────── quotes | crawl ─────────────────────────────┐
65
+ │ 60 pages 150 records 60 fetched 0 from cache │
66
+ │ 0 queued 0 retries 0 failed 0 blocked │
67
+ │ 272.5 KB downloaded 93 pages/min 39s elapsed 0 skipped │
68
+ └─────────────────────────────────────────────────────────────────────────┘
69
+ complete run 20261003T044028Z
70
+ records: 50 author, 100 quote
71
+ ```
72
+
73
+ ## Highlights
74
+
75
+ | | |
76
+ |---|---|
77
+ | **Fetch once, parse forever** | Raw responses are stored gzip-compressed and keyed by SHA-256. `harvester reparse` replays the whole crawl graph from disk with **zero network requests**, so fixing a parser never costs another crawl. |
78
+ | **Resumable by design** | The frontier is a SQLite queue. Ctrl+C (or a crash, or a reboot) loses nothing; the next `run` continues where it stopped. |
79
+ | **Polite by default** | robots.txt per RFC 9309, including its 4xx/5xx rules and `Crawl-delay`. Per-host rate limits and concurrency caps. `Retry-After` honoured. The delay adapts, backing off on 429/503 and recovering slowly. |
80
+ | **Honest** | Requests carry a real User-Agent with a contact URL. Scrapling's browser impersonation and header forging are **turned off**. |
81
+ | **Smart caching** | Each request picks a cache policy: documents are fetched once, listings are revalidated with `If-None-Match` / `If-Modified-Since`, and a `304` reuses the stored body. |
82
+ | **Provenance and licensing** | Every record states its source URL, fetch time, content hash, parser version and the data's licence. Every run writes a manifest. |
83
+ | **Change tracking** | The store keeps a history of every fetch. `harvester changes` lists pages whose content changed between crawls. |
84
+ | **Bulk-data connectors** | Built-in **OAI-PMH** and **DSpace 7+** sources cover thousands of libraries, archives and government repositories. Downloads are verified against published MD5s. |
85
+ | **Small API** | A source is a class with a start URL and a parse method. Load one from a file, or publish it as a plugin through entry points. |
86
+
87
+ ## Quick start
88
+
89
+ ```bash
90
+ pip install harvester-kit # installs the `harvester` command and package
91
+
92
+ harvester run examples/quotes.py # crawl the scraping sandbox
93
+ harvester reparse examples/quotes.py # rebuild every record offline
94
+ harvester status examples/quotes.py # queue, store and run history
95
+ ```
96
+
97
+ Output lands in `.harvester/<source>/runs/<run-id>/records.jsonl`, one record per line:
98
+
99
+ ```json
100
+ {
101
+ "kind": "author",
102
+ "id": "Albert-Einstein",
103
+ "data": {"name": "Albert Einstein", "born": "March 14, 1879", "born_in": "in Ulm, Germany"},
104
+ "provenance": {
105
+ "source": "quotes",
106
+ "url": "https://quotes.toscrape.com/author/Albert-Einstein/",
107
+ "fetched_at": "2026-10-03T04:41:36Z",
108
+ "sha256": "9d1c…",
109
+ "status": 200,
110
+ "parser_version": "1",
111
+ "harvester_version": "0.1.0",
112
+ "license": {"name": "Scraping sandbox by Zyte; sample data"}
113
+ }
114
+ }
115
+ ```
116
+
117
+ ## Writing a source
118
+
119
+ ```python
120
+ from harvester import CachePolicy, Page, Request, Source
121
+
122
+
123
+ class Quotes(Source):
124
+ name = "quotes"
125
+ start_urls = ("https://quotes.toscrape.com/",)
126
+ allowed_domains = ("quotes.toscrape.com",)
127
+
128
+ def start(self):
129
+ # Listings change, so revalidate them on each pass.
130
+ yield Request(url=self.start_urls[0], cache=CachePolicy.REVALIDATE)
131
+
132
+ def parse(self, page: Page):
133
+ for quote in page.css("div.quote"):
134
+ yield page.record(
135
+ "quote",
136
+ id=quote.css("span.text::text").get(),
137
+ author=quote.css("small.author::text").get(),
138
+ tags=quote.css("a.tag::text").getall(),
139
+ )
140
+ if next_href := page.css("li.next a::attr(href)").get():
141
+ yield page.follow(next_href, cache=CachePolicy.REVALIDATE)
142
+ ```
143
+
144
+ `harvester new my-site` scaffolds one. Callbacks receive a `Page`, which offers CSS/XPath via Scrapling, `.json()` and `.text`. They yield `Request`s to follow and `Record`s to keep. The one rule: **callbacks must not do their own I/O**. Keeping them pure is what makes offline re-parsing exact.
145
+
146
+ See [docs/writing-sources.md](docs/writing-sources.md) for options, multiple callbacks, file downloads and packaging a source as a plugin.
147
+
148
+ ## How it works
149
+
150
+ ```mermaid
151
+ flowchart LR
152
+ S[Source.start] --> F[(Frontier<br/>SQLite queue)]
153
+ F -->|claim| R[Scope + robots.txt check]
154
+ R --> C{In raw store?}
155
+ C -->|PREFER| PG[Page]
156
+ C -->|REVALIDATE / miss| T[Host throttle] --> H[Scrapling fetch]
157
+ H -->|200| ST[(Raw store<br/>SHA-256 blobs)] --> PG
158
+ H -->|304| PG
159
+ H -->|429 / 503 / 5xx| B[Back off + retry] --> F
160
+ PG --> CB[Parse callback]
161
+ CB -->|Request| F
162
+ CB -->|Record + provenance| O[(records.jsonl<br/>+ manifest)]
163
+ ```
164
+
165
+ ```text
166
+ .harvester/<source>/
167
+ ├── frontier.sqlite # queue: pending / done / failed / skipped, retries, priorities
168
+ ├── store/
169
+ │ ├── index.sqlite # latest response per URL + full fetch history
170
+ │ └── blobs/ab/cd/…gz # bodies, content-addressed and deduplicated
171
+ └── runs/<run-id>/
172
+ ├── records.jsonl # output
173
+ └── manifest.json # settings, stats, licence, outcome
174
+ ```
175
+
176
+ ## Built-in sources
177
+
178
+ | Source | What it harvests |
179
+ |---|---|
180
+ | `oai-pmh` | Any [OAI-PMH 2.0](https://www.openarchives.org/OAI/openarchivesprotocol.html) repository: `ListRecords` with resumption tokens, deleted records, any metadata format. `-o base_url=… -o set=… -o metadata_prefix=…` |
181
+ | `dspace` | Any DSpace 7+ repository via its REST API: items, then files from chosen bundles, MD5-verified, with text files inlined. `-o base_url=… -o scope=<uuid> -o bundles=ORIGINAL` |
182
+ | `india-code` | India's central Acts from [India Code](https://indiacode.gov.in): clean act records (number, year, ministry, enforcement date, repeal status) plus the official text extraction. `-o in_force_only=true -o pdf=true` |
183
+
184
+ ## CLI
185
+
186
+ | Command | |
187
+ |---|---|
188
+ | `harvester run SOURCE [-o k=v] [--limit N] [--delay S] [--refresh] [--restart]` | Crawl. Resumes interrupted work; otherwise starts a new pass that reuses the cache. |
189
+ | `harvester reparse SOURCE` | Re-run parsers over the raw store. No network. |
190
+ | `harvester status SOURCE` | Queue counts, store size, failures with reasons, recent runs. |
191
+ | `harvester changes SOURCE` | URLs whose content changed between fetches. |
192
+ | `harvester show URL -s SOURCE [--save FILE]` | Inspect or extract a stored response. |
193
+ | `harvester robots URL` | Explain whether a URL may be fetched, and why. |
194
+ | `harvester list` / `harvester new NAME` | Installed sources / scaffold a new one. |
195
+
196
+ `SOURCE` is an installed source name, `path/to/file.py`, or `path/to/file.py:ClassName`.
197
+
198
+ ## Politeness, precisely
199
+
200
+ harvester is meant for collecting data you are entitled to collect, without being a burden on the sites that host it.
201
+
202
+ - **robots.txt (RFC 9309).**
203
+ - `2xx`: rules and `Crawl-delay` are obeyed.
204
+ - `4xx`: no restrictions.
205
+ - `5xx` or network error: **complete disallow**.
206
+ - A source may opt out of that last rule only by declaring `robots_unavailable = "allow"` *with a written reason*. The choice is recorded in every run manifest.
207
+ - **Rate limits.** Per-host minimum delay (default 1 s) and concurrency (default 1). On `429`/`503` the delay doubles, honouring `Retry-After`. It recovers by 10% per healthy response.
208
+ - **Identity.** `harvester/<version> (+https://github.com/sarthak213/harvester)`. Change it with `--user-agent`, but keep a contact URL in it.
209
+ - **No evasion.** No browser fingerprint spoofing, no CAPTCHA solving, no proxy rotation to get around blocks. If a site says no, harvester listens.
210
+
211
+ ## Development
212
+
213
+ ```bash
214
+ git clone https://github.com/sarthak213/harvester && cd harvester
215
+ python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
216
+ pip install -e ".[dev]"
217
+ pytest # 41 tests, including a real HTTP server for end-to-end runs
218
+ ruff check . && ruff format --check . && mypy src
219
+ ```
220
+
221
+ ## Roadmap
222
+
223
+ - Browser rendering for JavaScript-only pages (Scrapling `DynamicFetcher`), opt-in per request
224
+ - Sitemap and RSS/Atom discovery sources
225
+ - Parquet / SQLite export with record de-duplication across runs
226
+ - Scheduled incremental runs and change notifications
227
+ - More open-data connectors: CKAN, Zenodo, S3 open-data buckets
228
+
229
+ ## License
230
+
231
+ MIT. See [LICENSE](LICENSE). The licence covers harvester's code. **The data you harvest has its own terms**; sources declare them, and harvester records them in every output.
@@ -0,0 +1,189 @@
1
+ <div align="center">
2
+
3
+ # harvester
4
+
5
+ **Polite, reproducible web harvesting. Fetch once, parse forever.**
6
+
7
+ [![CI](https://github.com/sarthak213/harvester/actions/workflows/ci.yml/badge.svg)](https://github.com/sarthak213/harvester/actions/workflows/ci.yml)
8
+ [![PyPI](https://img.shields.io/pypi/v/harvester-kit)](https://pypi.org/project/harvester-kit/)
9
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10%E2%80%933.14-blue)](https://www.python.org)
10
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
11
+ [![Typed: mypy strict](https://img.shields.io/badge/typed-mypy%20strict-informational)](https://mypy-lang.org)
12
+
13
+ </div>
14
+
15
+ Most scrapers are throwaway scripts. They re-download everything whenever a parser changes, start over after a crash, hammer servers until they get blocked, and produce data nobody can trace back to its source.
16
+
17
+ **harvester** treats harvesting as a data pipeline. Every response goes into a content-addressed store, so parsers can be fixed and re-run **offline**. The crawl queue lives on disk, so interrupted runs **resume** where they stopped. Politeness is built into the engine, not left as an afterthought. Every record carries its **provenance and licence**.
18
+
19
+ It uses [Scrapling](https://github.com/D4Vinci/Scrapling) for fetching and parsing, and adds the engine around it.
20
+
21
+ ```text
22
+ ┌──────────────────────────── quotes | crawl ─────────────────────────────┐
23
+ │ 60 pages 150 records 60 fetched 0 from cache │
24
+ │ 0 queued 0 retries 0 failed 0 blocked │
25
+ │ 272.5 KB downloaded 93 pages/min 39s elapsed 0 skipped │
26
+ └─────────────────────────────────────────────────────────────────────────┘
27
+ complete run 20261003T044028Z
28
+ records: 50 author, 100 quote
29
+ ```
30
+
31
+ ## Highlights
32
+
33
+ | | |
34
+ |---|---|
35
+ | **Fetch once, parse forever** | Raw responses are stored gzip-compressed and keyed by SHA-256. `harvester reparse` replays the whole crawl graph from disk with **zero network requests**, so fixing a parser never costs another crawl. |
36
+ | **Resumable by design** | The frontier is a SQLite queue. Ctrl+C (or a crash, or a reboot) loses nothing; the next `run` continues where it stopped. |
37
+ | **Polite by default** | robots.txt per RFC 9309, including its 4xx/5xx rules and `Crawl-delay`. Per-host rate limits and concurrency caps. `Retry-After` honoured. The delay adapts, backing off on 429/503 and recovering slowly. |
38
+ | **Honest** | Requests carry a real User-Agent with a contact URL. Scrapling's browser impersonation and header forging are **turned off**. |
39
+ | **Smart caching** | Each request picks a cache policy: documents are fetched once, listings are revalidated with `If-None-Match` / `If-Modified-Since`, and a `304` reuses the stored body. |
40
+ | **Provenance and licensing** | Every record states its source URL, fetch time, content hash, parser version and the data's licence. Every run writes a manifest. |
41
+ | **Change tracking** | The store keeps a history of every fetch. `harvester changes` lists pages whose content changed between crawls. |
42
+ | **Bulk-data connectors** | Built-in **OAI-PMH** and **DSpace 7+** sources cover thousands of libraries, archives and government repositories. Downloads are verified against published MD5s. |
43
+ | **Small API** | A source is a class with a start URL and a parse method. Load one from a file, or publish it as a plugin through entry points. |
44
+
45
+ ## Quick start
46
+
47
+ ```bash
48
+ pip install harvester-kit # installs the `harvester` command and package
49
+
50
+ harvester run examples/quotes.py # crawl the scraping sandbox
51
+ harvester reparse examples/quotes.py # rebuild every record offline
52
+ harvester status examples/quotes.py # queue, store and run history
53
+ ```
54
+
55
+ Output lands in `.harvester/<source>/runs/<run-id>/records.jsonl`, one record per line:
56
+
57
+ ```json
58
+ {
59
+ "kind": "author",
60
+ "id": "Albert-Einstein",
61
+ "data": {"name": "Albert Einstein", "born": "March 14, 1879", "born_in": "in Ulm, Germany"},
62
+ "provenance": {
63
+ "source": "quotes",
64
+ "url": "https://quotes.toscrape.com/author/Albert-Einstein/",
65
+ "fetched_at": "2026-10-03T04:41:36Z",
66
+ "sha256": "9d1c…",
67
+ "status": 200,
68
+ "parser_version": "1",
69
+ "harvester_version": "0.1.0",
70
+ "license": {"name": "Scraping sandbox by Zyte; sample data"}
71
+ }
72
+ }
73
+ ```
74
+
75
+ ## Writing a source
76
+
77
+ ```python
78
+ from harvester import CachePolicy, Page, Request, Source
79
+
80
+
81
+ class Quotes(Source):
82
+ name = "quotes"
83
+ start_urls = ("https://quotes.toscrape.com/",)
84
+ allowed_domains = ("quotes.toscrape.com",)
85
+
86
+ def start(self):
87
+ # Listings change, so revalidate them on each pass.
88
+ yield Request(url=self.start_urls[0], cache=CachePolicy.REVALIDATE)
89
+
90
+ def parse(self, page: Page):
91
+ for quote in page.css("div.quote"):
92
+ yield page.record(
93
+ "quote",
94
+ id=quote.css("span.text::text").get(),
95
+ author=quote.css("small.author::text").get(),
96
+ tags=quote.css("a.tag::text").getall(),
97
+ )
98
+ if next_href := page.css("li.next a::attr(href)").get():
99
+ yield page.follow(next_href, cache=CachePolicy.REVALIDATE)
100
+ ```
101
+
102
+ `harvester new my-site` scaffolds one. Callbacks receive a `Page`, which offers CSS/XPath via Scrapling, `.json()` and `.text`. They yield `Request`s to follow and `Record`s to keep. The one rule: **callbacks must not do their own I/O**. Keeping them pure is what makes offline re-parsing exact.
103
+
104
+ See [docs/writing-sources.md](docs/writing-sources.md) for options, multiple callbacks, file downloads and packaging a source as a plugin.
105
+
106
+ ## How it works
107
+
108
+ ```mermaid
109
+ flowchart LR
110
+ S[Source.start] --> F[(Frontier<br/>SQLite queue)]
111
+ F -->|claim| R[Scope + robots.txt check]
112
+ R --> C{In raw store?}
113
+ C -->|PREFER| PG[Page]
114
+ C -->|REVALIDATE / miss| T[Host throttle] --> H[Scrapling fetch]
115
+ H -->|200| ST[(Raw store<br/>SHA-256 blobs)] --> PG
116
+ H -->|304| PG
117
+ H -->|429 / 503 / 5xx| B[Back off + retry] --> F
118
+ PG --> CB[Parse callback]
119
+ CB -->|Request| F
120
+ CB -->|Record + provenance| O[(records.jsonl<br/>+ manifest)]
121
+ ```
122
+
123
+ ```text
124
+ .harvester/<source>/
125
+ ├── frontier.sqlite # queue: pending / done / failed / skipped, retries, priorities
126
+ ├── store/
127
+ │ ├── index.sqlite # latest response per URL + full fetch history
128
+ │ └── blobs/ab/cd/…gz # bodies, content-addressed and deduplicated
129
+ └── runs/<run-id>/
130
+ ├── records.jsonl # output
131
+ └── manifest.json # settings, stats, licence, outcome
132
+ ```
133
+
134
+ ## Built-in sources
135
+
136
+ | Source | What it harvests |
137
+ |---|---|
138
+ | `oai-pmh` | Any [OAI-PMH 2.0](https://www.openarchives.org/OAI/openarchivesprotocol.html) repository: `ListRecords` with resumption tokens, deleted records, any metadata format. `-o base_url=… -o set=… -o metadata_prefix=…` |
139
+ | `dspace` | Any DSpace 7+ repository via its REST API: items, then files from chosen bundles, MD5-verified, with text files inlined. `-o base_url=… -o scope=<uuid> -o bundles=ORIGINAL` |
140
+ | `india-code` | India's central Acts from [India Code](https://indiacode.gov.in): clean act records (number, year, ministry, enforcement date, repeal status) plus the official text extraction. `-o in_force_only=true -o pdf=true` |
141
+
142
+ ## CLI
143
+
144
+ | Command | |
145
+ |---|---|
146
+ | `harvester run SOURCE [-o k=v] [--limit N] [--delay S] [--refresh] [--restart]` | Crawl. Resumes interrupted work; otherwise starts a new pass that reuses the cache. |
147
+ | `harvester reparse SOURCE` | Re-run parsers over the raw store. No network. |
148
+ | `harvester status SOURCE` | Queue counts, store size, failures with reasons, recent runs. |
149
+ | `harvester changes SOURCE` | URLs whose content changed between fetches. |
150
+ | `harvester show URL -s SOURCE [--save FILE]` | Inspect or extract a stored response. |
151
+ | `harvester robots URL` | Explain whether a URL may be fetched, and why. |
152
+ | `harvester list` / `harvester new NAME` | Installed sources / scaffold a new one. |
153
+
154
+ `SOURCE` is an installed source name, `path/to/file.py`, or `path/to/file.py:ClassName`.
155
+
156
+ ## Politeness, precisely
157
+
158
+ harvester is meant for collecting data you are entitled to collect, without being a burden on the sites that host it.
159
+
160
+ - **robots.txt (RFC 9309).**
161
+ - `2xx`: rules and `Crawl-delay` are obeyed.
162
+ - `4xx`: no restrictions.
163
+ - `5xx` or network error: **complete disallow**.
164
+ - A source may opt out of that last rule only by declaring `robots_unavailable = "allow"` *with a written reason*. The choice is recorded in every run manifest.
165
+ - **Rate limits.** Per-host minimum delay (default 1 s) and concurrency (default 1). On `429`/`503` the delay doubles, honouring `Retry-After`. It recovers by 10% per healthy response.
166
+ - **Identity.** `harvester/<version> (+https://github.com/sarthak213/harvester)`. Change it with `--user-agent`, but keep a contact URL in it.
167
+ - **No evasion.** No browser fingerprint spoofing, no CAPTCHA solving, no proxy rotation to get around blocks. If a site says no, harvester listens.
168
+
169
+ ## Development
170
+
171
+ ```bash
172
+ git clone https://github.com/sarthak213/harvester && cd harvester
173
+ python -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
174
+ pip install -e ".[dev]"
175
+ pytest # 41 tests, including a real HTTP server for end-to-end runs
176
+ ruff check . && ruff format --check . && mypy src
177
+ ```
178
+
179
+ ## Roadmap
180
+
181
+ - Browser rendering for JavaScript-only pages (Scrapling `DynamicFetcher`), opt-in per request
182
+ - Sitemap and RSS/Atom discovery sources
183
+ - Parquet / SQLite export with record de-duplication across runs
184
+ - Scheduled incremental runs and change notifications
185
+ - More open-data connectors: CKAN, Zenodo, S3 open-data buckets
186
+
187
+ ## License
188
+
189
+ MIT. See [LICENSE](LICENSE). The licence covers harvester's code. **The data you harvest has its own terms**; sources declare them, and harvester records them in every output.
@@ -0,0 +1,96 @@
1
+ # Writing sources
2
+
3
+ A source tells harvester where to start and how to read what comes back. Everything else (queueing, caching, politeness, retries, output) is the engine's job.
4
+
5
+ ## Anatomy
6
+
7
+ ```python
8
+ from harvester import CachePolicy, DataLicense, Page, Request, Source
9
+
10
+
11
+ class Gazette(Source):
12
+ name = "gazette" # CLI name and data-directory name
13
+ description = "Notifications from the Example Gazette"
14
+ allowed_domains = ("gazette.example.org",) # anything else is skipped
15
+ license = DataLicense(
16
+ name="Government open data licence",
17
+ url="https://gazette.example.org/terms",
18
+ attribution="Source: Example Gazette",
19
+ commercial_use=True,
20
+ )
21
+ parser_version = "2" # bump when output changes
22
+ delay = 2.0 # seconds between requests to this host
23
+ concurrency = 1 # simultaneous requests to this host
24
+
25
+ @classmethod
26
+ def option_defaults(cls):
27
+ return {"year": "2026"} # harvester run gazette -o year=2025
28
+
29
+ def start(self):
30
+ url = f"https://gazette.example.org/{self.options['year']}/"
31
+ yield Request(url=url, callback="parse_index", cache=CachePolicy.REVALIDATE)
32
+
33
+ def parse_index(self, page: Page):
34
+ for href in page.css("a.notice::attr(href)").getall():
35
+ yield page.follow(href, callback="parse_notice", meta={"year": self.options["year"]})
36
+ if nxt := page.css("a[rel=next]::attr(href)").get():
37
+ yield page.follow(nxt, cache=CachePolicy.REVALIDATE)
38
+
39
+ def parse_notice(self, page: Page):
40
+ yield page.record(
41
+ "notice",
42
+ id=page.css("meta[name=notice-id]::attr(content)").get(),
43
+ title=page.css("h1::text").get(),
44
+ year=page.meta["year"],
45
+ )
46
+ if pdf := page.css("a.pdf::attr(href)").get():
47
+ yield page.follow(pdf, callback="parse_pdf")
48
+
49
+ def parse_pdf(self, page: Page):
50
+ # The PDF body is already in the raw store; record where to find it.
51
+ yield page.record("file", id=page.sha256, url=page.url, bytes=len(page.body))
52
+ ```
53
+
54
+ ## Callbacks
55
+
56
+ - A callback is any method named in `Request.callback` (default `parse`).
57
+ - It receives a `Page` and yields `Request` and `Record` objects. Use `page.follow()` and `page.record()` for convenience.
58
+ - **Callbacks must not perform I/O.** Everything they need is in the page: `body`, `headers`, `text`, `json()`, `css()`, `xpath()` and `meta`. Purity is what lets `harvester reparse` reproduce a crawl exactly from the raw store.
59
+ - An exception inside a callback fails only that request. The reason is visible in `harvester status`, and the crawl continues.
60
+
61
+ ## Cache policies
62
+
63
+ | Policy | Use for | Behaviour on later passes |
64
+ |---|---|---|
65
+ | `PREFER` (default) | documents, files, anything immutable | served from the raw store; no request |
66
+ | `REVALIDATE` | listings, feeds, search results | conditional request; `304` reuses the stored body |
67
+ | `BYPASS` | volatile endpoints | always re-downloaded |
68
+
69
+ `harvester run --refresh` upgrades every `PREFER` request to `REVALIDATE` for one pass.
70
+
71
+ ## Passing data between callbacks
72
+
73
+ Put it in `meta`. It is stored with the request in the frontier, so it survives restarts. `page.follow()` merges the current page's meta with any you add.
74
+
75
+ ## robots.txt that cannot be fetched
76
+
77
+ If a site's robots.txt returns `5xx` or times out, RFC 9309 says crawlers must assume everything is disallowed, and harvester does. If you have good reason to proceed (for example, the operator has confirmed it, or the endpoint is a published bulk API), declare it:
78
+
79
+ ```python
80
+ robots_unavailable = "allow"
81
+ robots_unavailable_reason = "Operator confirmed by email on 2026-10-01; bulk API is public."
82
+ ```
83
+
84
+ The reason is written into every run manifest.
85
+
86
+ ## Packaging a source as a plugin
87
+
88
+ Any installed package can register sources:
89
+
90
+ ```toml
91
+ # pyproject.toml of your package
92
+ [project.entry-points."harvester.sources"]
93
+ gazette = "my_package.sources:Gazette"
94
+ ```
95
+
96
+ After `pip install`, `harvester list` shows it and `harvester run gazette` works.