puffinparse 0.1.2__tar.gz → 0.1.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {puffinparse-0.1.2 → puffinparse-0.1.4}/Cargo.lock +5 -5
- {puffinparse-0.1.2 → puffinparse-0.1.4}/Cargo.toml +2 -2
- {puffinparse-0.1.2 → puffinparse-0.1.4}/PKG-INFO +77 -33
- {puffinparse-0.1.2 → puffinparse-0.1.4}/README.md +76 -32
- puffinparse-0.1.4/crates/puffinparse-core/README.md +18 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/bench/tables.rs +5 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/bench.rs +7 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/provider.rs +36 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/google_documentai.rs +15 -5
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/tesseract.rs +74 -6
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/textract.rs +3 -0
- puffinparse-0.1.4/crates/puffinparse-core/tests/fixtures/README.md +14 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/pyproject.toml +1 -1
- puffinparse-0.1.2/crates/puffinparse-core/README.md +0 -17
- {puffinparse-0.1.2 → puffinparse-0.1.4}/LICENSE +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/Cargo.toml +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/compat/extend.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/compat/llamaparse.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/compat/mod.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/compat/reducto.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/compat/roundtrip.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/error.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/http.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/jobs.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/lib.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/model.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/pricing.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/pricing.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/anthropic.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/azure.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/datalab.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/docling.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/extend.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/gemini.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/landingai.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/llamaparse.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/local.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/mathpix.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/mistral.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/mod.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/openai.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/paddleocr.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/reducto.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/unstructured.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/upstage.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/providers/vlm.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/router.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/testutil.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/types.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/src/util.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/benchmark_datasets.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/anthropic_messages_extract.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/anthropic_messages_parse.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/azure_invoice.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/azure_layout.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/azure_read.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/datalab_convert.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/docling_headings.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/docling_multipage.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/extend_extract_run.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/extend_parse_run.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/extend_parse_run_url.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/extend_parse_run_url_output.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/gemini_extract_invoice.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/gemini_parse_multipage.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/google_documentai_form.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/google_documentai_layout.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/google_documentai_ocr.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/landingai_extract.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/landingai_parse.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/llamaparse_extract_job.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/llamaparse_result_json.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/mathpix_pdf_lines.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/mathpix_text.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/mistral_annotation.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/mistral_ocr.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/mistral_ocr_blocks.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/openai_responses_extract.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/openai_responses_parse.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/paddleocr_layout_parsing.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/paddleocr_ocr.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/reducto_extract.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/reducto_extract_plain.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/reducto_parse.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/reducto_parse_url.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/reducto_parse_url_result.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/rules_text_sparse_note.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/tesseract_headings.tsv +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/textract_detect_text.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/textract_forms.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/textract_layout.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/textract_queries.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/unstructured_elements.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/fixtures/upstage_document_parse.json +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/live.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-core/tests/live_jobs.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-python/Cargo.toml +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/crates/puffinparse-python/src/lib.rs +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/python/puffinparse/__init__.py +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/python/puffinparse/_core.pyi +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/python/puffinparse/exceptions.py +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/python/puffinparse/jobs.py +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/python/puffinparse/main.py +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/python/puffinparse/py.typed +0 -0
- {puffinparse-0.1.2 → puffinparse-0.1.4}/python/puffinparse/types.py +0 -0
|
@@ -1395,7 +1395,7 @@ dependencies = [
|
|
|
1395
1395
|
|
|
1396
1396
|
[[package]]
|
|
1397
1397
|
name = "puffinparse-cli"
|
|
1398
|
-
version = "0.1.
|
|
1398
|
+
version = "0.1.4"
|
|
1399
1399
|
dependencies = [
|
|
1400
1400
|
"anyhow",
|
|
1401
1401
|
"chrono",
|
|
@@ -1415,7 +1415,7 @@ dependencies = [
|
|
|
1415
1415
|
|
|
1416
1416
|
[[package]]
|
|
1417
1417
|
name = "puffinparse-core"
|
|
1418
|
-
version = "0.1.
|
|
1418
|
+
version = "0.1.4"
|
|
1419
1419
|
dependencies = [
|
|
1420
1420
|
"async-trait",
|
|
1421
1421
|
"bytes",
|
|
@@ -1439,7 +1439,7 @@ dependencies = [
|
|
|
1439
1439
|
|
|
1440
1440
|
[[package]]
|
|
1441
1441
|
name = "puffinparse-node"
|
|
1442
|
-
version = "0.1.
|
|
1442
|
+
version = "0.1.4"
|
|
1443
1443
|
dependencies = [
|
|
1444
1444
|
"bytes",
|
|
1445
1445
|
"napi",
|
|
@@ -1453,7 +1453,7 @@ dependencies = [
|
|
|
1453
1453
|
|
|
1454
1454
|
[[package]]
|
|
1455
1455
|
name = "puffinparse-python"
|
|
1456
|
-
version = "0.1.
|
|
1456
|
+
version = "0.1.4"
|
|
1457
1457
|
dependencies = [
|
|
1458
1458
|
"bytes",
|
|
1459
1459
|
"puffinparse-core",
|
|
@@ -1468,7 +1468,7 @@ dependencies = [
|
|
|
1468
1468
|
|
|
1469
1469
|
[[package]]
|
|
1470
1470
|
name = "puffinparse-server"
|
|
1471
|
-
version = "0.1.
|
|
1471
|
+
version = "0.1.4"
|
|
1472
1472
|
dependencies = [
|
|
1473
1473
|
"axum",
|
|
1474
1474
|
"bytes",
|
|
@@ -3,7 +3,7 @@ resolver = "2"
|
|
|
3
3
|
members = ["crates/puffinparse-core", "crates/puffinparse-python"]
|
|
4
4
|
|
|
5
5
|
[workspace.package]
|
|
6
|
-
version = "0.1.
|
|
6
|
+
version = "0.1.4"
|
|
7
7
|
edition = "2021"
|
|
8
8
|
rust-version = "1.80"
|
|
9
9
|
license = "MIT"
|
|
@@ -15,7 +15,7 @@ keywords = ["ocr", "document-parsing", "pdf", "api", "benchmark"]
|
|
|
15
15
|
categories = ["api-bindings", "text-processing"]
|
|
16
16
|
|
|
17
17
|
[workspace.dependencies]
|
|
18
|
-
puffinparse-core = { path = "crates/puffinparse-core", version = "0.1.
|
|
18
|
+
puffinparse-core = { path = "crates/puffinparse-core", version = "0.1.4" }
|
|
19
19
|
reqwest = { version = "0.13", default-features = false, features = ["json", "multipart", "rustls", "stream", "gzip"] }
|
|
20
20
|
tokio = { version = "1", features = ["rt-multi-thread", "macros", "fs", "time", "sync"] }
|
|
21
21
|
serde = { version = "1", features = ["derive"] }
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: puffinparse
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.4
|
|
4
4
|
Classifier: Development Status :: 3 - Alpha
|
|
5
5
|
Classifier: Intended Audience :: Developers
|
|
6
6
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -36,7 +36,7 @@ Project-URL: Repository, https://github.com/ajinkyashejul/puffinparse
|
|
|
36
36
|
|
|
37
37
|
# PuffinParse
|
|
38
38
|
|
|
39
|
-
[](https://pypi.org/project/puffinparse/) [](https://github.com/ajinkyashejul/puffinparse/stargazers) [](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [](https://puffinparse.com/docs/) [](https://puffinparse.com/benchmark-results/) [](pyproject.toml)
|
|
39
|
+
[](https://pypi.org/project/puffinparse/) [](https://github.com/ajinkyashejul/puffinparse/stargazers) [](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [](https://puffinparse.com/docs/) [](https://puffinparse.com/benchmark-results/) [](https://github.com/ajinkyashejul/puffinparse/blob/main/pyproject.toml)
|
|
40
40
|
|
|
41
41
|
**One API for every document parser: parse, OCR and extract.** Rust core, Python and TypeScript SDKs, a CLI, a self-hosted gateway, and an open benchmark that ranks providers on accuracy, latency and cost.
|
|
42
42
|
|
|
@@ -64,24 +64,28 @@ Two things make switching real rather than aspirational. **Modes**: every call n
|
|
|
64
64
|
| **Providers (v0.1)** | 18 providers · 60 models · 3 modes. **Live-verified** against the real APIs: [Reducto](https://reducto.ai), [Extend](https://extend.ai), [LlamaParse](https://cloud.llamaindex.ai). **Verified locally**: the self-hosted Tesseract and Docling. **Docs-only** (implemented from the provider's API documentation and tested against fixture payloads, not yet run live): Mistral, Azure, Textract, Gemini, OpenAI, Anthropic, Mathpix, Datalab, Unstructured, Upstage, Landing AI, Google Document AI and the self-hosted PaddleOCR ([help verify them](https://github.com/ajinkyashejul/puffinparse/issues/10)). [Full table](#model-names) |
|
|
65
65
|
| **Modes** | `parse` (markdown + blocks), `ocr` (plain text + boxes), `extract` (JSON from a schema) |
|
|
66
66
|
| **Core** | Rust (`puffinparse-core`): `reqwest` + `tokio`, no vendor SDKs, `#![forbid(unsafe_code)]` |
|
|
67
|
-
| **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](js/README.md)), both on the same Rust core |
|
|
68
|
-
| **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](docs/SERVER.md)) |
|
|
67
|
+
| **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md)), both on the same Rust core |
|
|
68
|
+
| **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md)) |
|
|
69
69
|
| **Long documents** | `submit` / `retrieve` jobs and provider webhooks instead of a blocking call ([below](#long-documents-jobs-and-webhooks)) |
|
|
70
70
|
| **CLI** | `puffinparse parse`, `puffinparse ocr`, `puffinparse extract`, `puffinparse providers`, `puffinparse bench` |
|
|
71
71
|
| **Reliability** | Retries with jittered backoff, whole-call deadlines, `Router` with ordered fallbacks / round-robin |
|
|
72
|
-
| **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](docs/COMPAT.md)) |
|
|
72
|
+
| **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md)) |
|
|
73
73
|
| **Cost** | Embedded, overridable price table → `cost_usd` on every response |
|
|
74
74
|
| **Benchmark** | One harness over synthetic data and public benchmarks (ParseBench, olmOCR-bench, OmniDocBench, DP-Bench); deterministic metrics and rule checks, latency, $/1k pages; every output inspectable at [puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/) |
|
|
75
75
|
|
|
76
76
|
## Install
|
|
77
77
|
|
|
78
78
|
```bash
|
|
79
|
-
pip install puffinparse # Python SDK (abi3 wheels: Linux
|
|
80
|
-
|
|
81
|
-
cargo install puffinparse-cli # CLI + gateway; or grab an archive from GitHub Releases
|
|
82
|
-
docker pull ghcr.io/ajinkyashejul/puffinparse # gateway image
|
|
79
|
+
pip install puffinparse # Python SDK (abi3 wheels: Linux glibc 2.28+, macOS, Windows)
|
|
80
|
+
docker pull --platform linux/amd64 ghcr.io/ajinkyashejul/puffinparse # gateway image (amd64 only for now)
|
|
83
81
|
```
|
|
84
82
|
|
|
83
|
+
The CLI (which also runs the gateway) is a single binary: download the archive for your platform
|
|
84
|
+
from [GitHub Releases](https://github.com/ajinkyashejul/puffinparse/releases/latest), or build it
|
|
85
|
+
with `cargo install --git https://github.com/ajinkyashejul/puffinparse puffinparse-cli`. On macOS, a
|
|
86
|
+
binary downloaded with a browser needs `xattr -d com.apple.quarantine ./puffinparse` once (it is not
|
|
87
|
+
notarized yet). The Node.js SDK is not on npm yet: build it from [`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md).
|
|
88
|
+
|
|
85
89
|
Set the keys for the providers you use:
|
|
86
90
|
|
|
87
91
|
```bash
|
|
@@ -94,10 +98,11 @@ No key yet? The self-hosted engines work out of the box once installed:
|
|
|
94
98
|
|
|
95
99
|
```bash
|
|
96
100
|
sudo apt-get install tesseract-ocr poppler-utils # or: brew install tesseract poppler
|
|
97
|
-
puffinparse ocr
|
|
101
|
+
python -c 'import puffinparse; print(puffinparse.ocr("scan.png", model="tesseract").text)'
|
|
102
|
+
puffinparse ocr scan.png -m tesseract # same with the CLI binary: word boxes + confidences
|
|
98
103
|
```
|
|
99
104
|
|
|
100
|
-
Every provider reads its own variable: [`.env.example`](.env.example) lists all of them, the
|
|
105
|
+
Every provider reads its own variable: [`.env.example`](https://github.com/ajinkyashejul/puffinparse/blob/main/.env.example) lists all of them, the
|
|
101
106
|
[model tables](#model-names) say which belongs to which provider, and `puffinparse providers` shows
|
|
102
107
|
which ones are set in your shell.
|
|
103
108
|
|
|
@@ -167,7 +172,8 @@ text.text # whole document, pages joined by a blank line
|
|
|
167
172
|
page = text.pages[0]
|
|
168
173
|
page.text # plain text in reading order
|
|
169
174
|
for line in page.lines: # Line(text, bbox, confidence)
|
|
170
|
-
|
|
175
|
+
if line.bbox and page.width and page.height: # normalised 0-1 box -> pixels
|
|
176
|
+
x0, y0, x1, y1 = line.bbox.to_pixels(page.width, page.height)
|
|
171
177
|
for word in page.words: # Word(text, bbox, confidence)
|
|
172
178
|
...
|
|
173
179
|
```
|
|
@@ -180,7 +186,7 @@ schema = {
|
|
|
180
186
|
"properties": {"invoice_number": {"type": "string"}, "total": {"type": "number"}},
|
|
181
187
|
"required": ["invoice_number", "total"],
|
|
182
188
|
}
|
|
183
|
-
result = puffinparse.extract("invoice.pdf", schema, model="
|
|
189
|
+
result = puffinparse.extract("invoice.pdf", schema, model="reducto/extract", citations=True)
|
|
184
190
|
|
|
185
191
|
result.data # {"invoice_number": "INV-42", "total": 1280.5}
|
|
186
192
|
result.fields["/total"].confidence # per-field confidence, keyed by JSON pointer
|
|
@@ -212,7 +218,7 @@ result = puffinparse.handle_webhook(request.json(), model="reducto") # verify t
|
|
|
212
218
|
|
|
213
219
|
Reducto, Extend and LlamaParse support jobs; `webhook_url` maps to each provider's per-job webhook
|
|
214
220
|
where one exists (Extend only has workspace-level webhooks, so it is rejected there). See
|
|
215
|
-
[`docs/SPEC.md`](docs/SPEC.md) §15.
|
|
221
|
+
[`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md) §15.
|
|
216
222
|
|
|
217
223
|
### TypeScript / Node.js
|
|
218
224
|
|
|
@@ -225,8 +231,9 @@ const doc = await parse("invoice.pdf", { model: "reducto/standard", fallbacks: [
|
|
|
225
231
|
console.log(doc.markdown, doc.usage.pages, doc.costUsd);
|
|
226
232
|
```
|
|
227
233
|
|
|
228
|
-
|
|
229
|
-
|
|
234
|
+
The npm package is not published yet; build it from a clone (`cd js && npm ci && npm run build`), see
|
|
235
|
+
[`js/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md). Prebuilt binaries for Linux x64/arm64 (glibc), macOS and Windows x64
|
|
236
|
+
will ship with the npm release.
|
|
230
237
|
|
|
231
238
|
### Model names
|
|
232
239
|
|
|
@@ -253,7 +260,7 @@ Each provider carries one of three verification labels:
|
|
|
253
260
|
until it is verified; [issue #10](https://github.com/ajinkyashejul/puffinparse/issues/10) tracks this, and a run with your own key is a welcome
|
|
254
261
|
contribution.
|
|
255
262
|
|
|
256
|
-
[`docs/providers/README.md`](docs/providers/README.md) tracks the state and links one reference
|
|
263
|
+
[`docs/providers/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/providers/README.md) tracks the state and links one reference
|
|
257
264
|
page per provider.
|
|
258
265
|
|
|
259
266
|
**Reducto** · `REDUCTO_API_KEY` · live-verified — Layout parsing plus a schema extractor with citations; the default parse target.
|
|
@@ -417,7 +424,7 @@ What is guaranteed is **structural fidelity** — key set and nesting, one chunk
|
|
|
417
424
|
content strings, the vendor's own block vocabulary and coordinate units, the billed page count —
|
|
418
425
|
not byte equality with what the vendor would have returned. Fields PuffinParse does not model
|
|
419
426
|
(presigned URLs, studio links, billing breakdowns, OCR word layers) are `null` or empty, and a few
|
|
420
|
-
block types are lossy. [`docs/COMPAT.md`](docs/COMPAT.md) enumerates all of it, per format;
|
|
427
|
+
block types are lossy. [`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md) enumerates all of it, per format;
|
|
421
428
|
`examples/switch_provider_keep_format.py` is a runnable version of the above.
|
|
422
429
|
|
|
423
430
|
### Provider-specific options
|
|
@@ -502,8 +509,8 @@ curl -H "Authorization: Bearer $TEAM_KEY" -F file=@invoice.pdf -F model=invoices
|
|
|
502
509
|
ordered or round-robin fallback), provider keys as `env:` references, and virtual keys with model
|
|
503
510
|
allow-lists, monthly USD budgets and per-minute limits. `/v1/models`, `/v1/usage`, `/health` and
|
|
504
511
|
Prometheus `/metrics` are built in; request logs are JSON lines that never contain document content
|
|
505
|
-
or secrets. Reference: [`docs/SERVER.md`](docs/SERVER.md), sample:
|
|
506
|
-
[`examples/server/puffinparse.toml`](examples/server/puffinparse.toml).
|
|
512
|
+
or secrets. Reference: [`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md), sample:
|
|
513
|
+
[`examples/server/puffinparse.toml`](https://github.com/ajinkyashejul/puffinparse/blob/main/examples/server/puffinparse.toml).
|
|
507
514
|
|
|
508
515
|
### Rust
|
|
509
516
|
|
|
@@ -516,38 +523,39 @@ println!("{} pages, ${:.4}\n{}", doc.usage.pages, doc.cost_usd.unwrap_or(0.0), d
|
|
|
516
523
|
let text = ocr(DocumentRequest::from_path("scan.png").model("llamaparse/fast")).await?;
|
|
517
524
|
println!("{} lines on page 1", text.pages[0].lines.len());
|
|
518
525
|
|
|
519
|
-
let
|
|
526
|
+
let schema = serde_json::json!({"type": "object", "properties": {"total": {"type": "number"}}});
|
|
527
|
+
let req = ExtractRequest::new(DocumentRequest::from_path("invoice.pdf").model("reducto/extract"), schema);
|
|
520
528
|
let data = extract(req.citations(true)).await?.data;
|
|
521
529
|
```
|
|
522
530
|
|
|
523
531
|
## Benchmark
|
|
524
532
|
|
|
525
|
-
PuffinParse ships an open, reproducible benchmark.
|
|
533
|
+
PuffinParse ships an open, reproducible benchmark. Pages from public benchmarks are scored against their own published truth or rules, the synthetic set's truth is exact by construction (its documents are rendered from the same source as the truth files), metrics are deterministic text comparisons with no LLM judge, and every run records the dataset hash, model, latency and cost.
|
|
526
534
|
|
|
527
535
|
```bash
|
|
528
536
|
python benchmark/generate_synthetic.py # regenerate the dataset (byte-reproducible)
|
|
529
537
|
puffinparse bench run --dataset benchmark/datasets/synthetic-v1 \
|
|
530
538
|
--models reducto/standard extend/parse_performance llamaparse/cost_effective
|
|
531
|
-
puffinparse bench report benchmark/results
|
|
539
|
+
puffinparse bench report benchmark/results/2026-09-25-combined-v3.json # one dataset's leaderboard section
|
|
532
540
|
puffinparse bench score prediction.md truth.md # metrics for one pair, no network
|
|
533
541
|
```
|
|
534
542
|
|
|
535
543
|
Metrics (after NFKC + markdown stripping + whitespace collapsing, case-insensitive by default):
|
|
536
544
|
|
|
537
|
-
- **Overall** = `100
|
|
545
|
+
- **Overall** = `100 ×` the mean of each document's headline metric: `char_similarity` (`1 − levenshtein / max(len)`) for transcripts, the table score for table-only pages, the rule pass rate for rule-checked pages; a failed call scores 0 (details in [`benchmark/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/README.md))
|
|
538
546
|
- **CER**, **WER**, **word F1** (bag-of-words precision / recall)
|
|
539
547
|
- **Order**: Kendall-τ-style agreement of shared line order (reading order)
|
|
540
548
|
- **Table**: character similarity restricted to markdown table rows
|
|
541
549
|
- **Latency** p50 / p95 and ms per page; **$/1k pages** from the price table
|
|
542
550
|
|
|
543
|
-
The current leaderboard is in [`benchmark/LEADERBOARD.md`](benchmark/LEADERBOARD.md), and every
|
|
551
|
+
The current leaderboard is in [`benchmark/LEADERBOARD.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/LEADERBOARD.md), and every
|
|
544
552
|
document, output, diff and rule check is browsable at
|
|
545
553
|
[puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/). Datasets:
|
|
546
554
|
`synthetic-v1` (exact truth by construction) and the headline `combined-v3`, which adds subsets of
|
|
547
555
|
[ParseBench](https://github.com/run-llama/ParseBench), [olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench),
|
|
548
556
|
[OmniDocBench](https://github.com/opendatalab/OmniDocBench) and
|
|
549
557
|
[DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) converted by
|
|
550
|
-
[`benchmark/adapters/`](docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
|
|
558
|
+
[`benchmark/adapters/`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
|
|
551
559
|
`combined-v2` runs are kept for comparison. Long runs are safe:
|
|
552
560
|
`--dry-run` and `--max-cost` show and cap the spend before any call, and `--resume` continues an
|
|
553
561
|
interrupted run.
|
|
@@ -562,7 +570,7 @@ interrupted run.
|
|
|
562
570
|
| Boxes | already normalised | divided by `metadata.page.width/height` | `bBox` divided by page `width/height` |
|
|
563
571
|
| Usage | `usage.num_pages`, `usage.credits` | `metrics.pageCount`, `usage.credits` | `job_metadata.job_pages` |
|
|
564
572
|
|
|
565
|
-
Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](docs/SPEC.md).
|
|
573
|
+
Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md).
|
|
566
574
|
|
|
567
575
|
## Project layout
|
|
568
576
|
|
|
@@ -583,12 +591,12 @@ docs/SPEC.md specification
|
|
|
583
591
|
Planned work is tracked in [GitHub Issues](https://github.com/ajinkyashejul/puffinparse/issues).
|
|
584
592
|
Open items:
|
|
585
593
|
|
|
586
|
-
- **Packaging**:
|
|
587
|
-
|
|
594
|
+
- **Packaging**: PyPI wheels and CLI binaries ship today; npm and crates.io are next, plus a
|
|
595
|
+
multi-arch gateway image and a notarized macOS binary ([#9](https://github.com/ajinkyashejul/puffinparse/issues/9)).
|
|
588
596
|
- **Providers**: live-verify the docs-only providers ([#10](https://github.com/ajinkyashejul/puffinparse/issues/10)), including PaddleOCR
|
|
589
597
|
against a real server ([#17](https://github.com/ajinkyashejul/puffinparse/issues/17)).
|
|
590
|
-
- **Benchmark**: add `
|
|
591
|
-
([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3
|
|
598
|
+
- **Benchmark**: add `docling/default` with a hardware note
|
|
599
|
+
([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3` (`tesseract/default` is already in it); regenerate the leaderboard in one command
|
|
592
600
|
([#13](https://github.com/ajinkyashejul/puffinparse/issues/13)); score table-cell neighbour relations exactly ([#15](https://github.com/ajinkyashejul/puffinparse/issues/15)); a READoc
|
|
593
601
|
long-document track ([#16](https://github.com/ajinkyashejul/puffinparse/issues/16)).
|
|
594
602
|
- **SDK**: `output_format="mistral"` for code written against Mistral OCR responses
|
|
@@ -604,9 +612,45 @@ developers find it. Reports of a wrong score, a missing provider or a confusing
|
|
|
604
612
|
|
|
605
613
|
## Contributing
|
|
606
614
|
|
|
607
|
-
See [CONTRIBUTING.md](CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](.github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](AGENTS.md) summarises the build commands and conventions for coding agents.
|
|
615
|
+
See [CONTRIBUTING.md](https://github.com/ajinkyashejul/puffinparse/blob/main/CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](https://github.com/ajinkyashejul/puffinparse/blob/main/github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](https://github.com/ajinkyashejul/puffinparse/blob/main/AGENTS.md) summarises the build commands and conventions for coding agents.
|
|
616
|
+
|
|
617
|
+
## Acknowledgements
|
|
618
|
+
|
|
619
|
+
PuffinParse stands on other people's work:
|
|
620
|
+
|
|
621
|
+
- **Benchmarks and datasets.** The combined benchmark is built on
|
|
622
|
+
[ParseBench](https://github.com/run-llama/ParseBench) (LlamaIndex; Zhang et al., 2026,
|
|
623
|
+
arXiv:2604.08538; Apache-2.0),
|
|
624
|
+
[olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench) (Allen Institute for AI;
|
|
625
|
+
Poznanski et al., 2025, arXiv:2502.18443; ODC-BY-1.0),
|
|
626
|
+
[OmniDocBench](https://github.com/opendatalab/OmniDocBench) (OpenDataLab / Shanghai AI
|
|
627
|
+
Laboratory; Ouyang et al., 2024, arXiv:2412.07626; research-only, so only an index is committed)
|
|
628
|
+
and [DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) (Upstage AI; MIT). Each
|
|
629
|
+
`benchmark/datasets/<name>/README.md` gives the pinned revision, the licence, what we changed and
|
|
630
|
+
the citation the authors ask for. If you use these numbers, please cite those benchmarks too.
|
|
631
|
+
- **Metrics.** The rule checks re-implement the test semantics of olmOCR-bench and ParseBench, and
|
|
632
|
+
table structure is scored with TEDS (Zhong, ShafieiBavani and Jimeno Yepes, 2020, "Image-based
|
|
633
|
+
table recognition: data, model, and evaluation"). The scorer is our own Rust code; the source
|
|
634
|
+
comments say whose behaviour each part mirrors.
|
|
635
|
+
- **API design.** The `"<provider>/<model>"` model strings and the gateway server follow
|
|
636
|
+
[LiteLLM](https://github.com/BerriAI/litellm), which did this first for LLM APIs.
|
|
637
|
+
- **Providers and engines.** The compatibility shapes mirror the public response formats of
|
|
638
|
+
Reducto, Extend and LlamaParse so existing code can switch, and the open baselines are
|
|
639
|
+
[Tesseract](https://github.com/tesseract-ocr/tesseract) and
|
|
640
|
+
[Docling](https://github.com/docling-project/docling). Provider names are trademarks of their
|
|
641
|
+
owners. PuffinParse is not affiliated with or endorsed by any of them.
|
|
642
|
+
- **Libraries.** [tokio](https://tokio.rs), [reqwest](https://github.com/seanmonstar/reqwest),
|
|
643
|
+
[serde](https://serde.rs), [axum](https://github.com/tokio-rs/axum),
|
|
644
|
+
[clap](https://github.com/clap-rs/clap), [PyO3](https://pyo3.rs),
|
|
645
|
+
[maturin](https://github.com/PyO3/maturin), [napi-rs](https://napi.rs) and the other crates in
|
|
646
|
+
[THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md).
|
|
647
|
+
The site uses GitHub's [Octicons](https://github.com/primer/octicons) GitHub mark (MIT), and the
|
|
648
|
+
results viewer uses [pdf.js](https://github.com/mozilla/pdf.js) (Apache-2.0).
|
|
608
649
|
|
|
609
650
|
## License
|
|
610
651
|
|
|
611
|
-
MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE).
|
|
652
|
+
MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE). The third-party
|
|
653
|
+
crates compiled into the binaries are listed in
|
|
654
|
+
[THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md),
|
|
655
|
+
and the benchmark data keeps its own licence, stated per dataset.
|
|
612
656
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
# PuffinParse
|
|
4
4
|
|
|
5
|
-
[](https://pypi.org/project/puffinparse/) [](https://github.com/ajinkyashejul/puffinparse/stargazers) [](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [](https://puffinparse.com/docs/) [](https://puffinparse.com/benchmark-results/) [](pyproject.toml)
|
|
5
|
+
[](https://pypi.org/project/puffinparse/) [](https://github.com/ajinkyashejul/puffinparse/stargazers) [](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [](https://puffinparse.com/docs/) [](https://puffinparse.com/benchmark-results/) [](https://github.com/ajinkyashejul/puffinparse/blob/main/pyproject.toml)
|
|
6
6
|
|
|
7
7
|
**One API for every document parser: parse, OCR and extract.** Rust core, Python and TypeScript SDKs, a CLI, a self-hosted gateway, and an open benchmark that ranks providers on accuracy, latency and cost.
|
|
8
8
|
|
|
@@ -30,24 +30,28 @@ Two things make switching real rather than aspirational. **Modes**: every call n
|
|
|
30
30
|
| **Providers (v0.1)** | 18 providers · 60 models · 3 modes. **Live-verified** against the real APIs: [Reducto](https://reducto.ai), [Extend](https://extend.ai), [LlamaParse](https://cloud.llamaindex.ai). **Verified locally**: the self-hosted Tesseract and Docling. **Docs-only** (implemented from the provider's API documentation and tested against fixture payloads, not yet run live): Mistral, Azure, Textract, Gemini, OpenAI, Anthropic, Mathpix, Datalab, Unstructured, Upstage, Landing AI, Google Document AI and the self-hosted PaddleOCR ([help verify them](https://github.com/ajinkyashejul/puffinparse/issues/10)). [Full table](#model-names) |
|
|
31
31
|
| **Modes** | `parse` (markdown + blocks), `ocr` (plain text + boxes), `extract` (JSON from a schema) |
|
|
32
32
|
| **Core** | Rust (`puffinparse-core`): `reqwest` + `tokio`, no vendor SDKs, `#![forbid(unsafe_code)]` |
|
|
33
|
-
| **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](js/README.md)), both on the same Rust core |
|
|
34
|
-
| **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](docs/SERVER.md)) |
|
|
33
|
+
| **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md)), both on the same Rust core |
|
|
34
|
+
| **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md)) |
|
|
35
35
|
| **Long documents** | `submit` / `retrieve` jobs and provider webhooks instead of a blocking call ([below](#long-documents-jobs-and-webhooks)) |
|
|
36
36
|
| **CLI** | `puffinparse parse`, `puffinparse ocr`, `puffinparse extract`, `puffinparse providers`, `puffinparse bench` |
|
|
37
37
|
| **Reliability** | Retries with jittered backoff, whole-call deadlines, `Router` with ordered fallbacks / round-robin |
|
|
38
|
-
| **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](docs/COMPAT.md)) |
|
|
38
|
+
| **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md)) |
|
|
39
39
|
| **Cost** | Embedded, overridable price table → `cost_usd` on every response |
|
|
40
40
|
| **Benchmark** | One harness over synthetic data and public benchmarks (ParseBench, olmOCR-bench, OmniDocBench, DP-Bench); deterministic metrics and rule checks, latency, $/1k pages; every output inspectable at [puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/) |
|
|
41
41
|
|
|
42
42
|
## Install
|
|
43
43
|
|
|
44
44
|
```bash
|
|
45
|
-
pip install puffinparse # Python SDK (abi3 wheels: Linux
|
|
46
|
-
|
|
47
|
-
cargo install puffinparse-cli # CLI + gateway; or grab an archive from GitHub Releases
|
|
48
|
-
docker pull ghcr.io/ajinkyashejul/puffinparse # gateway image
|
|
45
|
+
pip install puffinparse # Python SDK (abi3 wheels: Linux glibc 2.28+, macOS, Windows)
|
|
46
|
+
docker pull --platform linux/amd64 ghcr.io/ajinkyashejul/puffinparse # gateway image (amd64 only for now)
|
|
49
47
|
```
|
|
50
48
|
|
|
49
|
+
The CLI (which also runs the gateway) is a single binary: download the archive for your platform
|
|
50
|
+
from [GitHub Releases](https://github.com/ajinkyashejul/puffinparse/releases/latest), or build it
|
|
51
|
+
with `cargo install --git https://github.com/ajinkyashejul/puffinparse puffinparse-cli`. On macOS, a
|
|
52
|
+
binary downloaded with a browser needs `xattr -d com.apple.quarantine ./puffinparse` once (it is not
|
|
53
|
+
notarized yet). The Node.js SDK is not on npm yet: build it from [`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md).
|
|
54
|
+
|
|
51
55
|
Set the keys for the providers you use:
|
|
52
56
|
|
|
53
57
|
```bash
|
|
@@ -60,10 +64,11 @@ No key yet? The self-hosted engines work out of the box once installed:
|
|
|
60
64
|
|
|
61
65
|
```bash
|
|
62
66
|
sudo apt-get install tesseract-ocr poppler-utils # or: brew install tesseract poppler
|
|
63
|
-
puffinparse ocr
|
|
67
|
+
python -c 'import puffinparse; print(puffinparse.ocr("scan.png", model="tesseract").text)'
|
|
68
|
+
puffinparse ocr scan.png -m tesseract # same with the CLI binary: word boxes + confidences
|
|
64
69
|
```
|
|
65
70
|
|
|
66
|
-
Every provider reads its own variable: [`.env.example`](.env.example) lists all of them, the
|
|
71
|
+
Every provider reads its own variable: [`.env.example`](https://github.com/ajinkyashejul/puffinparse/blob/main/.env.example) lists all of them, the
|
|
67
72
|
[model tables](#model-names) say which belongs to which provider, and `puffinparse providers` shows
|
|
68
73
|
which ones are set in your shell.
|
|
69
74
|
|
|
@@ -133,7 +138,8 @@ text.text # whole document, pages joined by a blank line
|
|
|
133
138
|
page = text.pages[0]
|
|
134
139
|
page.text # plain text in reading order
|
|
135
140
|
for line in page.lines: # Line(text, bbox, confidence)
|
|
136
|
-
|
|
141
|
+
if line.bbox and page.width and page.height: # normalised 0-1 box -> pixels
|
|
142
|
+
x0, y0, x1, y1 = line.bbox.to_pixels(page.width, page.height)
|
|
137
143
|
for word in page.words: # Word(text, bbox, confidence)
|
|
138
144
|
...
|
|
139
145
|
```
|
|
@@ -146,7 +152,7 @@ schema = {
|
|
|
146
152
|
"properties": {"invoice_number": {"type": "string"}, "total": {"type": "number"}},
|
|
147
153
|
"required": ["invoice_number", "total"],
|
|
148
154
|
}
|
|
149
|
-
result = puffinparse.extract("invoice.pdf", schema, model="
|
|
155
|
+
result = puffinparse.extract("invoice.pdf", schema, model="reducto/extract", citations=True)
|
|
150
156
|
|
|
151
157
|
result.data # {"invoice_number": "INV-42", "total": 1280.5}
|
|
152
158
|
result.fields["/total"].confidence # per-field confidence, keyed by JSON pointer
|
|
@@ -178,7 +184,7 @@ result = puffinparse.handle_webhook(request.json(), model="reducto") # verify t
|
|
|
178
184
|
|
|
179
185
|
Reducto, Extend and LlamaParse support jobs; `webhook_url` maps to each provider's per-job webhook
|
|
180
186
|
where one exists (Extend only has workspace-level webhooks, so it is rejected there). See
|
|
181
|
-
[`docs/SPEC.md`](docs/SPEC.md) §15.
|
|
187
|
+
[`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md) §15.
|
|
182
188
|
|
|
183
189
|
### TypeScript / Node.js
|
|
184
190
|
|
|
@@ -191,8 +197,9 @@ const doc = await parse("invoice.pdf", { model: "reducto/standard", fallbacks: [
|
|
|
191
197
|
console.log(doc.markdown, doc.usage.pages, doc.costUsd);
|
|
192
198
|
```
|
|
193
199
|
|
|
194
|
-
|
|
195
|
-
|
|
200
|
+
The npm package is not published yet; build it from a clone (`cd js && npm ci && npm run build`), see
|
|
201
|
+
[`js/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md). Prebuilt binaries for Linux x64/arm64 (glibc), macOS and Windows x64
|
|
202
|
+
will ship with the npm release.
|
|
196
203
|
|
|
197
204
|
### Model names
|
|
198
205
|
|
|
@@ -219,7 +226,7 @@ Each provider carries one of three verification labels:
|
|
|
219
226
|
until it is verified; [issue #10](https://github.com/ajinkyashejul/puffinparse/issues/10) tracks this, and a run with your own key is a welcome
|
|
220
227
|
contribution.
|
|
221
228
|
|
|
222
|
-
[`docs/providers/README.md`](docs/providers/README.md) tracks the state and links one reference
|
|
229
|
+
[`docs/providers/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/providers/README.md) tracks the state and links one reference
|
|
223
230
|
page per provider.
|
|
224
231
|
|
|
225
232
|
**Reducto** · `REDUCTO_API_KEY` · live-verified — Layout parsing plus a schema extractor with citations; the default parse target.
|
|
@@ -383,7 +390,7 @@ What is guaranteed is **structural fidelity** — key set and nesting, one chunk
|
|
|
383
390
|
content strings, the vendor's own block vocabulary and coordinate units, the billed page count —
|
|
384
391
|
not byte equality with what the vendor would have returned. Fields PuffinParse does not model
|
|
385
392
|
(presigned URLs, studio links, billing breakdowns, OCR word layers) are `null` or empty, and a few
|
|
386
|
-
block types are lossy. [`docs/COMPAT.md`](docs/COMPAT.md) enumerates all of it, per format;
|
|
393
|
+
block types are lossy. [`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md) enumerates all of it, per format;
|
|
387
394
|
`examples/switch_provider_keep_format.py` is a runnable version of the above.
|
|
388
395
|
|
|
389
396
|
### Provider-specific options
|
|
@@ -468,8 +475,8 @@ curl -H "Authorization: Bearer $TEAM_KEY" -F file=@invoice.pdf -F model=invoices
|
|
|
468
475
|
ordered or round-robin fallback), provider keys as `env:` references, and virtual keys with model
|
|
469
476
|
allow-lists, monthly USD budgets and per-minute limits. `/v1/models`, `/v1/usage`, `/health` and
|
|
470
477
|
Prometheus `/metrics` are built in; request logs are JSON lines that never contain document content
|
|
471
|
-
or secrets. Reference: [`docs/SERVER.md`](docs/SERVER.md), sample:
|
|
472
|
-
[`examples/server/puffinparse.toml`](examples/server/puffinparse.toml).
|
|
478
|
+
or secrets. Reference: [`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md), sample:
|
|
479
|
+
[`examples/server/puffinparse.toml`](https://github.com/ajinkyashejul/puffinparse/blob/main/examples/server/puffinparse.toml).
|
|
473
480
|
|
|
474
481
|
### Rust
|
|
475
482
|
|
|
@@ -482,38 +489,39 @@ println!("{} pages, ${:.4}\n{}", doc.usage.pages, doc.cost_usd.unwrap_or(0.0), d
|
|
|
482
489
|
let text = ocr(DocumentRequest::from_path("scan.png").model("llamaparse/fast")).await?;
|
|
483
490
|
println!("{} lines on page 1", text.pages[0].lines.len());
|
|
484
491
|
|
|
485
|
-
let
|
|
492
|
+
let schema = serde_json::json!({"type": "object", "properties": {"total": {"type": "number"}}});
|
|
493
|
+
let req = ExtractRequest::new(DocumentRequest::from_path("invoice.pdf").model("reducto/extract"), schema);
|
|
486
494
|
let data = extract(req.citations(true)).await?.data;
|
|
487
495
|
```
|
|
488
496
|
|
|
489
497
|
## Benchmark
|
|
490
498
|
|
|
491
|
-
PuffinParse ships an open, reproducible benchmark.
|
|
499
|
+
PuffinParse ships an open, reproducible benchmark. Pages from public benchmarks are scored against their own published truth or rules, the synthetic set's truth is exact by construction (its documents are rendered from the same source as the truth files), metrics are deterministic text comparisons with no LLM judge, and every run records the dataset hash, model, latency and cost.
|
|
492
500
|
|
|
493
501
|
```bash
|
|
494
502
|
python benchmark/generate_synthetic.py # regenerate the dataset (byte-reproducible)
|
|
495
503
|
puffinparse bench run --dataset benchmark/datasets/synthetic-v1 \
|
|
496
504
|
--models reducto/standard extend/parse_performance llamaparse/cost_effective
|
|
497
|
-
puffinparse bench report benchmark/results
|
|
505
|
+
puffinparse bench report benchmark/results/2026-09-25-combined-v3.json # one dataset's leaderboard section
|
|
498
506
|
puffinparse bench score prediction.md truth.md # metrics for one pair, no network
|
|
499
507
|
```
|
|
500
508
|
|
|
501
509
|
Metrics (after NFKC + markdown stripping + whitespace collapsing, case-insensitive by default):
|
|
502
510
|
|
|
503
|
-
- **Overall** = `100
|
|
511
|
+
- **Overall** = `100 ×` the mean of each document's headline metric: `char_similarity` (`1 − levenshtein / max(len)`) for transcripts, the table score for table-only pages, the rule pass rate for rule-checked pages; a failed call scores 0 (details in [`benchmark/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/README.md))
|
|
504
512
|
- **CER**, **WER**, **word F1** (bag-of-words precision / recall)
|
|
505
513
|
- **Order**: Kendall-τ-style agreement of shared line order (reading order)
|
|
506
514
|
- **Table**: character similarity restricted to markdown table rows
|
|
507
515
|
- **Latency** p50 / p95 and ms per page; **$/1k pages** from the price table
|
|
508
516
|
|
|
509
|
-
The current leaderboard is in [`benchmark/LEADERBOARD.md`](benchmark/LEADERBOARD.md), and every
|
|
517
|
+
The current leaderboard is in [`benchmark/LEADERBOARD.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/LEADERBOARD.md), and every
|
|
510
518
|
document, output, diff and rule check is browsable at
|
|
511
519
|
[puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/). Datasets:
|
|
512
520
|
`synthetic-v1` (exact truth by construction) and the headline `combined-v3`, which adds subsets of
|
|
513
521
|
[ParseBench](https://github.com/run-llama/ParseBench), [olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench),
|
|
514
522
|
[OmniDocBench](https://github.com/opendatalab/OmniDocBench) and
|
|
515
523
|
[DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) converted by
|
|
516
|
-
[`benchmark/adapters/`](docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
|
|
524
|
+
[`benchmark/adapters/`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
|
|
517
525
|
`combined-v2` runs are kept for comparison. Long runs are safe:
|
|
518
526
|
`--dry-run` and `--max-cost` show and cap the spend before any call, and `--resume` continues an
|
|
519
527
|
interrupted run.
|
|
@@ -528,7 +536,7 @@ interrupted run.
|
|
|
528
536
|
| Boxes | already normalised | divided by `metadata.page.width/height` | `bBox` divided by page `width/height` |
|
|
529
537
|
| Usage | `usage.num_pages`, `usage.credits` | `metrics.pageCount`, `usage.credits` | `job_metadata.job_pages` |
|
|
530
538
|
|
|
531
|
-
Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](docs/SPEC.md).
|
|
539
|
+
Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md).
|
|
532
540
|
|
|
533
541
|
## Project layout
|
|
534
542
|
|
|
@@ -549,12 +557,12 @@ docs/SPEC.md specification
|
|
|
549
557
|
Planned work is tracked in [GitHub Issues](https://github.com/ajinkyashejul/puffinparse/issues).
|
|
550
558
|
Open items:
|
|
551
559
|
|
|
552
|
-
- **Packaging**:
|
|
553
|
-
|
|
560
|
+
- **Packaging**: PyPI wheels and CLI binaries ship today; npm and crates.io are next, plus a
|
|
561
|
+
multi-arch gateway image and a notarized macOS binary ([#9](https://github.com/ajinkyashejul/puffinparse/issues/9)).
|
|
554
562
|
- **Providers**: live-verify the docs-only providers ([#10](https://github.com/ajinkyashejul/puffinparse/issues/10)), including PaddleOCR
|
|
555
563
|
against a real server ([#17](https://github.com/ajinkyashejul/puffinparse/issues/17)).
|
|
556
|
-
- **Benchmark**: add `
|
|
557
|
-
([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3
|
|
564
|
+
- **Benchmark**: add `docling/default` with a hardware note
|
|
565
|
+
([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3` (`tesseract/default` is already in it); regenerate the leaderboard in one command
|
|
558
566
|
([#13](https://github.com/ajinkyashejul/puffinparse/issues/13)); score table-cell neighbour relations exactly ([#15](https://github.com/ajinkyashejul/puffinparse/issues/15)); a READoc
|
|
559
567
|
long-document track ([#16](https://github.com/ajinkyashejul/puffinparse/issues/16)).
|
|
560
568
|
- **SDK**: `output_format="mistral"` for code written against Mistral OCR responses
|
|
@@ -570,8 +578,44 @@ developers find it. Reports of a wrong score, a missing provider or a confusing
|
|
|
570
578
|
|
|
571
579
|
## Contributing
|
|
572
580
|
|
|
573
|
-
See [CONTRIBUTING.md](CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](.github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](AGENTS.md) summarises the build commands and conventions for coding agents.
|
|
581
|
+
See [CONTRIBUTING.md](https://github.com/ajinkyashejul/puffinparse/blob/main/CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](https://github.com/ajinkyashejul/puffinparse/blob/main/github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](https://github.com/ajinkyashejul/puffinparse/blob/main/AGENTS.md) summarises the build commands and conventions for coding agents.
|
|
582
|
+
|
|
583
|
+
## Acknowledgements
|
|
584
|
+
|
|
585
|
+
PuffinParse stands on other people's work:
|
|
586
|
+
|
|
587
|
+
- **Benchmarks and datasets.** The combined benchmark is built on
|
|
588
|
+
[ParseBench](https://github.com/run-llama/ParseBench) (LlamaIndex; Zhang et al., 2026,
|
|
589
|
+
arXiv:2604.08538; Apache-2.0),
|
|
590
|
+
[olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench) (Allen Institute for AI;
|
|
591
|
+
Poznanski et al., 2025, arXiv:2502.18443; ODC-BY-1.0),
|
|
592
|
+
[OmniDocBench](https://github.com/opendatalab/OmniDocBench) (OpenDataLab / Shanghai AI
|
|
593
|
+
Laboratory; Ouyang et al., 2024, arXiv:2412.07626; research-only, so only an index is committed)
|
|
594
|
+
and [DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) (Upstage AI; MIT). Each
|
|
595
|
+
`benchmark/datasets/<name>/README.md` gives the pinned revision, the licence, what we changed and
|
|
596
|
+
the citation the authors ask for. If you use these numbers, please cite those benchmarks too.
|
|
597
|
+
- **Metrics.** The rule checks re-implement the test semantics of olmOCR-bench and ParseBench, and
|
|
598
|
+
table structure is scored with TEDS (Zhong, ShafieiBavani and Jimeno Yepes, 2020, "Image-based
|
|
599
|
+
table recognition: data, model, and evaluation"). The scorer is our own Rust code; the source
|
|
600
|
+
comments say whose behaviour each part mirrors.
|
|
601
|
+
- **API design.** The `"<provider>/<model>"` model strings and the gateway server follow
|
|
602
|
+
[LiteLLM](https://github.com/BerriAI/litellm), which did this first for LLM APIs.
|
|
603
|
+
- **Providers and engines.** The compatibility shapes mirror the public response formats of
|
|
604
|
+
Reducto, Extend and LlamaParse so existing code can switch, and the open baselines are
|
|
605
|
+
[Tesseract](https://github.com/tesseract-ocr/tesseract) and
|
|
606
|
+
[Docling](https://github.com/docling-project/docling). Provider names are trademarks of their
|
|
607
|
+
owners. PuffinParse is not affiliated with or endorsed by any of them.
|
|
608
|
+
- **Libraries.** [tokio](https://tokio.rs), [reqwest](https://github.com/seanmonstar/reqwest),
|
|
609
|
+
[serde](https://serde.rs), [axum](https://github.com/tokio-rs/axum),
|
|
610
|
+
[clap](https://github.com/clap-rs/clap), [PyO3](https://pyo3.rs),
|
|
611
|
+
[maturin](https://github.com/PyO3/maturin), [napi-rs](https://napi.rs) and the other crates in
|
|
612
|
+
[THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md).
|
|
613
|
+
The site uses GitHub's [Octicons](https://github.com/primer/octicons) GitHub mark (MIT), and the
|
|
614
|
+
results viewer uses [pdf.js](https://github.com/mozilla/pdf.js) (Apache-2.0).
|
|
574
615
|
|
|
575
616
|
## License
|
|
576
617
|
|
|
577
|
-
MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE).
|
|
618
|
+
MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE). The third-party
|
|
619
|
+
crates compiled into the binaries are listed in
|
|
620
|
+
[THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md),
|
|
621
|
+
and the benchmark data keeps its own licence, stated per dataset.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# puffinparse-core
|
|
2
|
+
|
|
3
|
+
Rust core of [PuffinParse](https://github.com/ajinkyashejul/puffinparse): one API for every
|
|
4
|
+
OCR / document-parsing provider (Reducto, Extend, LlamaParse, Mistral, Azure, Textract, …).
|
|
5
|
+
|
|
6
|
+
```rust
|
|
7
|
+
use puffinparse_core::{parse, DocumentRequest};
|
|
8
|
+
|
|
9
|
+
#[tokio::main]
|
|
10
|
+
async fn main() -> Result<(), puffinparse_core::Error> {
|
|
11
|
+
let doc = parse(DocumentRequest::from_path("invoice.pdf").model("reducto/standard")).await?;
|
|
12
|
+
println!("{} pages, ${:.4}: {}", doc.usage.pages, doc.cost_usd.unwrap_or(0.0), doc.markdown);
|
|
13
|
+
Ok(())
|
|
14
|
+
}
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
`ocr` returns plain text with line and word boxes, and `extract` fills a JSON Schema; see the
|
|
18
|
+
repository README for the Python and TypeScript SDKs, the CLI, the gateway and the benchmark.
|
|
@@ -390,6 +390,11 @@ pub fn decode_entities(s: &str) -> String {
|
|
|
390
390
|
// grids — the same formula, costs and Zhang–Shasha edit distance, just without `thead`/`tbody` and
|
|
391
391
|
// span attributes. It still punishes missing, extra, split or merged rows and cells, which the
|
|
392
392
|
// flat `table_score` string similarity does not see.
|
|
393
|
+
//
|
|
394
|
+
// Credit: the metric is from the paper above; its reference implementation is IBM's PubTabNet
|
|
395
|
+
// `src/metric.py` (github.com/ibm-aur-nlp/PubTabNet, Apache-2.0), which uses APTED. This file is an
|
|
396
|
+
// independent Rust implementation (Zhang & Shasha, 1989, "Simple fast algorithms for the editing
|
|
397
|
+
// distance between trees and related problems"); no PubTabNet code is copied.
|
|
393
398
|
|
|
394
399
|
/// Node budget above which [`teds_grid`] declines (`None`): Zhang–Shasha is O(n·m) memory.
|
|
395
400
|
const TEDS_MAX_PAIRS: usize = 16_000_000;
|