puffinparse 0.1.2__tar.gz → 0.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. {puffinparse-0.1.2 → puffinparse-0.1.3}/Cargo.lock +5 -5
  2. {puffinparse-0.1.2 → puffinparse-0.1.3}/Cargo.toml +2 -2
  3. {puffinparse-0.1.2 → puffinparse-0.1.3}/PKG-INFO +77 -33
  4. {puffinparse-0.1.2 → puffinparse-0.1.3}/README.md +76 -32
  5. puffinparse-0.1.3/crates/puffinparse-core/README.md +18 -0
  6. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/bench/tables.rs +5 -0
  7. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/bench.rs +7 -0
  8. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/tesseract.rs +74 -6
  9. puffinparse-0.1.3/crates/puffinparse-core/tests/fixtures/README.md +14 -0
  10. {puffinparse-0.1.2 → puffinparse-0.1.3}/pyproject.toml +1 -1
  11. puffinparse-0.1.2/crates/puffinparse-core/README.md +0 -17
  12. {puffinparse-0.1.2 → puffinparse-0.1.3}/LICENSE +0 -0
  13. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/Cargo.toml +0 -0
  14. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/compat/extend.rs +0 -0
  15. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/compat/llamaparse.rs +0 -0
  16. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/compat/mod.rs +0 -0
  17. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/compat/reducto.rs +0 -0
  18. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/compat/roundtrip.rs +0 -0
  19. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/error.rs +0 -0
  20. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/http.rs +0 -0
  21. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/jobs.rs +0 -0
  22. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/lib.rs +0 -0
  23. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/model.rs +0 -0
  24. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/pricing.json +0 -0
  25. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/pricing.rs +0 -0
  26. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/provider.rs +0 -0
  27. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/anthropic.rs +0 -0
  28. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/azure.rs +0 -0
  29. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/datalab.rs +0 -0
  30. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/docling.rs +0 -0
  31. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/extend.rs +0 -0
  32. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/gemini.rs +0 -0
  33. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/google_documentai.rs +0 -0
  34. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/landingai.rs +0 -0
  35. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/llamaparse.rs +0 -0
  36. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/local.rs +0 -0
  37. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/mathpix.rs +0 -0
  38. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/mistral.rs +0 -0
  39. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/mod.rs +0 -0
  40. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/openai.rs +0 -0
  41. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/paddleocr.rs +0 -0
  42. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/reducto.rs +0 -0
  43. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/textract.rs +0 -0
  44. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/unstructured.rs +0 -0
  45. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/upstage.rs +0 -0
  46. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/providers/vlm.rs +0 -0
  47. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/router.rs +0 -0
  48. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/testutil.rs +0 -0
  49. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/types.rs +0 -0
  50. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/src/util.rs +0 -0
  51. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/benchmark_datasets.rs +0 -0
  52. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/anthropic_messages_extract.json +0 -0
  53. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/anthropic_messages_parse.json +0 -0
  54. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/azure_invoice.json +0 -0
  55. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/azure_layout.json +0 -0
  56. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/azure_read.json +0 -0
  57. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/datalab_convert.json +0 -0
  58. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/docling_headings.json +0 -0
  59. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/docling_multipage.json +0 -0
  60. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/extend_extract_run.json +0 -0
  61. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/extend_parse_run.json +0 -0
  62. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/extend_parse_run_url.json +0 -0
  63. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/extend_parse_run_url_output.json +0 -0
  64. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/gemini_extract_invoice.json +0 -0
  65. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/gemini_parse_multipage.json +0 -0
  66. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/google_documentai_form.json +0 -0
  67. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/google_documentai_layout.json +0 -0
  68. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/google_documentai_ocr.json +0 -0
  69. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/landingai_extract.json +0 -0
  70. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/landingai_parse.json +0 -0
  71. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/llamaparse_extract_job.json +0 -0
  72. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/llamaparse_result_json.json +0 -0
  73. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/mathpix_pdf_lines.json +0 -0
  74. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/mathpix_text.json +0 -0
  75. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/mistral_annotation.json +0 -0
  76. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/mistral_ocr.json +0 -0
  77. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/mistral_ocr_blocks.json +0 -0
  78. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/openai_responses_extract.json +0 -0
  79. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/openai_responses_parse.json +0 -0
  80. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/paddleocr_layout_parsing.json +0 -0
  81. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/paddleocr_ocr.json +0 -0
  82. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/reducto_extract.json +0 -0
  83. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/reducto_extract_plain.json +0 -0
  84. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/reducto_parse.json +0 -0
  85. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/reducto_parse_url.json +0 -0
  86. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/reducto_parse_url_result.json +0 -0
  87. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/rules_text_sparse_note.json +0 -0
  88. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/tesseract_headings.tsv +0 -0
  89. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/textract_detect_text.json +0 -0
  90. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/textract_forms.json +0 -0
  91. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/textract_layout.json +0 -0
  92. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/textract_queries.json +0 -0
  93. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/unstructured_elements.json +0 -0
  94. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/fixtures/upstage_document_parse.json +0 -0
  95. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/live.rs +0 -0
  96. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-core/tests/live_jobs.rs +0 -0
  97. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-python/Cargo.toml +0 -0
  98. {puffinparse-0.1.2 → puffinparse-0.1.3}/crates/puffinparse-python/src/lib.rs +0 -0
  99. {puffinparse-0.1.2 → puffinparse-0.1.3}/python/puffinparse/__init__.py +0 -0
  100. {puffinparse-0.1.2 → puffinparse-0.1.3}/python/puffinparse/_core.pyi +0 -0
  101. {puffinparse-0.1.2 → puffinparse-0.1.3}/python/puffinparse/exceptions.py +0 -0
  102. {puffinparse-0.1.2 → puffinparse-0.1.3}/python/puffinparse/jobs.py +0 -0
  103. {puffinparse-0.1.2 → puffinparse-0.1.3}/python/puffinparse/main.py +0 -0
  104. {puffinparse-0.1.2 → puffinparse-0.1.3}/python/puffinparse/py.typed +0 -0
  105. {puffinparse-0.1.2 → puffinparse-0.1.3}/python/puffinparse/types.py +0 -0
@@ -1395,7 +1395,7 @@ dependencies = [
1395
1395
 
1396
1396
  [[package]]
1397
1397
  name = "puffinparse-cli"
1398
- version = "0.1.2"
1398
+ version = "0.1.3"
1399
1399
  dependencies = [
1400
1400
  "anyhow",
1401
1401
  "chrono",
@@ -1415,7 +1415,7 @@ dependencies = [
1415
1415
 
1416
1416
  [[package]]
1417
1417
  name = "puffinparse-core"
1418
- version = "0.1.2"
1418
+ version = "0.1.3"
1419
1419
  dependencies = [
1420
1420
  "async-trait",
1421
1421
  "bytes",
@@ -1439,7 +1439,7 @@ dependencies = [
1439
1439
 
1440
1440
  [[package]]
1441
1441
  name = "puffinparse-node"
1442
- version = "0.1.2"
1442
+ version = "0.1.3"
1443
1443
  dependencies = [
1444
1444
  "bytes",
1445
1445
  "napi",
@@ -1453,7 +1453,7 @@ dependencies = [
1453
1453
 
1454
1454
  [[package]]
1455
1455
  name = "puffinparse-python"
1456
- version = "0.1.2"
1456
+ version = "0.1.3"
1457
1457
  dependencies = [
1458
1458
  "bytes",
1459
1459
  "puffinparse-core",
@@ -1468,7 +1468,7 @@ dependencies = [
1468
1468
 
1469
1469
  [[package]]
1470
1470
  name = "puffinparse-server"
1471
- version = "0.1.2"
1471
+ version = "0.1.3"
1472
1472
  dependencies = [
1473
1473
  "axum",
1474
1474
  "bytes",
@@ -3,7 +3,7 @@ resolver = "2"
3
3
  members = ["crates/puffinparse-core", "crates/puffinparse-python"]
4
4
 
5
5
  [workspace.package]
6
- version = "0.1.2"
6
+ version = "0.1.3"
7
7
  edition = "2021"
8
8
  rust-version = "1.80"
9
9
  license = "MIT"
@@ -15,7 +15,7 @@ keywords = ["ocr", "document-parsing", "pdf", "api", "benchmark"]
15
15
  categories = ["api-bindings", "text-processing"]
16
16
 
17
17
  [workspace.dependencies]
18
- puffinparse-core = { path = "crates/puffinparse-core", version = "0.1.2" }
18
+ puffinparse-core = { path = "crates/puffinparse-core", version = "0.1.3" }
19
19
  reqwest = { version = "0.13", default-features = false, features = ["json", "multipart", "rustls", "stream", "gzip"] }
20
20
  tokio = { version = "1", features = ["rt-multi-thread", "macros", "fs", "time", "sync"] }
21
21
  serde = { version = "1", features = ["derive"] }
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: puffinparse
3
- Version: 0.1.2
3
+ Version: 0.1.3
4
4
  Classifier: Development Status :: 3 - Alpha
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: MIT License
@@ -36,7 +36,7 @@ Project-URL: Repository, https://github.com/ajinkyashejul/puffinparse
36
36
 
37
37
  # PuffinParse
38
38
 
39
- [![PyPI](https://img.shields.io/pypi/v/puffinparse?color=E95C20)](https://pypi.org/project/puffinparse/) [![GitHub stars](https://img.shields.io/github/stars/ajinkyashejul/puffinparse?style=flat&color=E95C20)](https://github.com/ajinkyashejul/puffinparse/stargazers) [![CI](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml/badge.svg)](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [![Docs](https://img.shields.io/badge/docs-puffinparse.com-E95C20)](https://puffinparse.com/docs/) [![Benchmark](https://img.shields.io/badge/benchmark-results-E95C20)](https://puffinparse.com/benchmark-results/) [![Python 3.9+](https://img.shields.io/badge/python-3.9%2B-blue)](pyproject.toml)
39
+ [![PyPI](https://img.shields.io/pypi/v/puffinparse?color=E95C20)](https://pypi.org/project/puffinparse/) [![GitHub stars](https://img.shields.io/github/stars/ajinkyashejul/puffinparse?style=flat&color=E95C20)](https://github.com/ajinkyashejul/puffinparse/stargazers) [![CI](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml/badge.svg)](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [![Docs](https://img.shields.io/badge/docs-puffinparse.com-E95C20)](https://puffinparse.com/docs/) [![Benchmark](https://img.shields.io/badge/benchmark-results-E95C20)](https://puffinparse.com/benchmark-results/) [![Python 3.9+](https://img.shields.io/badge/python-3.9%2B-blue)](https://github.com/ajinkyashejul/puffinparse/blob/main/pyproject.toml)
40
40
 
41
41
  **One API for every document parser: parse, OCR and extract.** Rust core, Python and TypeScript SDKs, a CLI, a self-hosted gateway, and an open benchmark that ranks providers on accuracy, latency and cost.
42
42
 
@@ -64,24 +64,28 @@ Two things make switching real rather than aspirational. **Modes**: every call n
64
64
  | **Providers (v0.1)** | 18 providers · 60 models · 3 modes. **Live-verified** against the real APIs: [Reducto](https://reducto.ai), [Extend](https://extend.ai), [LlamaParse](https://cloud.llamaindex.ai). **Verified locally**: the self-hosted Tesseract and Docling. **Docs-only** (implemented from the provider's API documentation and tested against fixture payloads, not yet run live): Mistral, Azure, Textract, Gemini, OpenAI, Anthropic, Mathpix, Datalab, Unstructured, Upstage, Landing AI, Google Document AI and the self-hosted PaddleOCR ([help verify them](https://github.com/ajinkyashejul/puffinparse/issues/10)). [Full table](#model-names) |
65
65
  | **Modes** | `parse` (markdown + blocks), `ocr` (plain text + boxes), `extract` (JSON from a schema) |
66
66
  | **Core** | Rust (`puffinparse-core`): `reqwest` + `tokio`, no vendor SDKs, `#![forbid(unsafe_code)]` |
67
- | **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](js/README.md)), both on the same Rust core |
68
- | **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](docs/SERVER.md)) |
67
+ | **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md)), both on the same Rust core |
68
+ | **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md)) |
69
69
  | **Long documents** | `submit` / `retrieve` jobs and provider webhooks instead of a blocking call ([below](#long-documents-jobs-and-webhooks)) |
70
70
  | **CLI** | `puffinparse parse`, `puffinparse ocr`, `puffinparse extract`, `puffinparse providers`, `puffinparse bench` |
71
71
  | **Reliability** | Retries with jittered backoff, whole-call deadlines, `Router` with ordered fallbacks / round-robin |
72
- | **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](docs/COMPAT.md)) |
72
+ | **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md)) |
73
73
  | **Cost** | Embedded, overridable price table → `cost_usd` on every response |
74
74
  | **Benchmark** | One harness over synthetic data and public benchmarks (ParseBench, olmOCR-bench, OmniDocBench, DP-Bench); deterministic metrics and rule checks, latency, $/1k pages; every output inspectable at [puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/) |
75
75
 
76
76
  ## Install
77
77
 
78
78
  ```bash
79
- pip install puffinparse # Python SDK (abi3 wheels: Linux, macOS, Windows)
80
- npm install puffinparse # Node.js SDK (prebuilt for Linux x64/arm64 glibc, macOS, Windows x64)
81
- cargo install puffinparse-cli # CLI + gateway; or grab an archive from GitHub Releases
82
- docker pull ghcr.io/ajinkyashejul/puffinparse # gateway image
79
+ pip install puffinparse # Python SDK (abi3 wheels: Linux glibc 2.28+, macOS, Windows)
80
+ docker pull --platform linux/amd64 ghcr.io/ajinkyashejul/puffinparse # gateway image (amd64 only for now)
83
81
  ```
84
82
 
83
+ The CLI (which also runs the gateway) is a single binary: download the archive for your platform
84
+ from [GitHub Releases](https://github.com/ajinkyashejul/puffinparse/releases/latest), or build it
85
+ with `cargo install --git https://github.com/ajinkyashejul/puffinparse puffinparse-cli`. On macOS, a
86
+ binary downloaded with a browser needs `xattr -d com.apple.quarantine ./puffinparse` once (it is not
87
+ notarized yet). The Node.js SDK is not on npm yet: build it from [`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md).
88
+
85
89
  Set the keys for the providers you use:
86
90
 
87
91
  ```bash
@@ -94,10 +98,11 @@ No key yet? The self-hosted engines work out of the box once installed:
94
98
 
95
99
  ```bash
96
100
  sudo apt-get install tesseract-ocr poppler-utils # or: brew install tesseract poppler
97
- puffinparse ocr scan.png -m tesseract # free, local, word boxes + confidences
101
+ python -c 'import puffinparse; print(puffinparse.ocr("scan.png", model="tesseract").text)'
102
+ puffinparse ocr scan.png -m tesseract # same with the CLI binary: word boxes + confidences
98
103
  ```
99
104
 
100
- Every provider reads its own variable: [`.env.example`](.env.example) lists all of them, the
105
+ Every provider reads its own variable: [`.env.example`](https://github.com/ajinkyashejul/puffinparse/blob/main/.env.example) lists all of them, the
101
106
  [model tables](#model-names) say which belongs to which provider, and `puffinparse providers` shows
102
107
  which ones are set in your shell.
103
108
 
@@ -167,7 +172,8 @@ text.text # whole document, pages joined by a blank line
167
172
  page = text.pages[0]
168
173
  page.text # plain text in reading order
169
174
  for line in page.lines: # Line(text, bbox, confidence)
170
- x0, y0, x1, y1 = line.bbox.to_pixels(page.width, page.height)
175
+ if line.bbox and page.width and page.height: # normalised 0-1 box -> pixels
176
+ x0, y0, x1, y1 = line.bbox.to_pixels(page.width, page.height)
171
177
  for word in page.words: # Word(text, bbox, confidence)
172
178
  ...
173
179
  ```
@@ -180,7 +186,7 @@ schema = {
180
186
  "properties": {"invoice_number": {"type": "string"}, "total": {"type": "number"}},
181
187
  "required": ["invoice_number", "total"],
182
188
  }
183
- result = puffinparse.extract("invoice.pdf", schema, model="...", citations=True)
189
+ result = puffinparse.extract("invoice.pdf", schema, model="reducto/extract", citations=True)
184
190
 
185
191
  result.data # {"invoice_number": "INV-42", "total": 1280.5}
186
192
  result.fields["/total"].confidence # per-field confidence, keyed by JSON pointer
@@ -212,7 +218,7 @@ result = puffinparse.handle_webhook(request.json(), model="reducto") # verify t
212
218
 
213
219
  Reducto, Extend and LlamaParse support jobs; `webhook_url` maps to each provider's per-job webhook
214
220
  where one exists (Extend only has workspace-level webhooks, so it is rejected there). See
215
- [`docs/SPEC.md`](docs/SPEC.md) §15.
221
+ [`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md) §15.
216
222
 
217
223
  ### TypeScript / Node.js
218
224
 
@@ -225,8 +231,9 @@ const doc = await parse("invoice.pdf", { model: "reducto/standard", fallbacks: [
225
231
  console.log(doc.markdown, doc.usage.pages, doc.costUsd);
226
232
  ```
227
233
 
228
- `npm install puffinparse` ships prebuilt binaries for Linux x64/arm64 (glibc), macOS and Windows x64;
229
- other platforms build from source (`cd js && npm ci && npm run build`), see [`js/README.md`](js/README.md).
234
+ The npm package is not published yet; build it from a clone (`cd js && npm ci && npm run build`), see
235
+ [`js/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md). Prebuilt binaries for Linux x64/arm64 (glibc), macOS and Windows x64
236
+ will ship with the npm release.
230
237
 
231
238
  ### Model names
232
239
 
@@ -253,7 +260,7 @@ Each provider carries one of three verification labels:
253
260
  until it is verified; [issue #10](https://github.com/ajinkyashejul/puffinparse/issues/10) tracks this, and a run with your own key is a welcome
254
261
  contribution.
255
262
 
256
- [`docs/providers/README.md`](docs/providers/README.md) tracks the state and links one reference
263
+ [`docs/providers/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/providers/README.md) tracks the state and links one reference
257
264
  page per provider.
258
265
 
259
266
  **Reducto** · `REDUCTO_API_KEY` · live-verified — Layout parsing plus a schema extractor with citations; the default parse target.
@@ -417,7 +424,7 @@ What is guaranteed is **structural fidelity** — key set and nesting, one chunk
417
424
  content strings, the vendor's own block vocabulary and coordinate units, the billed page count —
418
425
  not byte equality with what the vendor would have returned. Fields PuffinParse does not model
419
426
  (presigned URLs, studio links, billing breakdowns, OCR word layers) are `null` or empty, and a few
420
- block types are lossy. [`docs/COMPAT.md`](docs/COMPAT.md) enumerates all of it, per format;
427
+ block types are lossy. [`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md) enumerates all of it, per format;
421
428
  `examples/switch_provider_keep_format.py` is a runnable version of the above.
422
429
 
423
430
  ### Provider-specific options
@@ -502,8 +509,8 @@ curl -H "Authorization: Bearer $TEAM_KEY" -F file=@invoice.pdf -F model=invoices
502
509
  ordered or round-robin fallback), provider keys as `env:` references, and virtual keys with model
503
510
  allow-lists, monthly USD budgets and per-minute limits. `/v1/models`, `/v1/usage`, `/health` and
504
511
  Prometheus `/metrics` are built in; request logs are JSON lines that never contain document content
505
- or secrets. Reference: [`docs/SERVER.md`](docs/SERVER.md), sample:
506
- [`examples/server/puffinparse.toml`](examples/server/puffinparse.toml).
512
+ or secrets. Reference: [`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md), sample:
513
+ [`examples/server/puffinparse.toml`](https://github.com/ajinkyashejul/puffinparse/blob/main/examples/server/puffinparse.toml).
507
514
 
508
515
  ### Rust
509
516
 
@@ -516,38 +523,39 @@ println!("{} pages, ${:.4}\n{}", doc.usage.pages, doc.cost_usd.unwrap_or(0.0), d
516
523
  let text = ocr(DocumentRequest::from_path("scan.png").model("llamaparse/fast")).await?;
517
524
  println!("{} lines on page 1", text.pages[0].lines.len());
518
525
 
519
- let req = ExtractRequest::new(DocumentRequest::from_path("invoice.pdf").model("..."), schema);
526
+ let schema = serde_json::json!({"type": "object", "properties": {"total": {"type": "number"}}});
527
+ let req = ExtractRequest::new(DocumentRequest::from_path("invoice.pdf").model("reducto/extract"), schema);
520
528
  let data = extract(req.citations(true)).await?.data;
521
529
  ```
522
530
 
523
531
  ## Benchmark
524
532
 
525
- PuffinParse ships an open, reproducible benchmark. Ground truth is exact by construction (the documents are rendered from the same source as the truth files), metrics are deterministic text comparisons, and every run records the dataset hash, model, latency and cost.
533
+ PuffinParse ships an open, reproducible benchmark. Pages from public benchmarks are scored against their own published truth or rules, the synthetic set's truth is exact by construction (its documents are rendered from the same source as the truth files), metrics are deterministic text comparisons with no LLM judge, and every run records the dataset hash, model, latency and cost.
526
534
 
527
535
  ```bash
528
536
  python benchmark/generate_synthetic.py # regenerate the dataset (byte-reproducible)
529
537
  puffinparse bench run --dataset benchmark/datasets/synthetic-v1 \
530
538
  --models reducto/standard extend/parse_performance llamaparse/cost_effective
531
- puffinparse bench report benchmark/results/*.json > benchmark/LEADERBOARD.md
539
+ puffinparse bench report benchmark/results/2026-09-25-combined-v3.json # one dataset's leaderboard section
532
540
  puffinparse bench score prediction.md truth.md # metrics for one pair, no network
533
541
  ```
534
542
 
535
543
  Metrics (after NFKC + markdown stripping + whitespace collapsing, case-insensitive by default):
536
544
 
537
- - **Overall** = `100 × mean(char_similarity)`, where `char_similarity = 1 − levenshtein / max(len)`
545
+ - **Overall** = `100 ×` the mean of each document's headline metric: `char_similarity` (`1 − levenshtein / max(len)`) for transcripts, the table score for table-only pages, the rule pass rate for rule-checked pages; a failed call scores 0 (details in [`benchmark/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/README.md))
538
546
  - **CER**, **WER**, **word F1** (bag-of-words precision / recall)
539
547
  - **Order**: Kendall-τ-style agreement of shared line order (reading order)
540
548
  - **Table**: character similarity restricted to markdown table rows
541
549
  - **Latency** p50 / p95 and ms per page; **$/1k pages** from the price table
542
550
 
543
- The current leaderboard is in [`benchmark/LEADERBOARD.md`](benchmark/LEADERBOARD.md), and every
551
+ The current leaderboard is in [`benchmark/LEADERBOARD.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/LEADERBOARD.md), and every
544
552
  document, output, diff and rule check is browsable at
545
553
  [puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/). Datasets:
546
554
  `synthetic-v1` (exact truth by construction) and the headline `combined-v3`, which adds subsets of
547
555
  [ParseBench](https://github.com/run-llama/ParseBench), [olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench),
548
556
  [OmniDocBench](https://github.com/opendatalab/OmniDocBench) and
549
557
  [DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) converted by
550
- [`benchmark/adapters/`](docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
558
+ [`benchmark/adapters/`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
551
559
  `combined-v2` runs are kept for comparison. Long runs are safe:
552
560
  `--dry-run` and `--max-cost` show and cap the spend before any call, and `--resume` continues an
553
561
  interrupted run.
@@ -562,7 +570,7 @@ interrupted run.
562
570
  | Boxes | already normalised | divided by `metadata.page.width/height` | `bBox` divided by page `width/height` |
563
571
  | Usage | `usage.num_pages`, `usage.credits` | `metrics.pageCount`, `usage.credits` | `job_metadata.job_pages` |
564
572
 
565
- Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](docs/SPEC.md).
573
+ Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md).
566
574
 
567
575
  ## Project layout
568
576
 
@@ -583,12 +591,12 @@ docs/SPEC.md specification
583
591
  Planned work is tracked in [GitHub Issues](https://github.com/ajinkyashejul/puffinparse/issues).
584
592
  Open items:
585
593
 
586
- - **Packaging**: publish 0.1.0 to PyPI, npm and crates.io with prebuilt wheels, addons and CLI
587
- binaries ([#9](https://github.com/ajinkyashejul/puffinparse/issues/9)).
594
+ - **Packaging**: PyPI wheels and CLI binaries ship today; npm and crates.io are next, plus a
595
+ multi-arch gateway image and a notarized macOS binary ([#9](https://github.com/ajinkyashejul/puffinparse/issues/9)).
588
596
  - **Providers**: live-verify the docs-only providers ([#10](https://github.com/ajinkyashejul/puffinparse/issues/10)), including PaddleOCR
589
597
  against a real server ([#17](https://github.com/ajinkyashejul/puffinparse/issues/17)).
590
- - **Benchmark**: add `tesseract/default` ([#11](https://github.com/ajinkyashejul/puffinparse/issues/11)) and `docling/default` with a hardware note
591
- ([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3`; regenerate the leaderboard in one command
598
+ - **Benchmark**: add `docling/default` with a hardware note
599
+ ([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3` (`tesseract/default` is already in it); regenerate the leaderboard in one command
592
600
  ([#13](https://github.com/ajinkyashejul/puffinparse/issues/13)); score table-cell neighbour relations exactly ([#15](https://github.com/ajinkyashejul/puffinparse/issues/15)); a READoc
593
601
  long-document track ([#16](https://github.com/ajinkyashejul/puffinparse/issues/16)).
594
602
  - **SDK**: `output_format="mistral"` for code written against Mistral OCR responses
@@ -604,9 +612,45 @@ developers find it. Reports of a wrong score, a missing provider or a confusing
604
612
 
605
613
  ## Contributing
606
614
 
607
- See [CONTRIBUTING.md](CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](.github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](AGENTS.md) summarises the build commands and conventions for coding agents.
615
+ See [CONTRIBUTING.md](https://github.com/ajinkyashejul/puffinparse/blob/main/CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](https://github.com/ajinkyashejul/puffinparse/blob/main/github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](https://github.com/ajinkyashejul/puffinparse/blob/main/AGENTS.md) summarises the build commands and conventions for coding agents.
616
+
617
+ ## Acknowledgements
618
+
619
+ PuffinParse stands on other people's work:
620
+
621
+ - **Benchmarks and datasets.** The combined benchmark is built on
622
+ [ParseBench](https://github.com/run-llama/ParseBench) (LlamaIndex; Zhang et al., 2026,
623
+ arXiv:2604.08538; Apache-2.0),
624
+ [olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench) (Allen Institute for AI;
625
+ Poznanski et al., 2025, arXiv:2502.18443; ODC-BY-1.0),
626
+ [OmniDocBench](https://github.com/opendatalab/OmniDocBench) (OpenDataLab / Shanghai AI
627
+ Laboratory; Ouyang et al., 2024, arXiv:2412.07626; research-only, so only an index is committed)
628
+ and [DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) (Upstage AI; MIT). Each
629
+ `benchmark/datasets/<name>/README.md` gives the pinned revision, the licence, what we changed and
630
+ the citation the authors ask for. If you use these numbers, please cite those benchmarks too.
631
+ - **Metrics.** The rule checks re-implement the test semantics of olmOCR-bench and ParseBench, and
632
+ table structure is scored with TEDS (Zhong, ShafieiBavani and Jimeno Yepes, 2020, "Image-based
633
+ table recognition: data, model, and evaluation"). The scorer is our own Rust code; the source
634
+ comments say whose behaviour each part mirrors.
635
+ - **API design.** The `"<provider>/<model>"` model strings and the gateway server follow
636
+ [LiteLLM](https://github.com/BerriAI/litellm), which did this first for LLM APIs.
637
+ - **Providers and engines.** The compatibility shapes mirror the public response formats of
638
+ Reducto, Extend and LlamaParse so existing code can switch, and the open baselines are
639
+ [Tesseract](https://github.com/tesseract-ocr/tesseract) and
640
+ [Docling](https://github.com/docling-project/docling). Provider names are trademarks of their
641
+ owners. PuffinParse is not affiliated with or endorsed by any of them.
642
+ - **Libraries.** [tokio](https://tokio.rs), [reqwest](https://github.com/seanmonstar/reqwest),
643
+ [serde](https://serde.rs), [axum](https://github.com/tokio-rs/axum),
644
+ [clap](https://github.com/clap-rs/clap), [PyO3](https://pyo3.rs),
645
+ [maturin](https://github.com/PyO3/maturin), [napi-rs](https://napi.rs) and the other crates in
646
+ [THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md).
647
+ The site uses GitHub's [Octicons](https://github.com/primer/octicons) GitHub mark (MIT), and the
648
+ results viewer uses [pdf.js](https://github.com/mozilla/pdf.js) (Apache-2.0).
608
649
 
609
650
  ## License
610
651
 
611
- MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE).
652
+ MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE). The third-party
653
+ crates compiled into the binaries are listed in
654
+ [THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md),
655
+ and the benchmark data keeps its own licence, stated per dataset.
612
656
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  # PuffinParse
4
4
 
5
- [![PyPI](https://img.shields.io/pypi/v/puffinparse?color=E95C20)](https://pypi.org/project/puffinparse/) [![GitHub stars](https://img.shields.io/github/stars/ajinkyashejul/puffinparse?style=flat&color=E95C20)](https://github.com/ajinkyashejul/puffinparse/stargazers) [![CI](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml/badge.svg)](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [![Docs](https://img.shields.io/badge/docs-puffinparse.com-E95C20)](https://puffinparse.com/docs/) [![Benchmark](https://img.shields.io/badge/benchmark-results-E95C20)](https://puffinparse.com/benchmark-results/) [![Python 3.9+](https://img.shields.io/badge/python-3.9%2B-blue)](pyproject.toml)
5
+ [![PyPI](https://img.shields.io/pypi/v/puffinparse?color=E95C20)](https://pypi.org/project/puffinparse/) [![GitHub stars](https://img.shields.io/github/stars/ajinkyashejul/puffinparse?style=flat&color=E95C20)](https://github.com/ajinkyashejul/puffinparse/stargazers) [![CI](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml/badge.svg)](https://github.com/ajinkyashejul/puffinparse/actions/workflows/ci.yml) [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE) [![Docs](https://img.shields.io/badge/docs-puffinparse.com-E95C20)](https://puffinparse.com/docs/) [![Benchmark](https://img.shields.io/badge/benchmark-results-E95C20)](https://puffinparse.com/benchmark-results/) [![Python 3.9+](https://img.shields.io/badge/python-3.9%2B-blue)](https://github.com/ajinkyashejul/puffinparse/blob/main/pyproject.toml)
6
6
 
7
7
  **One API for every document parser: parse, OCR and extract.** Rust core, Python and TypeScript SDKs, a CLI, a self-hosted gateway, and an open benchmark that ranks providers on accuracy, latency and cost.
8
8
 
@@ -30,24 +30,28 @@ Two things make switching real rather than aspirational. **Modes**: every call n
30
30
  | **Providers (v0.1)** | 18 providers · 60 models · 3 modes. **Live-verified** against the real APIs: [Reducto](https://reducto.ai), [Extend](https://extend.ai), [LlamaParse](https://cloud.llamaindex.ai). **Verified locally**: the self-hosted Tesseract and Docling. **Docs-only** (implemented from the provider's API documentation and tested against fixture payloads, not yet run live): Mistral, Azure, Textract, Gemini, OpenAI, Anthropic, Mathpix, Datalab, Unstructured, Upstage, Landing AI, Google Document AI and the self-hosted PaddleOCR ([help verify them](https://github.com/ajinkyashejul/puffinparse/issues/10)). [Full table](#model-names) |
31
31
  | **Modes** | `parse` (markdown + blocks), `ocr` (plain text + boxes), `extract` (JSON from a schema) |
32
32
  | **Core** | Rust (`puffinparse-core`): `reqwest` + `tokio`, no vendor SDKs, `#![forbid(unsafe_code)]` |
33
- | **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](js/README.md)), both on the same Rust core |
34
- | **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](docs/SERVER.md)) |
33
+ | **SDKs** | Python 3.9+ (sync + async, fully typed) and Node.js / TypeScript ([`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md)), both on the same Rust core |
34
+ | **Gateway** | `puffinparse serve`: one HTTP endpoint with aliases, fallbacks, virtual keys, budgets, rate limits, JSON logs and Prometheus metrics ([`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md)) |
35
35
  | **Long documents** | `submit` / `retrieve` jobs and provider webhooks instead of a blocking call ([below](#long-documents-jobs-and-webhooks)) |
36
36
  | **CLI** | `puffinparse parse`, `puffinparse ocr`, `puffinparse extract`, `puffinparse providers`, `puffinparse bench` |
37
37
  | **Reliability** | Retries with jittered backoff, whole-call deadlines, `Router` with ordered fallbacks / round-robin |
38
- | **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](docs/COMPAT.md)) |
38
+ | **Compatibility** | `output_format` renders any provider's result in Reducto's, Extend's or LlamaParse's own JSON, so an existing integration keeps its parser ([`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md)) |
39
39
  | **Cost** | Embedded, overridable price table → `cost_usd` on every response |
40
40
  | **Benchmark** | One harness over synthetic data and public benchmarks (ParseBench, olmOCR-bench, OmniDocBench, DP-Bench); deterministic metrics and rule checks, latency, $/1k pages; every output inspectable at [puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/) |
41
41
 
42
42
  ## Install
43
43
 
44
44
  ```bash
45
- pip install puffinparse # Python SDK (abi3 wheels: Linux, macOS, Windows)
46
- npm install puffinparse # Node.js SDK (prebuilt for Linux x64/arm64 glibc, macOS, Windows x64)
47
- cargo install puffinparse-cli # CLI + gateway; or grab an archive from GitHub Releases
48
- docker pull ghcr.io/ajinkyashejul/puffinparse # gateway image
45
+ pip install puffinparse # Python SDK (abi3 wheels: Linux glibc 2.28+, macOS, Windows)
46
+ docker pull --platform linux/amd64 ghcr.io/ajinkyashejul/puffinparse # gateway image (amd64 only for now)
49
47
  ```
50
48
 
49
+ The CLI (which also runs the gateway) is a single binary: download the archive for your platform
50
+ from [GitHub Releases](https://github.com/ajinkyashejul/puffinparse/releases/latest), or build it
51
+ with `cargo install --git https://github.com/ajinkyashejul/puffinparse puffinparse-cli`. On macOS, a
52
+ binary downloaded with a browser needs `xattr -d com.apple.quarantine ./puffinparse` once (it is not
53
+ notarized yet). The Node.js SDK is not on npm yet: build it from [`js/`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md).
54
+
51
55
  Set the keys for the providers you use:
52
56
 
53
57
  ```bash
@@ -60,10 +64,11 @@ No key yet? The self-hosted engines work out of the box once installed:
60
64
 
61
65
  ```bash
62
66
  sudo apt-get install tesseract-ocr poppler-utils # or: brew install tesseract poppler
63
- puffinparse ocr scan.png -m tesseract # free, local, word boxes + confidences
67
+ python -c 'import puffinparse; print(puffinparse.ocr("scan.png", model="tesseract").text)'
68
+ puffinparse ocr scan.png -m tesseract # same with the CLI binary: word boxes + confidences
64
69
  ```
65
70
 
66
- Every provider reads its own variable: [`.env.example`](.env.example) lists all of them, the
71
+ Every provider reads its own variable: [`.env.example`](https://github.com/ajinkyashejul/puffinparse/blob/main/.env.example) lists all of them, the
67
72
  [model tables](#model-names) say which belongs to which provider, and `puffinparse providers` shows
68
73
  which ones are set in your shell.
69
74
 
@@ -133,7 +138,8 @@ text.text # whole document, pages joined by a blank line
133
138
  page = text.pages[0]
134
139
  page.text # plain text in reading order
135
140
  for line in page.lines: # Line(text, bbox, confidence)
136
- x0, y0, x1, y1 = line.bbox.to_pixels(page.width, page.height)
141
+ if line.bbox and page.width and page.height: # normalised 0-1 box -> pixels
142
+ x0, y0, x1, y1 = line.bbox.to_pixels(page.width, page.height)
137
143
  for word in page.words: # Word(text, bbox, confidence)
138
144
  ...
139
145
  ```
@@ -146,7 +152,7 @@ schema = {
146
152
  "properties": {"invoice_number": {"type": "string"}, "total": {"type": "number"}},
147
153
  "required": ["invoice_number", "total"],
148
154
  }
149
- result = puffinparse.extract("invoice.pdf", schema, model="...", citations=True)
155
+ result = puffinparse.extract("invoice.pdf", schema, model="reducto/extract", citations=True)
150
156
 
151
157
  result.data # {"invoice_number": "INV-42", "total": 1280.5}
152
158
  result.fields["/total"].confidence # per-field confidence, keyed by JSON pointer
@@ -178,7 +184,7 @@ result = puffinparse.handle_webhook(request.json(), model="reducto") # verify t
178
184
 
179
185
  Reducto, Extend and LlamaParse support jobs; `webhook_url` maps to each provider's per-job webhook
180
186
  where one exists (Extend only has workspace-level webhooks, so it is rejected there). See
181
- [`docs/SPEC.md`](docs/SPEC.md) §15.
187
+ [`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md) §15.
182
188
 
183
189
  ### TypeScript / Node.js
184
190
 
@@ -191,8 +197,9 @@ const doc = await parse("invoice.pdf", { model: "reducto/standard", fallbacks: [
191
197
  console.log(doc.markdown, doc.usage.pages, doc.costUsd);
192
198
  ```
193
199
 
194
- `npm install puffinparse` ships prebuilt binaries for Linux x64/arm64 (glibc), macOS and Windows x64;
195
- other platforms build from source (`cd js && npm ci && npm run build`), see [`js/README.md`](js/README.md).
200
+ The npm package is not published yet; build it from a clone (`cd js && npm ci && npm run build`), see
201
+ [`js/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/js/README.md). Prebuilt binaries for Linux x64/arm64 (glibc), macOS and Windows x64
202
+ will ship with the npm release.
196
203
 
197
204
  ### Model names
198
205
 
@@ -219,7 +226,7 @@ Each provider carries one of three verification labels:
219
226
  until it is verified; [issue #10](https://github.com/ajinkyashejul/puffinparse/issues/10) tracks this, and a run with your own key is a welcome
220
227
  contribution.
221
228
 
222
- [`docs/providers/README.md`](docs/providers/README.md) tracks the state and links one reference
229
+ [`docs/providers/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/providers/README.md) tracks the state and links one reference
223
230
  page per provider.
224
231
 
225
232
  **Reducto** · `REDUCTO_API_KEY` · live-verified — Layout parsing plus a schema extractor with citations; the default parse target.
@@ -383,7 +390,7 @@ What is guaranteed is **structural fidelity** — key set and nesting, one chunk
383
390
  content strings, the vendor's own block vocabulary and coordinate units, the billed page count —
384
391
  not byte equality with what the vendor would have returned. Fields PuffinParse does not model
385
392
  (presigned URLs, studio links, billing breakdowns, OCR word layers) are `null` or empty, and a few
386
- block types are lossy. [`docs/COMPAT.md`](docs/COMPAT.md) enumerates all of it, per format;
393
+ block types are lossy. [`docs/COMPAT.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/COMPAT.md) enumerates all of it, per format;
387
394
  `examples/switch_provider_keep_format.py` is a runnable version of the above.
388
395
 
389
396
  ### Provider-specific options
@@ -468,8 +475,8 @@ curl -H "Authorization: Bearer $TEAM_KEY" -F file=@invoice.pdf -F model=invoices
468
475
  ordered or round-robin fallback), provider keys as `env:` references, and virtual keys with model
469
476
  allow-lists, monthly USD budgets and per-minute limits. `/v1/models`, `/v1/usage`, `/health` and
470
477
  Prometheus `/metrics` are built in; request logs are JSON lines that never contain document content
471
- or secrets. Reference: [`docs/SERVER.md`](docs/SERVER.md), sample:
472
- [`examples/server/puffinparse.toml`](examples/server/puffinparse.toml).
478
+ or secrets. Reference: [`docs/SERVER.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SERVER.md), sample:
479
+ [`examples/server/puffinparse.toml`](https://github.com/ajinkyashejul/puffinparse/blob/main/examples/server/puffinparse.toml).
473
480
 
474
481
  ### Rust
475
482
 
@@ -482,38 +489,39 @@ println!("{} pages, ${:.4}\n{}", doc.usage.pages, doc.cost_usd.unwrap_or(0.0), d
482
489
  let text = ocr(DocumentRequest::from_path("scan.png").model("llamaparse/fast")).await?;
483
490
  println!("{} lines on page 1", text.pages[0].lines.len());
484
491
 
485
- let req = ExtractRequest::new(DocumentRequest::from_path("invoice.pdf").model("..."), schema);
492
+ let schema = serde_json::json!({"type": "object", "properties": {"total": {"type": "number"}}});
493
+ let req = ExtractRequest::new(DocumentRequest::from_path("invoice.pdf").model("reducto/extract"), schema);
486
494
  let data = extract(req.citations(true)).await?.data;
487
495
  ```
488
496
 
489
497
  ## Benchmark
490
498
 
491
- PuffinParse ships an open, reproducible benchmark. Ground truth is exact by construction (the documents are rendered from the same source as the truth files), metrics are deterministic text comparisons, and every run records the dataset hash, model, latency and cost.
499
+ PuffinParse ships an open, reproducible benchmark. Pages from public benchmarks are scored against their own published truth or rules, the synthetic set's truth is exact by construction (its documents are rendered from the same source as the truth files), metrics are deterministic text comparisons with no LLM judge, and every run records the dataset hash, model, latency and cost.
492
500
 
493
501
  ```bash
494
502
  python benchmark/generate_synthetic.py # regenerate the dataset (byte-reproducible)
495
503
  puffinparse bench run --dataset benchmark/datasets/synthetic-v1 \
496
504
  --models reducto/standard extend/parse_performance llamaparse/cost_effective
497
- puffinparse bench report benchmark/results/*.json > benchmark/LEADERBOARD.md
505
+ puffinparse bench report benchmark/results/2026-09-25-combined-v3.json # one dataset's leaderboard section
498
506
  puffinparse bench score prediction.md truth.md # metrics for one pair, no network
499
507
  ```
500
508
 
501
509
  Metrics (after NFKC + markdown stripping + whitespace collapsing, case-insensitive by default):
502
510
 
503
- - **Overall** = `100 × mean(char_similarity)`, where `char_similarity = 1 − levenshtein / max(len)`
511
+ - **Overall** = `100 ×` the mean of each document's headline metric: `char_similarity` (`1 − levenshtein / max(len)`) for transcripts, the table score for table-only pages, the rule pass rate for rule-checked pages; a failed call scores 0 (details in [`benchmark/README.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/README.md))
504
512
  - **CER**, **WER**, **word F1** (bag-of-words precision / recall)
505
513
  - **Order**: Kendall-τ-style agreement of shared line order (reading order)
506
514
  - **Table**: character similarity restricted to markdown table rows
507
515
  - **Latency** p50 / p95 and ms per page; **$/1k pages** from the price table
508
516
 
509
- The current leaderboard is in [`benchmark/LEADERBOARD.md`](benchmark/LEADERBOARD.md), and every
517
+ The current leaderboard is in [`benchmark/LEADERBOARD.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/benchmark/LEADERBOARD.md), and every
510
518
  document, output, diff and rule check is browsable at
511
519
  [puffinparse.com/benchmark-results](https://puffinparse.com/benchmark-results/). Datasets:
512
520
  `synthetic-v1` (exact truth by construction) and the headline `combined-v3`, which adds subsets of
513
521
  [ParseBench](https://github.com/run-llama/ParseBench), [olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench),
514
522
  [OmniDocBench](https://github.com/opendatalab/OmniDocBench) and
515
523
  [DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) converted by
516
- [`benchmark/adapters/`](docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
524
+ [`benchmark/adapters/`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/benchmarks/adapters.md) at pinned revisions. Older `combined-v1` and
517
525
  `combined-v2` runs are kept for comparison. Long runs are safe:
518
526
  `--dry-run` and `--max-cost` show and cap the spend before any call, and `--resume` continues an
519
527
  interrupted run.
@@ -528,7 +536,7 @@ interrupted run.
528
536
  | Boxes | already normalised | divided by `metadata.page.width/height` | `bBox` divided by page `width/height` |
529
537
  | Usage | `usage.num_pages`, `usage.credits` | `metrics.pageCount`, `usage.credits` | `job_metadata.job_pages` |
530
538
 
531
- Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](docs/SPEC.md).
539
+ Full details, including the exact wire formats verified against live responses, are in [`docs/SPEC.md`](https://github.com/ajinkyashejul/puffinparse/blob/main/docs/SPEC.md).
532
540
 
533
541
  ## Project layout
534
542
 
@@ -549,12 +557,12 @@ docs/SPEC.md specification
549
557
  Planned work is tracked in [GitHub Issues](https://github.com/ajinkyashejul/puffinparse/issues).
550
558
  Open items:
551
559
 
552
- - **Packaging**: publish 0.1.0 to PyPI, npm and crates.io with prebuilt wheels, addons and CLI
553
- binaries ([#9](https://github.com/ajinkyashejul/puffinparse/issues/9)).
560
+ - **Packaging**: PyPI wheels and CLI binaries ship today; npm and crates.io are next, plus a
561
+ multi-arch gateway image and a notarized macOS binary ([#9](https://github.com/ajinkyashejul/puffinparse/issues/9)).
554
562
  - **Providers**: live-verify the docs-only providers ([#10](https://github.com/ajinkyashejul/puffinparse/issues/10)), including PaddleOCR
555
563
  against a real server ([#17](https://github.com/ajinkyashejul/puffinparse/issues/17)).
556
- - **Benchmark**: add `tesseract/default` ([#11](https://github.com/ajinkyashejul/puffinparse/issues/11)) and `docling/default` with a hardware note
557
- ([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3`; regenerate the leaderboard in one command
564
+ - **Benchmark**: add `docling/default` with a hardware note
565
+ ([#12](https://github.com/ajinkyashejul/puffinparse/issues/12)) to `combined-v3` (`tesseract/default` is already in it); regenerate the leaderboard in one command
558
566
  ([#13](https://github.com/ajinkyashejul/puffinparse/issues/13)); score table-cell neighbour relations exactly ([#15](https://github.com/ajinkyashejul/puffinparse/issues/15)); a READoc
559
567
  long-document track ([#16](https://github.com/ajinkyashejul/puffinparse/issues/16)).
560
568
  - **SDK**: `output_format="mistral"` for code written against Mistral OCR responses
@@ -570,8 +578,44 @@ developers find it. Reports of a wrong score, a missing provider or a confusing
570
578
 
571
579
  ## Contributing
572
580
 
573
- See [CONTRIBUTING.md](CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](.github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](AGENTS.md) summarises the build commands and conventions for coding agents.
581
+ See [CONTRIBUTING.md](https://github.com/ajinkyashejul/puffinparse/blob/main/CONTRIBUTING.md). Adding a provider is one Rust file plus a fixture test; the checklist is in the [new provider issue template](https://github.com/ajinkyashejul/puffinparse/blob/main/github/ISSUE_TEMPLATE/new_provider.md). Issues labelled [`good first issue`](https://github.com/ajinkyashejul/puffinparse/labels/good%20first%20issue) are a good place to start, and [AGENTS.md](https://github.com/ajinkyashejul/puffinparse/blob/main/AGENTS.md) summarises the build commands and conventions for coding agents.
582
+
583
+ ## Acknowledgements
584
+
585
+ PuffinParse stands on other people's work:
586
+
587
+ - **Benchmarks and datasets.** The combined benchmark is built on
588
+ [ParseBench](https://github.com/run-llama/ParseBench) (LlamaIndex; Zhang et al., 2026,
589
+ arXiv:2604.08538; Apache-2.0),
590
+ [olmOCR-bench](https://huggingface.co/datasets/allenai/olmOCR-bench) (Allen Institute for AI;
591
+ Poznanski et al., 2025, arXiv:2502.18443; ODC-BY-1.0),
592
+ [OmniDocBench](https://github.com/opendatalab/OmniDocBench) (OpenDataLab / Shanghai AI
593
+ Laboratory; Ouyang et al., 2024, arXiv:2412.07626; research-only, so only an index is committed)
594
+ and [DP-Bench](https://huggingface.co/datasets/upstage/dp-bench) (Upstage AI; MIT). Each
595
+ `benchmark/datasets/<name>/README.md` gives the pinned revision, the licence, what we changed and
596
+ the citation the authors ask for. If you use these numbers, please cite those benchmarks too.
597
+ - **Metrics.** The rule checks re-implement the test semantics of olmOCR-bench and ParseBench, and
598
+ table structure is scored with TEDS (Zhong, ShafieiBavani and Jimeno Yepes, 2020, "Image-based
599
+ table recognition: data, model, and evaluation"). The scorer is our own Rust code; the source
600
+ comments say whose behaviour each part mirrors.
601
+ - **API design.** The `"<provider>/<model>"` model strings and the gateway server follow
602
+ [LiteLLM](https://github.com/BerriAI/litellm), which did this first for LLM APIs.
603
+ - **Providers and engines.** The compatibility shapes mirror the public response formats of
604
+ Reducto, Extend and LlamaParse so existing code can switch, and the open baselines are
605
+ [Tesseract](https://github.com/tesseract-ocr/tesseract) and
606
+ [Docling](https://github.com/docling-project/docling). Provider names are trademarks of their
607
+ owners. PuffinParse is not affiliated with or endorsed by any of them.
608
+ - **Libraries.** [tokio](https://tokio.rs), [reqwest](https://github.com/seanmonstar/reqwest),
609
+ [serde](https://serde.rs), [axum](https://github.com/tokio-rs/axum),
610
+ [clap](https://github.com/clap-rs/clap), [PyO3](https://pyo3.rs),
611
+ [maturin](https://github.com/PyO3/maturin), [napi-rs](https://napi.rs) and the other crates in
612
+ [THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md).
613
+ The site uses GitHub's [Octicons](https://github.com/primer/octicons) GitHub mark (MIT), and the
614
+ results viewer uses [pdf.js](https://github.com/mozilla/pdf.js) (Apache-2.0).
574
615
 
575
616
  ## License
576
617
 
577
- MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE).
618
+ MIT. See [LICENSE](https://github.com/ajinkyashejul/puffinparse/blob/main/LICENSE). The third-party
619
+ crates compiled into the binaries are listed in
620
+ [THIRD_PARTY_NOTICES.md](https://github.com/ajinkyashejul/puffinparse/blob/main/THIRD_PARTY_NOTICES.md),
621
+ and the benchmark data keeps its own licence, stated per dataset.
@@ -0,0 +1,18 @@
1
+ # puffinparse-core
2
+
3
+ Rust core of [PuffinParse](https://github.com/ajinkyashejul/puffinparse): one API for every
4
+ OCR / document-parsing provider (Reducto, Extend, LlamaParse, Mistral, Azure, Textract, …).
5
+
6
+ ```rust
7
+ use puffinparse_core::{parse, DocumentRequest};
8
+
9
+ #[tokio::main]
10
+ async fn main() -> Result<(), puffinparse_core::Error> {
11
+ let doc = parse(DocumentRequest::from_path("invoice.pdf").model("reducto/standard")).await?;
12
+ println!("{} pages, ${:.4}: {}", doc.usage.pages, doc.cost_usd.unwrap_or(0.0), doc.markdown);
13
+ Ok(())
14
+ }
15
+ ```
16
+
17
+ `ocr` returns plain text with line and word boxes, and `extract` fills a JSON Schema; see the
18
+ repository README for the Python and TypeScript SDKs, the CLI, the gateway and the benchmark.
@@ -390,6 +390,11 @@ pub fn decode_entities(s: &str) -> String {
390
390
  // grids — the same formula, costs and Zhang–Shasha edit distance, just without `thead`/`tbody` and
391
391
  // span attributes. It still punishes missing, extra, split or merged rows and cells, which the
392
392
  // flat `table_score` string similarity does not see.
393
+ //
394
+ // Credit: the metric is from the paper above; its reference implementation is IBM's PubTabNet
395
+ // `src/metric.py` (github.com/ibm-aur-nlp/PubTabNet, Apache-2.0), which uses APTED. This file is an
396
+ // independent Rust implementation (Zhang & Shasha, 1989, "Simple fast algorithms for the editing
397
+ // distance between trees and related problems"); no PubTabNet code is copied.
393
398
 
394
399
  /// Node budget above which [`teds_grid`] declines (`None`): Zhang–Shasha is O(n·m) memory.
395
400
  const TEDS_MAX_PAIRS: usize = 16_000_000;
@@ -444,6 +444,13 @@ pub fn summarize_with(metrics: &[Option<Metrics>], table_only: &[bool]) -> Summa
444
444
  // of a reference transcript: `kind: "rules"` documents in `docs/benchmarks/adapters.md`. A document
445
445
  // scores `passed / total`, which has the same shape as an accuracy in `0..=1` and therefore slots
446
446
  // into [`Metrics`] via [`metrics_from_rules`].
447
+ //
448
+ // Credit: the rule types and their semantics come from the benchmarks that define them and are
449
+ // re-implemented here from their published descriptions and code, without copying any of it:
450
+ // olmOCR-bench (`olmocr/bench/tests.py` in github.com/allenai/olmocr, Apache-2.0: `present`,
451
+ // `absent`, `order`, `table`, and `max_diffs` fuzzy matching) and ParseBench
452
+ // (github.com/run-llama/ParseBench, Apache-2.0: `missing_sentence_percent` → `bag_of_sentences`).
453
+ // The fuzzy substring search is Sellers' algorithm (1980), see [`approx_contains`].
447
454
 
448
455
  /// What a [`Rule`] asserts about the parsed markdown.
449
456
  #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
@@ -245,6 +245,7 @@ async fn run(request: &DocumentRequest) -> Result<Run> {
245
245
  &deadline,
246
246
  "pdftoppm",
247
247
  "install poppler-utils (apt install poppler-utils / brew install poppler) or set PDFTOPPM_CMD",
248
+ PDFTOPPM_INPUT_EXITS,
248
249
  )
249
250
  .await?;
250
251
  for (n, path) in rendered_pages(scratch.path()).await? {
@@ -285,6 +286,7 @@ async fn run(request: &DocumentRequest) -> Result<Run> {
285
286
  &deadline,
286
287
  "tesseract",
287
288
  "install Tesseract (apt install tesseract-ocr / brew install tesseract) or set TESSERACT_CMD",
289
+ &[],
288
290
  )
289
291
  .await?;
290
292
  let tsv = String::from_utf8_lossy(&stdout).into_owned();
@@ -319,8 +321,34 @@ async fn rendered_pages(dir: &Path) -> Result<Vec<(u32, PathBuf)>> {
319
321
  Ok(out)
320
322
  }
321
323
 
322
- /// Run a local binary with the call's deadline; non-zero exits carry the tool's stderr verbatim.
323
- async fn exec(cmd: &str, args: &[String], deadline: &Deadline, tool: &str, install_hint: &str) -> Result<Vec<u8>> {
324
+ /// `pdftoppm` exit codes that mean the input is at fault: 1 cannot open/read the PDF (corrupt or not
325
+ /// a PDF), 3 PDF permissions, 99 anything else, which in practice is a page range past the end.
326
+ /// These are input errors, not retryable: another model would fail on the same file.
327
+ const PDFTOPPM_INPUT_EXITS: &[i32] = &[1, 3, 99];
328
+
329
+ /// Lines of a tool's stderr kept in an error message; a corrupt PDF can produce hundreds.
330
+ const STDERR_LINES: usize = 5;
331
+
332
+ /// The first [`STDERR_LINES`] lines of a tool's stderr, noting how many were dropped.
333
+ fn stderr_excerpt(stderr: &str) -> String {
334
+ let lines: Vec<&str> = stderr.trim().lines().collect();
335
+ if lines.len() <= STDERR_LINES {
336
+ return lines.join("\n");
337
+ }
338
+ format!("{}\n... ({} more lines)", lines[..STDERR_LINES].join("\n"), lines.len() - STDERR_LINES)
339
+ }
340
+
341
+ /// Run a local binary with the call's deadline. A non-zero exit carries the start of the tool's
342
+ /// stderr; exit codes in `input_exits` are reported as input errors, and 127 (the shell's "command
343
+ /// not found", as seen in minimal containers) as a missing binary.
344
+ async fn exec(
345
+ cmd: &str,
346
+ args: &[String],
347
+ deadline: &Deadline,
348
+ tool: &str,
349
+ install_hint: &str,
350
+ input_exits: &[i32],
351
+ ) -> Result<Vec<u8>> {
324
352
  let mut command = tokio::process::Command::new(cmd);
325
353
  // Tesseract's OpenMP threads oversubscribe the CPU when several pages run concurrently (a
326
354
  // small PNG took 70 s instead of 0.7 s on 4 cores); one thread per process is the documented
@@ -348,10 +376,16 @@ async fn exec(cmd: &str, args: &[String], deadline: &Deadline, tool: &str, insta
348
376
  Err(_) => return Err(Error::timeout(format!("deadline exceeded while running {tool}")).with_provider(NAME)),
349
377
  };
350
378
  if !output.status.success() {
351
- let stderr = String::from_utf8_lossy(&output.stderr);
352
- return Err(
353
- Error::provider(format!("{tool} exited with {}: {}", output.status, stderr.trim())).with_provider(NAME)
354
- );
379
+ let stderr = stderr_excerpt(&String::from_utf8_lossy(&output.stderr));
380
+ let code = output.status.code();
381
+ let err = match code {
382
+ Some(127) => Error::provider(format!("{tool} binary '{cmd}' could not be run (exit 127): {install_hint}")),
383
+ Some(c) if input_exits.contains(&c) => {
384
+ Error::input(format!("{tool} could not read the input (exit {c}): {stderr}"))
385
+ }
386
+ _ => Error::provider(format!("{tool} exited with {}: {stderr}", output.status)),
387
+ };
388
+ return Err(err.with_provider(NAME));
355
389
  }
356
390
  Ok(output.stdout)
357
391
  }
@@ -622,6 +656,40 @@ mod tests {
622
656
  assert!(e.message.contains("pdftoppm binary") && e.message.contains("poppler"), "{e}");
623
657
  }
624
658
 
659
+ #[test]
660
+ fn stderr_is_trimmed_to_a_few_lines() {
661
+ assert_eq!(stderr_excerpt(" one\ntwo\n"), "one\ntwo");
662
+ let long: String = (1..=300).map(|i| format!("Syntax Error ({i}): Illegal character\n")).collect();
663
+ let short = stderr_excerpt(&long);
664
+ assert_eq!(short.lines().count(), STDERR_LINES + 1);
665
+ assert!(short.ends_with("... (295 more lines)"), "{short}");
666
+ }
667
+
668
+ #[cfg(unix)]
669
+ #[tokio::test]
670
+ async fn exit_codes_are_classified() {
671
+ let deadline = Deadline::new(10.0);
672
+ let sh = |script: &str| vec!["-c".to_string(), script.to_string()];
673
+ let e = exec(
674
+ "sh",
675
+ &sh("echo 'Wrong page range' >&2; exit 99"),
676
+ &deadline,
677
+ "pdftoppm",
678
+ "hint",
679
+ PDFTOPPM_INPUT_EXITS,
680
+ )
681
+ .await
682
+ .unwrap_err();
683
+ assert_eq!(e.kind, crate::error::ErrorKind::Input);
684
+ assert!(!e.retryable && e.message.contains("Wrong page range"), "{e}");
685
+ let e = exec("sh", &sh("exit 127"), &deadline, "tesseract", "install it", &[]).await.unwrap_err();
686
+ assert_eq!(e.kind, crate::error::ErrorKind::Provider);
687
+ assert!(e.message.contains("could not be run") && e.message.contains("install it"), "{e}");
688
+ let e = exec("sh", &sh("exit 2"), &deadline, "pdftoppm", "hint", PDFTOPPM_INPUT_EXITS).await.unwrap_err();
689
+ assert_eq!(e.kind, crate::error::ErrorKind::Provider);
690
+ assert!(e.retryable, "{e}");
691
+ }
692
+
625
693
  #[tokio::test]
626
694
  async fn rejects_non_image_input() {
627
695
  let req = DocumentRequest::from_bytes(&b"PK\x03\x04"[..], "a.docx");
@@ -0,0 +1,14 @@
1
+ # Test fixtures: where each payload comes from
2
+
3
+ Every provider parse path is tested against a payload in this directory. They come from three
4
+ places, matching the **Status** column in [`docs/providers/README.md`](../../../../docs/providers/README.md):
5
+
6
+ | Fixtures | Provider status | Origin |
7
+ |---|---|---|
8
+ | `reducto_*.json`, `extend_*.json`, `llamaparse_*.json` | live-verified | Trimmed responses captured from our own calls to the live APIs, on documents we made (`benchmark/datasets/synthetic-v1`, the "Hello LiteOCR" / "Cedar Ridge Supply" sample invoices). Job ids, signed URLs, project ids and similar values are redacted or replaced. The provider pages say which file is which. |
9
+ | `docling_*.json`, `tesseract_headings.tsv` | verified locally | Real output of a locally installed docling-serve / Tesseract on `synthetic-v1` documents. |
10
+ | `anthropic_*`, `azure_*`, `datalab_*`, `gemini_*`, `google_documentai_*`, `landingai_*`, `mathpix_*`, `mistral_*`, `openai_*`, `paddleocr_*`, `textract_*`, `unstructured_*`, `upstage_*` | docs-only | Hand-built to the response *schema* each provider documents, filled with our own sample content. They are not copies of the providers' documentation examples. |
11
+ | `rules_text_sparse_note.json` | n/a | A verbatim copy of `benchmark/datasets/parsebench/rules/text_sparse_note.json`, derived from [ParseBench](https://huggingface.co/datasets/llamaindex/ParseBench) (LlamaIndex, Apache-2.0); see that dataset's README for attribution. |
12
+
13
+ When you replace a hand-built fixture with a real response, redact every credential, signed URL,
14
+ account or project id and any personal data before committing, and update this table.
@@ -4,7 +4,7 @@ build-backend = "maturin"
4
4
 
5
5
  [project]
6
6
  name = "puffinparse"
7
- version = "0.1.2"
7
+ version = "0.1.3"
8
8
  description = "One API for every OCR / document-parsing provider (Reducto, Extend, LlamaParse, ...). Rust core, Python SDK, open benchmark."
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -1,17 +0,0 @@
1
- # puffinparse-core
2
-
3
- Rust core of [PuffinParse](https://github.com/ajinkyashejul/puffinparse): a single API
4
- for every OCR / document-parsing provider (Reducto, Extend, LlamaParse, …).
5
-
6
- ```rust
7
- use puffinparse_core::{ocr, OcrRequest};
8
-
9
- #[tokio::main]
10
- async fn main() -> Result<(), puffinparse_core::Error> {
11
- let resp = ocr(OcrRequest::from_path("invoice.pdf").model("reducto/standard")).await?;
12
- println!("{} pages, ${:.4}: {}", resp.usage.pages, resp.cost_usd.unwrap_or(0.0), resp.markdown);
13
- Ok(())
14
- }
15
- ```
16
-
17
- See the repository README for the Python SDK, CLI and benchmark.
File without changes