veriscript 1.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. veriscript-1.5.0/LICENSE +21 -0
  2. veriscript-1.5.0/PKG-INFO +342 -0
  3. veriscript-1.5.0/README.md +270 -0
  4. veriscript-1.5.0/calibration/rapidocr_devanagari_v1.json +60 -0
  5. veriscript-1.5.0/deva_crnn/__init__.py +1 -0
  6. veriscript-1.5.0/deva_crnn/augment.py +242 -0
  7. veriscript-1.5.0/deva_crnn/charset.py +31 -0
  8. veriscript-1.5.0/deva_crnn/data.py +48 -0
  9. veriscript-1.5.0/deva_crnn/gate.py +97 -0
  10. veriscript-1.5.0/deva_crnn/model.py +44 -0
  11. veriscript-1.5.0/deva_crnn/predict.py +127 -0
  12. veriscript-1.5.0/deva_crnn/train.py +254 -0
  13. veriscript-1.5.0/fonts/Mukta-Regular.ttf +0 -0
  14. veriscript-1.5.0/fonts/OFL.txt +93 -0
  15. veriscript-1.5.0/pyproject.toml +84 -0
  16. veriscript-1.5.0/setup.cfg +4 -0
  17. veriscript-1.5.0/tests/test_anchor_gemini.py +151 -0
  18. veriscript-1.5.0/tests/test_bakeoff_models.py +145 -0
  19. veriscript-1.5.0/tests/test_build_deva_wheel.py +44 -0
  20. veriscript-1.5.0/tests/test_cli_document.py +131 -0
  21. veriscript-1.5.0/tests/test_demo_asset.py +38 -0
  22. veriscript-1.5.0/tests/test_deva_corpus_fetch.py +89 -0
  23. veriscript-1.5.0/tests/test_deva_crnn.py +292 -0
  24. veriscript-1.5.0/tests/test_deva_gate.py +38 -0
  25. veriscript-1.5.0/tests/test_deva_lines.py +149 -0
  26. veriscript-1.5.0/tests/test_deva_train_out.py +32 -0
  27. veriscript-1.5.0/tests/test_doc_data_gt_validity.py +36 -0
  28. veriscript-1.5.0/tests/test_doc_data_io.py +39 -0
  29. veriscript-1.5.0/tests/test_doc_data_mixed.py +133 -0
  30. veriscript-1.5.0/tests/test_doc_metrics_agreement.py +51 -0
  31. veriscript-1.5.0/tests/test_doc_metrics_boxes.py +177 -0
  32. veriscript-1.5.0/tests/test_doc_metrics_queue.py +55 -0
  33. veriscript-1.5.0/tests/test_doc_metrics_validity.py +133 -0
  34. veriscript-1.5.0/tests/test_document_confusions.py +74 -0
  35. veriscript-1.5.0/tests/test_document_corrections.py +196 -0
  36. veriscript-1.5.0/tests/test_document_date_flags.py +67 -0
  37. veriscript-1.5.0/tests/test_document_deva_lines.py +365 -0
  38. veriscript-1.5.0/tests/test_document_eval_smoke.py +221 -0
  39. veriscript-1.5.0/tests/test_document_export.py +150 -0
  40. veriscript-1.5.0/tests/test_document_export_unicode.py +106 -0
  41. veriscript-1.5.0/tests/test_document_flags.py +173 -0
  42. veriscript-1.5.0/tests/test_document_lang.py +78 -0
  43. veriscript-1.5.0/tests/test_document_layout.py +298 -0
  44. veriscript-1.5.0/tests/test_document_markdown.py +75 -0
  45. veriscript-1.5.0/tests/test_document_memory.py +237 -0
  46. veriscript-1.5.0/tests/test_document_multi_read.py +137 -0
  47. veriscript-1.5.0/tests/test_document_number_split.py +220 -0
  48. veriscript-1.5.0/tests/test_document_ocr_state.py +48 -0
  49. veriscript-1.5.0/tests/test_document_orientation.py +176 -0
  50. veriscript-1.5.0/tests/test_document_reconcile.py +98 -0
  51. veriscript-1.5.0/tests/test_document_restore.py +123 -0
  52. veriscript-1.5.0/tests/test_document_router.py +96 -0
  53. veriscript-1.5.0/tests/test_document_routing.py +95 -0
  54. veriscript-1.5.0/tests/test_document_verifier.py +166 -0
  55. veriscript-1.5.0/tests/test_error_taxonomy.py +62 -0
  56. veriscript-1.5.0/tests/test_eval_anchor.py +119 -0
  57. veriscript-1.5.0/tests/test_eval_freeze.py +108 -0
  58. veriscript-1.5.0/tests/test_eval_lines.py +99 -0
  59. veriscript-1.5.0/tests/test_eval_mixed.py +60 -0
  60. veriscript-1.5.0/tests/test_eval_sanity.py +79 -0
  61. veriscript-1.5.0/tests/test_frontend_shell.py +47 -0
  62. veriscript-1.5.0/tests/test_harvest_gate.py +92 -0
  63. veriscript-1.5.0/tests/test_harvest_textbooks.py +148 -0
  64. veriscript-1.5.0/tests/test_lexicon.py +78 -0
  65. veriscript-1.5.0/tests/test_lexicon_fetch.py +81 -0
  66. veriscript-1.5.0/tests/test_logging_setup.py +44 -0
  67. veriscript-1.5.0/tests/test_profile_textbook_scans.py +48 -0
  68. veriscript-1.5.0/tests/test_real_lines.py +205 -0
  69. veriscript-1.5.0/tests/test_smart_upscaler.py +152 -0
  70. veriscript-1.5.0/tests/test_sr_engine_tier_c_gate.py +52 -0
  71. veriscript-1.5.0/tests/test_vector_raster.py +197 -0
  72. veriscript-1.5.0/tests/test_webapp_api.py +496 -0
  73. veriscript-1.5.0/tests/test_webapp_security.py +196 -0
  74. veriscript-1.5.0/veriscript/__init__.py +9 -0
  75. veriscript-1.5.0/veriscript/__main__.py +4 -0
  76. veriscript-1.5.0/veriscript/branding.py +26 -0
  77. veriscript-1.5.0/veriscript/calibration.py +74 -0
  78. veriscript-1.5.0/veriscript/cli.py +473 -0
  79. veriscript-1.5.0/veriscript/core/__init__.py +1 -0
  80. veriscript-1.5.0/veriscript/core/data.py +1586 -0
  81. veriscript-1.5.0/veriscript/core/degradation.py +230 -0
  82. veriscript-1.5.0/veriscript/core/metrics.py +591 -0
  83. veriscript-1.5.0/veriscript/deva/__init__.py +1 -0
  84. veriscript-1.5.0/veriscript/deva/reader.py +218 -0
  85. veriscript-1.5.0/veriscript/document/__init__.py +1 -0
  86. veriscript-1.5.0/veriscript/document/confusions.py +142 -0
  87. veriscript-1.5.0/veriscript/document/corrections.py +378 -0
  88. veriscript-1.5.0/veriscript/document/export.py +388 -0
  89. veriscript-1.5.0/veriscript/document/layout.py +528 -0
  90. veriscript-1.5.0/veriscript/document/memory.py +561 -0
  91. veriscript-1.5.0/veriscript/document/ocr.py +1295 -0
  92. veriscript-1.5.0/veriscript/document/orientation.py +247 -0
  93. veriscript-1.5.0/veriscript/document/pipeline.py +367 -0
  94. veriscript-1.5.0/veriscript/document/reconcile.py +225 -0
  95. veriscript-1.5.0/veriscript/document/restore.py +222 -0
  96. veriscript-1.5.0/veriscript/document/router.py +76 -0
  97. veriscript-1.5.0/veriscript/document/verifier.py +132 -0
  98. veriscript-1.5.0/veriscript/lexicon.py +105 -0
  99. veriscript-1.5.0/veriscript/logging_setup.py +44 -0
  100. veriscript-1.5.0/veriscript/paths.py +13 -0
  101. veriscript-1.5.0/veriscript/photo/__init__.py +1 -0
  102. veriscript-1.5.0/veriscript/photo/hybrid.py +736 -0
  103. veriscript-1.5.0/veriscript/photo/rrdbnet.py +70 -0
  104. veriscript-1.5.0/veriscript/photo/sr_engine.py +390 -0
  105. veriscript-1.5.0/veriscript/photo/srvggnet.py +69 -0
  106. veriscript-1.5.0/veriscript/photo/tv_refinement.py +142 -0
  107. veriscript-1.5.0/veriscript/photo/upscaler.py +360 -0
  108. veriscript-1.5.0/veriscript.egg-info/PKG-INFO +342 -0
  109. veriscript-1.5.0/veriscript.egg-info/SOURCES.txt +111 -0
  110. veriscript-1.5.0/veriscript.egg-info/dependency_links.txt +1 -0
  111. veriscript-1.5.0/veriscript.egg-info/entry_points.txt +3 -0
  112. veriscript-1.5.0/veriscript.egg-info/requires.txt +30 -0
  113. veriscript-1.5.0/veriscript.egg-info/top_level.txt +2 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 DontHash
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,342 @@
1
+ Metadata-Version: 2.4
2
+ Name: veriscript
3
+ Version: 1.5.0
4
+ Summary: Offline document restoration: photo/scan in, searchable PDF + honest review flags out
5
+ Author: DontHash
6
+ License: MIT License
7
+
8
+ Copyright (c) 2026 DontHash
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Project-URL: Homepage, https://github.com/DontHash/VeriScript
29
+ Project-URL: Live demo, https://veriscript.live
30
+ Project-URL: Issues, https://github.com/DontHash/VeriScript/issues
31
+ Keywords: ocr,devanagari,nepali,document-restore,searchable-pdf,offline
32
+ Classifier: Development Status :: 4 - Beta
33
+ Classifier: Environment :: Console
34
+ Classifier: Intended Audience :: End Users/Desktop
35
+ Classifier: License :: OSI Approved :: MIT License
36
+ Classifier: Operating System :: Microsoft :: Windows
37
+ Classifier: Operating System :: POSIX :: Linux
38
+ Classifier: Programming Language :: Python :: 3
39
+ Classifier: Topic :: Multimedia :: Graphics :: Graphics Conversion
40
+ Classifier: Topic :: Text Processing :: Linguistic
41
+ Requires-Python: >=3.10
42
+ Description-Content-Type: text/markdown
43
+ License-File: LICENSE
44
+ Requires-Dist: numpy
45
+ Requires-Dist: opencv-python
46
+ Requires-Dist: scipy
47
+ Requires-Dist: scikit-image
48
+ Requires-Dist: onnxruntime
49
+ Requires-Dist: Pillow
50
+ Requires-Dist: jiwer
51
+ Requires-Dist: rapidocr==3.9.*
52
+ Requires-Dist: reportlab
53
+ Requires-Dist: pypdfium2
54
+ Provides-Extra: photo
55
+ Requires-Dist: torch; extra == "photo"
56
+ Provides-Extra: verifier
57
+ Requires-Dist: torch; extra == "verifier"
58
+ Requires-Dist: transformers; extra == "verifier"
59
+ Requires-Dist: huggingface_hub; extra == "verifier"
60
+ Provides-Extra: dev
61
+ Requires-Dist: pytest; extra == "dev"
62
+ Requires-Dist: httpx; extra == "dev"
63
+ Requires-Dist: matplotlib; extra == "dev"
64
+ Requires-Dist: scipy; extra == "dev"
65
+ Requires-Dist: torch; extra == "dev"
66
+ Requires-Dist: pyiqa; extra == "dev"
67
+ Requires-Dist: datasets; extra == "dev"
68
+ Requires-Dist: transformers; extra == "dev"
69
+ Requires-Dist: PySide6; extra == "dev"
70
+ Requires-Dist: google-genai; extra == "dev"
71
+ Dynamic: license-file
72
+
73
+ # VeriScript
74
+
75
+ **Offline document restoration.** A photo, scan or PDF goes in; a cleaned page
76
+ with a **searchable PDF**, overlay, transcript, **Markdown** and OCR JSON comes
77
+ out — with honest flags on the numbers it is not sure about. Devanagari
78
+ (Nepali/Hindi) first, English supported by the same pipeline.
79
+
80
+ Everything runs on your machine. No cloud calls, no telemetry, no account.
81
+
82
+ **Live demo:** <https://veriscript.live> — one page at
83
+ a time, no signup; demo uploads are deleted after the 60-minute retention
84
+ window.
85
+
86
+ ```
87
+ photo/scan/PDF ──▶ orientation ──▶ layout ──▶ OCR ──▶ risk flags ──▶ searchable PDF
88
+ ├─ overlay.png (review boxes)
89
+ ├─ transcript.txt
90
+ ├─ transcript.md (Markdown)
91
+ └─ ocr.json (tokens + queue)
92
+ ```
93
+
94
+ ## Before / after
95
+
96
+ ![VeriScript before and after: an aged letterpress page and a shadowed invoice photo, restored with the numbered review queue](docs/assets/before_after.png)
97
+
98
+ Real pipeline output. **Green** = accepted, **amber** = uncertain, **red** =
99
+ digit conflict with the alternative reading kept. The letterpress page is from
100
+ heiDATA (doi:10.11588/data/EGOKEI, CC BY 4.0); the photo row is this
101
+ repository's own synthetic fixture under a simulated phone shadow. Regenerate
102
+ with `python scripts/make_demo_image.py`.
103
+
104
+ ## Why it exists
105
+
106
+ - **It will not invent the numbers on your bill.** Digits are the highest-risk
107
+ tokens, so they are re-read, risk-ranked and put in a review queue instead of
108
+ being silently guessed. The queue is measured, not asserted
109
+ ([EVALUATION.md](docs/EVALUATION.md)).
110
+ - **Devanagari is a first-class citizen, not a language pack.** Letterpress
111
+ books, modern government PDFs and phone photos each have frozen evaluation
112
+ sets with published numbers and confidence intervals.
113
+ - **Offline by default.** Privacy is the reason to use this instead of a cloud
114
+ lens: the pipeline runs with networking disabled.
115
+
116
+ ## Install
117
+
118
+ From a source checkout (the package is not on PyPI yet):
119
+
120
+ ```bash
121
+ pip install -r requirements.txt # full runtime (includes torch)
122
+ pip install -r requirements-dev.txt # + dev/eval tooling (optional)
123
+ ```
124
+
125
+ Or install the Python package itself — the document pipeline is torch-free
126
+ (`photo`/`verifier` are opt-in extras):
127
+
128
+ ```bash
129
+ pip install -e . # document pipeline + web/CLI
130
+ pip install -e ".[photo]" # + photo upscaling (torch)
131
+ pip install -e ".[verifier]" # + the optional bodhan digit verifier
132
+ ```
133
+
134
+ Optional: [Tesseract 5](https://github.com/UB-Mannheim/tesseract/wiki) as a
135
+ second OCR backend (clean-scan fallback and the OSD rotation fallback).
136
+
137
+ ## Use
138
+
139
+ ### CLI
140
+
141
+ ```bash
142
+ # Photo/scan in: searchable PDF + overlay + transcript + Markdown + JSON out
143
+ python cli.py --mode document --input page.jpg --output out/run1
144
+
145
+ # PDF in: first page by default; --max-pages 3 = first three pages
146
+ python cli.py --mode document --input scan.pdf --output out/run1 --max-pages 3
147
+
148
+ # PDF → Markdown: per-page .md always; --max-pages 0 adds one combined .md
149
+ python cli.py --mode document --input report.pdf --output out/report --max-pages 0
150
+
151
+ # Same CLI as a module / installed console script:
152
+ python -m veriscript --mode document --input report.pdf --output out/report
153
+
154
+ # Photo upscaling (shares the same CLI/app)
155
+ python cli.py --mode photo --input photos/ --output out/upscaled --scale 4
156
+ ```
157
+
158
+ Useful knobs (all default to the measured best configuration):
159
+
160
+ ```
161
+ --ocr rapidocr|tesseract --lang en|ne|hi --deskew
162
+ --no-reading-order --rotate auto|off --repass-digits/--no-repass-digits
163
+ --digit-verifier off|bodhan (optional second-model digit check)
164
+ --no-pdf --no-overlay --no-txt --no-md
165
+ ```
166
+
167
+ ### Web app (the studio)
168
+
169
+ The single UI — FastAPI + SolidJS, running the same pipeline and outputs
170
+ (searchable PDF, transcript, **Markdown**, OCR JSON) as the CLI.
171
+
172
+ ```bash
173
+ pip install -r webapp/requirements-web.txt # backend, on top of requirements.txt
174
+ cd webapp/frontend && npm install && npm run build && cd ../..
175
+ python webapp/server.py # http://127.0.0.1:8000
176
+ ```
177
+
178
+ Hosted deployments (Docker, basic auth, HF Spaces/Render/Fly) are covered in
179
+ [docs/DEPLOY.md](docs/DEPLOY.md). The server binds to localhost by default;
180
+ the pipeline, models and any training artifacts stay server-side. The photo
181
+ upscaler is CLI-only (its weights are non-commercial); the web studio is
182
+ document-only — see [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md).
183
+
184
+ For JS/TS integrations there is a typed HTTP client in
185
+ [`packages/veriscript-client`](packages/veriscript-client) (npm:
186
+ `veriscript-client`; not published yet).
187
+
188
+ ### Outputs (per page)
189
+
190
+ For an input `page.jpg`, every run writes:
191
+
192
+ | File | Contents |
193
+ |---|---|
194
+ | `page.pdf` | searchable PDF — cleaned image + invisible Unicode text layer (Devanagari included; bundled Mukta font), selectable/copyable |
195
+ | `page_overlay.png` | cleaned page with numbered review boxes for the flagged tokens |
196
+ | `page.txt` | reading-order plain text |
197
+ | `page.md` | **Markdown transcript** — the text plus the review queue as a table (this is the PDF → Markdown output) |
198
+ | `page.json` | tokens, risk flags, review queue, orientation/reading-order evidence |
199
+ | `report_pNNN.*` | per-page set for a multi-page PDF (`report_p000.pdf`, `report_p000.md`, …) |
200
+ | `report_combined.pdf` / `.txt` / `.md` | all pages joined with `--max-pages 0`; the combined Markdown has one `## Page N` section per page |
201
+
202
+ The web studio serves the same artifacts under canonical names
203
+ (`restored.png`, `searchable.pdf`, `transcript.txt`, `transcript.md`,
204
+ `ocr.json`, `combined.md`). `--no-md` (CLI) or the Markdown toggle (web)
205
+ skips the `.md` files.
206
+
207
+ ## Measured behaviour
208
+
209
+ Numbers below come from hash-frozen evaluation sets with bootstrap 95% CIs
210
+ (full tables, provenance and rebuild commands: [docs/EVALUATION.md](docs/EVALUATION.md)).
211
+ No number in this repo is marketing.
212
+
213
+ | Capability | Measured |
214
+ |---|---|
215
+ | Orientation (EXIF + 0/90/180/270 via OCR evidence) | 16/16 synthetic; 0/30 false rotations on upright real scans; 90/90 rotated pages decided |
216
+ | Devanagari — modern government PDFs (raw engine) | CER **0.135 [0.097–0.186]**, exact-token recall 0.877 (n=41, Gemini-2.5-Pro anchor GT, engine-corroborated) |
217
+ | Devanagari — letterpress books (the honest bar) | CER 0.434 [0.387–0.481] raw; **0.373** under the default `auto` line reader (42/69 pages engaged, Appendix AE); exact-token recall 0.647 (n=69, human-corrected ALTO GT) |
218
+ | Devanagari — rendered fixtures | clean CER 0.030/0.028; degraded ne 0.094 (pass) / hi 0.170 (heavy fails) |
219
+ | Digit honesty — 2× re-pass (default ON for Devanagari) | review-queue digit recall@10 **0.88** letterpress / **0.38** PDFs (was 0.73 / 0.15), ≈ +0.4 s/page |
220
+ | Flags + calibration | `script_mismatch`, `invalid_sequence` (impossible combining sequence), isotonic `cal_conf`; ECE 0.82 → 0.30 on scans; optional `unknown_word` lexicon flag (needs `scripts/fetch_nepali_lexicon.py`) lifts the frozen-PDF review queue R@10 **0.148 → 0.175** with the digit queue unchanged |
221
+ | Optional digit verifier (`--digit-verifier bodhan`) | flags 0.71 recall / 0.81 precision; **opt-in** (cost gate failed: +8.3 s/page) |
222
+ | Multi-page PDF | `--max-pages 0` → every page + `<stem>_combined.pdf`/`.txt`; GUI "All pages" checkbox |
223
+ | Two-column reading order | arXiv set WER 0.870 → 0.268 (−69%) on split pages; >2 columns unsupported |
224
+ | Phone-photo proxy (not real photos) | CER 0.370 medium / 0.757 heavy (clean 0.135) — recognition, not honesty, is what breaks |
225
+ | Mixed-page router (opt-in) | restored text + untouched logos/photos (non-text PSNR 60–67 dB vs 18–23 dB naive); 1 invented token on photo texture → not auto-enabled |
226
+ | Detection on real pages | covers 97.5% of ALTO line boxes (area view) — recognition, not detection, is the bottleneck |
227
+
228
+ Known limitations are listed with their evidence in
229
+ [docs/EVALUATION.md](docs/EVALUATION.md) and the full research log in
230
+ [docs/PLAN.md](docs/PLAN.md). Since 2026-10-01 the image paths (CLI and web
231
+ studio) default to the `auto` line-reader gate — running text + aged paper
232
+ only. PDF inputs keep the engine reading, and the hosted document-only demo
233
+ ships no reader checkpoint, so it stays on the engine there.
234
+
235
+ ## How it compares
236
+
237
+ Head-to-head on the frozen Devanagari hard-10 (the table-heavy hardest 10 of
238
+ the 41-page `nepali_pdf_v2` set, Gemini-2.5-Pro anchor GT; every arm scored by
239
+ the same metric code — canonical write-up in
240
+ [docs/BENCHMARK.md](docs/BENCHMARK.md), raw lab notes and reproduction
241
+ commands in [evals/bakeoff_results.md](evals/bakeoff_results.md)):
242
+
243
+ | Engine | CER (mean [95% CI]) | Median CER | bagCER | Speed (s/page) | Runs on CPU |
244
+ |---|---|---|---|---|---|
245
+ | Surya OCR 2 | 0.166 [0.100–0.258] | 0.107 | 0.360 | 172.0 | ✗ |
246
+ | Qwen3-VL-8B NF4 | 0.176 [0.129–0.235] | 0.141 | 0.369 | 152.7 | ✗ |
247
+ | **VeriScript pipeline (default)** | **0.255 [0.202–0.329]** | **0.227** | 0.376 | **7.3** | ✓ |
248
+ | raw RapidOCR engine | 0.530 [0.463–0.585] | 0.543 | 0.380 | 3.8 | ✓ |
249
+ | PaddleOCR 3.x full pipeline | 0.662 [0.585–0.713] | 0.687 | 0.400 | 21.8 | ✓ |
250
+
251
+ Speeds are per page on the harness hardware — ours, raw and Surya on an
252
+ RTX 2050 4 GB (Surya via llama.cpp/Vulkan), Qwen on a T4 (NF4), PaddleOCR on
253
+ **CPU**. Runs on CPU: ✓ = CPU-first (PaddleOCR was measured CPU-only here;
254
+ VeriScript and RapidOCR ship CPU-first); ✗ = GPU-class in practice — a GPU was
255
+ used and CPU would be far slower.
256
+
257
+ **Metrics.** **CER** (character error rate) is the edit distance between the
258
+ OCR output and the ground truth (insertions + deletions + substitutions)
259
+ divided by the number of ground-truth characters: order-sensitive, 0 is
260
+ perfect, lower is better. **bagCER** is the same character distance after
261
+ alphabetically sorting the tokens on both sides — it ignores reading order, so
262
+ it isolates glyph recognition from layout (a page with perfect words in the
263
+ wrong order still scores 0). **Median CER** is the middle page's CER, robust
264
+ to a single catastrophic page; **s/page** is seconds per page.
265
+
266
+ **Reading.** bagCER is tied across every arm (0.360–0.400): glyph recognition
267
+ is a commodity, and the page-CER spread is reading order, table structure and
268
+ digit handling — which is where this pipeline earns its keep (0.530 → 0.255
269
+ over the raw engine on the same frozen pages, while staying CPU-first,
270
+ offline and Apache-2.0). Surya 2 reads best by median at ~23× our page time
271
+ and OpenRAIL-M weights, and the GPU arms invent text the source never
272
+ contained (82–224 tokens over the 10 pages); only this stack ships an
273
+ alternative-reading digit review queue.
274
+
275
+ ## Project layout
276
+
277
+ ```
278
+ veriscript/ the product package
279
+ cli.py command-line entry point (document + photo)
280
+ core/ dataset IO + metrics (doc_data, doc_metrics), degradation
281
+ document/ pipeline: OCR, layout, orientation, restore, export, router, verifier
282
+ deva/ Devanagari line reader (CRNN+CTC), opt-in via --deva-lines
283
+ photo/ photo/vector stack (parked; CLI photo mode)
284
+ branding.py, logging_setup.py, lexicon.py, calibration.py
285
+ cli.py compatibility launcher for `python -m veriscript`
286
+ webapp/ the studio: FastAPI backend (vvweb/) + SolidJS frontend
287
+ deva_crnn/ training package for the line reader (repo tooling)
288
+ evals/harness/ evaluation harnesses (frozen sets, metrics, gates)
289
+ evals/manifests/ content-hash frozen evaluation sets
290
+ scripts/ data acquisition, training export, gates, release tooling
291
+ tests/ the test suite (pytest counts them; ~338)
292
+ docs/ architecture, evaluation, deployment, licensing, plan
293
+ ```
294
+
295
+ ## Docs
296
+
297
+ | Document | Contents |
298
+ |---|---|
299
+ | [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) | module map, data flow, extension points |
300
+ | [docs/EVALUATION.md](docs/EVALUATION.md) | frozen sets, metrics, how to reproduce every number |
301
+ | [docs/DEPLOY.md](docs/DEPLOY.md) | hosted web app (Docker, auth, spaces) |
302
+ | [docs/LICENSES.md](docs/LICENSES.md) | dependency + model license gate |
303
+ | [docs/RELEASE.md](docs/RELEASE.md) | release checklist, versioning, desktop packaging path |
304
+ | [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md) | web studio: audit, diet and phased plan |
305
+ | [docs/MARKET.md](docs/MARKET.md) | market landscape, competitors, positioning |
306
+ | [evals/bakeoff_results.md](evals/bakeoff_results.md) | head-to-head bake-off results on the frozen sets |
307
+ | [docs/PLAN.md](docs/PLAN.md) | the full research log and decision record |
308
+
309
+ ## Tests
310
+
311
+ ```bash
312
+ python -m pytest tests/ -q
313
+ ```
314
+
315
+ The suite (≈336 collected) covers: OCR/export/routing/CLI/web, layout, orientation, language plumbing,
316
+ frozen-manifest and CI guards, GT-validity audits, error taxonomy, anchor
317
+ harness, queue metrics, Unicode-path IO, Devanagari line synthesis, CRNN
318
+ plumbing (height/width round-trip, beam search, cosine fine-tune), the
319
+ real-line mining gate, the opt-in line reader (merge geometry, table-rule
320
+ blocking, auto policy, graceful degradation, provenance), the lexicon
321
+ fetch/build + `unknown_word` flag mechanics, the textbook harvest parsers +
322
+ real-print scan probe, the CC-100 corpus builder + controlled sampler, the
323
+ Vertex wheel builder and the trainer GCS output path, the Devanagari
324
+ PDF text-layer extraction round-trips, the logging configuration +
325
+ server-error capture, and the Markdown export (per page and combined).
326
+
327
+ ## Data, models and licensing
328
+
329
+ This repository contains **code and evaluation evidence only**: no training
330
+ datasets, no model checkpoints and no redistributed third-party corpora.
331
+ Evaluation manifests record hashes and provenance; the data itself stays local
332
+ (`data/`, `weights/` and `artifacts/` are git-ignored). The optional
333
+ Devanagari lexicon behind the `unknown_word` flag is built locally by
334
+ `scripts/fetch_nepali_lexicon.py` (Apache-2.0 + MIT sources) and is not
335
+ redistributed. Third-party license
336
+ obligations and the air-gap story are tracked in
337
+ [docs/LICENSES.md](docs/LICENSES.md).
338
+
339
+ MIT licensed — see [LICENSE](LICENSE). Third-party components keep their own
340
+ licenses (RapidOCR/PP-OCR Apache-2.0, reportlab/pypdfium2 permissive, PySide6
341
+ LGPL, UltraSharp weights CC-BY-NC-SA and therefore excluded from commercial
342
+ builds).
@@ -0,0 +1,270 @@
1
+ # VeriScript
2
+
3
+ **Offline document restoration.** A photo, scan or PDF goes in; a cleaned page
4
+ with a **searchable PDF**, overlay, transcript, **Markdown** and OCR JSON comes
5
+ out — with honest flags on the numbers it is not sure about. Devanagari
6
+ (Nepali/Hindi) first, English supported by the same pipeline.
7
+
8
+ Everything runs on your machine. No cloud calls, no telemetry, no account.
9
+
10
+ **Live demo:** <https://veriscript.live> — one page at
11
+ a time, no signup; demo uploads are deleted after the 60-minute retention
12
+ window.
13
+
14
+ ```
15
+ photo/scan/PDF ──▶ orientation ──▶ layout ──▶ OCR ──▶ risk flags ──▶ searchable PDF
16
+ ├─ overlay.png (review boxes)
17
+ ├─ transcript.txt
18
+ ├─ transcript.md (Markdown)
19
+ └─ ocr.json (tokens + queue)
20
+ ```
21
+
22
+ ## Before / after
23
+
24
+ ![VeriScript before and after: an aged letterpress page and a shadowed invoice photo, restored with the numbered review queue](docs/assets/before_after.png)
25
+
26
+ Real pipeline output. **Green** = accepted, **amber** = uncertain, **red** =
27
+ digit conflict with the alternative reading kept. The letterpress page is from
28
+ heiDATA (doi:10.11588/data/EGOKEI, CC BY 4.0); the photo row is this
29
+ repository's own synthetic fixture under a simulated phone shadow. Regenerate
30
+ with `python scripts/make_demo_image.py`.
31
+
32
+ ## Why it exists
33
+
34
+ - **It will not invent the numbers on your bill.** Digits are the highest-risk
35
+ tokens, so they are re-read, risk-ranked and put in a review queue instead of
36
+ being silently guessed. The queue is measured, not asserted
37
+ ([EVALUATION.md](docs/EVALUATION.md)).
38
+ - **Devanagari is a first-class citizen, not a language pack.** Letterpress
39
+ books, modern government PDFs and phone photos each have frozen evaluation
40
+ sets with published numbers and confidence intervals.
41
+ - **Offline by default.** Privacy is the reason to use this instead of a cloud
42
+ lens: the pipeline runs with networking disabled.
43
+
44
+ ## Install
45
+
46
+ From a source checkout (the package is not on PyPI yet):
47
+
48
+ ```bash
49
+ pip install -r requirements.txt # full runtime (includes torch)
50
+ pip install -r requirements-dev.txt # + dev/eval tooling (optional)
51
+ ```
52
+
53
+ Or install the Python package itself — the document pipeline is torch-free
54
+ (`photo`/`verifier` are opt-in extras):
55
+
56
+ ```bash
57
+ pip install -e . # document pipeline + web/CLI
58
+ pip install -e ".[photo]" # + photo upscaling (torch)
59
+ pip install -e ".[verifier]" # + the optional bodhan digit verifier
60
+ ```
61
+
62
+ Optional: [Tesseract 5](https://github.com/UB-Mannheim/tesseract/wiki) as a
63
+ second OCR backend (clean-scan fallback and the OSD rotation fallback).
64
+
65
+ ## Use
66
+
67
+ ### CLI
68
+
69
+ ```bash
70
+ # Photo/scan in: searchable PDF + overlay + transcript + Markdown + JSON out
71
+ python cli.py --mode document --input page.jpg --output out/run1
72
+
73
+ # PDF in: first page by default; --max-pages 3 = first three pages
74
+ python cli.py --mode document --input scan.pdf --output out/run1 --max-pages 3
75
+
76
+ # PDF → Markdown: per-page .md always; --max-pages 0 adds one combined .md
77
+ python cli.py --mode document --input report.pdf --output out/report --max-pages 0
78
+
79
+ # Same CLI as a module / installed console script:
80
+ python -m veriscript --mode document --input report.pdf --output out/report
81
+
82
+ # Photo upscaling (shares the same CLI/app)
83
+ python cli.py --mode photo --input photos/ --output out/upscaled --scale 4
84
+ ```
85
+
86
+ Useful knobs (all default to the measured best configuration):
87
+
88
+ ```
89
+ --ocr rapidocr|tesseract --lang en|ne|hi --deskew
90
+ --no-reading-order --rotate auto|off --repass-digits/--no-repass-digits
91
+ --digit-verifier off|bodhan (optional second-model digit check)
92
+ --no-pdf --no-overlay --no-txt --no-md
93
+ ```
94
+
95
+ ### Web app (the studio)
96
+
97
+ The single UI — FastAPI + SolidJS, running the same pipeline and outputs
98
+ (searchable PDF, transcript, **Markdown**, OCR JSON) as the CLI.
99
+
100
+ ```bash
101
+ pip install -r webapp/requirements-web.txt # backend, on top of requirements.txt
102
+ cd webapp/frontend && npm install && npm run build && cd ../..
103
+ python webapp/server.py # http://127.0.0.1:8000
104
+ ```
105
+
106
+ Hosted deployments (Docker, basic auth, HF Spaces/Render/Fly) are covered in
107
+ [docs/DEPLOY.md](docs/DEPLOY.md). The server binds to localhost by default;
108
+ the pipeline, models and any training artifacts stay server-side. The photo
109
+ upscaler is CLI-only (its weights are non-commercial); the web studio is
110
+ document-only — see [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md).
111
+
112
+ For JS/TS integrations there is a typed HTTP client in
113
+ [`packages/veriscript-client`](packages/veriscript-client) (npm:
114
+ `veriscript-client`; not published yet).
115
+
116
+ ### Outputs (per page)
117
+
118
+ For an input `page.jpg`, every run writes:
119
+
120
+ | File | Contents |
121
+ |---|---|
122
+ | `page.pdf` | searchable PDF — cleaned image + invisible Unicode text layer (Devanagari included; bundled Mukta font), selectable/copyable |
123
+ | `page_overlay.png` | cleaned page with numbered review boxes for the flagged tokens |
124
+ | `page.txt` | reading-order plain text |
125
+ | `page.md` | **Markdown transcript** — the text plus the review queue as a table (this is the PDF → Markdown output) |
126
+ | `page.json` | tokens, risk flags, review queue, orientation/reading-order evidence |
127
+ | `report_pNNN.*` | per-page set for a multi-page PDF (`report_p000.pdf`, `report_p000.md`, …) |
128
+ | `report_combined.pdf` / `.txt` / `.md` | all pages joined with `--max-pages 0`; the combined Markdown has one `## Page N` section per page |
129
+
130
+ The web studio serves the same artifacts under canonical names
131
+ (`restored.png`, `searchable.pdf`, `transcript.txt`, `transcript.md`,
132
+ `ocr.json`, `combined.md`). `--no-md` (CLI) or the Markdown toggle (web)
133
+ skips the `.md` files.
134
+
135
+ ## Measured behaviour
136
+
137
+ Numbers below come from hash-frozen evaluation sets with bootstrap 95% CIs
138
+ (full tables, provenance and rebuild commands: [docs/EVALUATION.md](docs/EVALUATION.md)).
139
+ No number in this repo is marketing.
140
+
141
+ | Capability | Measured |
142
+ |---|---|
143
+ | Orientation (EXIF + 0/90/180/270 via OCR evidence) | 16/16 synthetic; 0/30 false rotations on upright real scans; 90/90 rotated pages decided |
144
+ | Devanagari — modern government PDFs (raw engine) | CER **0.135 [0.097–0.186]**, exact-token recall 0.877 (n=41, Gemini-2.5-Pro anchor GT, engine-corroborated) |
145
+ | Devanagari — letterpress books (the honest bar) | CER 0.434 [0.387–0.481] raw; **0.373** under the default `auto` line reader (42/69 pages engaged, Appendix AE); exact-token recall 0.647 (n=69, human-corrected ALTO GT) |
146
+ | Devanagari — rendered fixtures | clean CER 0.030/0.028; degraded ne 0.094 (pass) / hi 0.170 (heavy fails) |
147
+ | Digit honesty — 2× re-pass (default ON for Devanagari) | review-queue digit recall@10 **0.88** letterpress / **0.38** PDFs (was 0.73 / 0.15), ≈ +0.4 s/page |
148
+ | Flags + calibration | `script_mismatch`, `invalid_sequence` (impossible combining sequence), isotonic `cal_conf`; ECE 0.82 → 0.30 on scans; optional `unknown_word` lexicon flag (needs `scripts/fetch_nepali_lexicon.py`) lifts the frozen-PDF review queue R@10 **0.148 → 0.175** with the digit queue unchanged |
149
+ | Optional digit verifier (`--digit-verifier bodhan`) | flags 0.71 recall / 0.81 precision; **opt-in** (cost gate failed: +8.3 s/page) |
150
+ | Multi-page PDF | `--max-pages 0` → every page + `<stem>_combined.pdf`/`.txt`; GUI "All pages" checkbox |
151
+ | Two-column reading order | arXiv set WER 0.870 → 0.268 (−69%) on split pages; >2 columns unsupported |
152
+ | Phone-photo proxy (not real photos) | CER 0.370 medium / 0.757 heavy (clean 0.135) — recognition, not honesty, is what breaks |
153
+ | Mixed-page router (opt-in) | restored text + untouched logos/photos (non-text PSNR 60–67 dB vs 18–23 dB naive); 1 invented token on photo texture → not auto-enabled |
154
+ | Detection on real pages | covers 97.5% of ALTO line boxes (area view) — recognition, not detection, is the bottleneck |
155
+
156
+ Known limitations are listed with their evidence in
157
+ [docs/EVALUATION.md](docs/EVALUATION.md) and the full research log in
158
+ [docs/PLAN.md](docs/PLAN.md). Since 2026-10-01 the image paths (CLI and web
159
+ studio) default to the `auto` line-reader gate — running text + aged paper
160
+ only. PDF inputs keep the engine reading, and the hosted document-only demo
161
+ ships no reader checkpoint, so it stays on the engine there.
162
+
163
+ ## How it compares
164
+
165
+ Head-to-head on the frozen Devanagari hard-10 (the table-heavy hardest 10 of
166
+ the 41-page `nepali_pdf_v2` set, Gemini-2.5-Pro anchor GT; every arm scored by
167
+ the same metric code — canonical write-up in
168
+ [docs/BENCHMARK.md](docs/BENCHMARK.md), raw lab notes and reproduction
169
+ commands in [evals/bakeoff_results.md](evals/bakeoff_results.md)):
170
+
171
+ | Engine | CER (mean [95% CI]) | Median CER | bagCER | Speed (s/page) | Runs on CPU |
172
+ |---|---|---|---|---|---|
173
+ | Surya OCR 2 | 0.166 [0.100–0.258] | 0.107 | 0.360 | 172.0 | ✗ |
174
+ | Qwen3-VL-8B NF4 | 0.176 [0.129–0.235] | 0.141 | 0.369 | 152.7 | ✗ |
175
+ | **VeriScript pipeline (default)** | **0.255 [0.202–0.329]** | **0.227** | 0.376 | **7.3** | ✓ |
176
+ | raw RapidOCR engine | 0.530 [0.463–0.585] | 0.543 | 0.380 | 3.8 | ✓ |
177
+ | PaddleOCR 3.x full pipeline | 0.662 [0.585–0.713] | 0.687 | 0.400 | 21.8 | ✓ |
178
+
179
+ Speeds are per page on the harness hardware — ours, raw and Surya on an
180
+ RTX 2050 4 GB (Surya via llama.cpp/Vulkan), Qwen on a T4 (NF4), PaddleOCR on
181
+ **CPU**. Runs on CPU: ✓ = CPU-first (PaddleOCR was measured CPU-only here;
182
+ VeriScript and RapidOCR ship CPU-first); ✗ = GPU-class in practice — a GPU was
183
+ used and CPU would be far slower.
184
+
185
+ **Metrics.** **CER** (character error rate) is the edit distance between the
186
+ OCR output and the ground truth (insertions + deletions + substitutions)
187
+ divided by the number of ground-truth characters: order-sensitive, 0 is
188
+ perfect, lower is better. **bagCER** is the same character distance after
189
+ alphabetically sorting the tokens on both sides — it ignores reading order, so
190
+ it isolates glyph recognition from layout (a page with perfect words in the
191
+ wrong order still scores 0). **Median CER** is the middle page's CER, robust
192
+ to a single catastrophic page; **s/page** is seconds per page.
193
+
194
+ **Reading.** bagCER is tied across every arm (0.360–0.400): glyph recognition
195
+ is a commodity, and the page-CER spread is reading order, table structure and
196
+ digit handling — which is where this pipeline earns its keep (0.530 → 0.255
197
+ over the raw engine on the same frozen pages, while staying CPU-first,
198
+ offline and Apache-2.0). Surya 2 reads best by median at ~23× our page time
199
+ and OpenRAIL-M weights, and the GPU arms invent text the source never
200
+ contained (82–224 tokens over the 10 pages); only this stack ships an
201
+ alternative-reading digit review queue.
202
+
203
+ ## Project layout
204
+
205
+ ```
206
+ veriscript/ the product package
207
+ cli.py command-line entry point (document + photo)
208
+ core/ dataset IO + metrics (doc_data, doc_metrics), degradation
209
+ document/ pipeline: OCR, layout, orientation, restore, export, router, verifier
210
+ deva/ Devanagari line reader (CRNN+CTC), opt-in via --deva-lines
211
+ photo/ photo/vector stack (parked; CLI photo mode)
212
+ branding.py, logging_setup.py, lexicon.py, calibration.py
213
+ cli.py compatibility launcher for `python -m veriscript`
214
+ webapp/ the studio: FastAPI backend (vvweb/) + SolidJS frontend
215
+ deva_crnn/ training package for the line reader (repo tooling)
216
+ evals/harness/ evaluation harnesses (frozen sets, metrics, gates)
217
+ evals/manifests/ content-hash frozen evaluation sets
218
+ scripts/ data acquisition, training export, gates, release tooling
219
+ tests/ the test suite (pytest counts them; ~338)
220
+ docs/ architecture, evaluation, deployment, licensing, plan
221
+ ```
222
+
223
+ ## Docs
224
+
225
+ | Document | Contents |
226
+ |---|---|
227
+ | [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) | module map, data flow, extension points |
228
+ | [docs/EVALUATION.md](docs/EVALUATION.md) | frozen sets, metrics, how to reproduce every number |
229
+ | [docs/DEPLOY.md](docs/DEPLOY.md) | hosted web app (Docker, auth, spaces) |
230
+ | [docs/LICENSES.md](docs/LICENSES.md) | dependency + model license gate |
231
+ | [docs/RELEASE.md](docs/RELEASE.md) | release checklist, versioning, desktop packaging path |
232
+ | [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md) | web studio: audit, diet and phased plan |
233
+ | [docs/MARKET.md](docs/MARKET.md) | market landscape, competitors, positioning |
234
+ | [evals/bakeoff_results.md](evals/bakeoff_results.md) | head-to-head bake-off results on the frozen sets |
235
+ | [docs/PLAN.md](docs/PLAN.md) | the full research log and decision record |
236
+
237
+ ## Tests
238
+
239
+ ```bash
240
+ python -m pytest tests/ -q
241
+ ```
242
+
243
+ The suite (≈336 collected) covers: OCR/export/routing/CLI/web, layout, orientation, language plumbing,
244
+ frozen-manifest and CI guards, GT-validity audits, error taxonomy, anchor
245
+ harness, queue metrics, Unicode-path IO, Devanagari line synthesis, CRNN
246
+ plumbing (height/width round-trip, beam search, cosine fine-tune), the
247
+ real-line mining gate, the opt-in line reader (merge geometry, table-rule
248
+ blocking, auto policy, graceful degradation, provenance), the lexicon
249
+ fetch/build + `unknown_word` flag mechanics, the textbook harvest parsers +
250
+ real-print scan probe, the CC-100 corpus builder + controlled sampler, the
251
+ Vertex wheel builder and the trainer GCS output path, the Devanagari
252
+ PDF text-layer extraction round-trips, the logging configuration +
253
+ server-error capture, and the Markdown export (per page and combined).
254
+
255
+ ## Data, models and licensing
256
+
257
+ This repository contains **code and evaluation evidence only**: no training
258
+ datasets, no model checkpoints and no redistributed third-party corpora.
259
+ Evaluation manifests record hashes and provenance; the data itself stays local
260
+ (`data/`, `weights/` and `artifacts/` are git-ignored). The optional
261
+ Devanagari lexicon behind the `unknown_word` flag is built locally by
262
+ `scripts/fetch_nepali_lexicon.py` (Apache-2.0 + MIT sources) and is not
263
+ redistributed. Third-party license
264
+ obligations and the air-gap story are tracked in
265
+ [docs/LICENSES.md](docs/LICENSES.md).
266
+
267
+ MIT licensed — see [LICENSE](LICENSE). Third-party components keep their own
268
+ licenses (RapidOCR/PP-OCR Apache-2.0, reportlab/pypdfium2 permissive, PySide6
269
+ LGPL, UltraSharp weights CC-BY-NC-SA and therefore excluded from commercial
270
+ builds).