veriscript 1.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- veriscript-1.5.0/LICENSE +21 -0
- veriscript-1.5.0/PKG-INFO +342 -0
- veriscript-1.5.0/README.md +270 -0
- veriscript-1.5.0/calibration/rapidocr_devanagari_v1.json +60 -0
- veriscript-1.5.0/deva_crnn/__init__.py +1 -0
- veriscript-1.5.0/deva_crnn/augment.py +242 -0
- veriscript-1.5.0/deva_crnn/charset.py +31 -0
- veriscript-1.5.0/deva_crnn/data.py +48 -0
- veriscript-1.5.0/deva_crnn/gate.py +97 -0
- veriscript-1.5.0/deva_crnn/model.py +44 -0
- veriscript-1.5.0/deva_crnn/predict.py +127 -0
- veriscript-1.5.0/deva_crnn/train.py +254 -0
- veriscript-1.5.0/fonts/Mukta-Regular.ttf +0 -0
- veriscript-1.5.0/fonts/OFL.txt +93 -0
- veriscript-1.5.0/pyproject.toml +84 -0
- veriscript-1.5.0/setup.cfg +4 -0
- veriscript-1.5.0/tests/test_anchor_gemini.py +151 -0
- veriscript-1.5.0/tests/test_bakeoff_models.py +145 -0
- veriscript-1.5.0/tests/test_build_deva_wheel.py +44 -0
- veriscript-1.5.0/tests/test_cli_document.py +131 -0
- veriscript-1.5.0/tests/test_demo_asset.py +38 -0
- veriscript-1.5.0/tests/test_deva_corpus_fetch.py +89 -0
- veriscript-1.5.0/tests/test_deva_crnn.py +292 -0
- veriscript-1.5.0/tests/test_deva_gate.py +38 -0
- veriscript-1.5.0/tests/test_deva_lines.py +149 -0
- veriscript-1.5.0/tests/test_deva_train_out.py +32 -0
- veriscript-1.5.0/tests/test_doc_data_gt_validity.py +36 -0
- veriscript-1.5.0/tests/test_doc_data_io.py +39 -0
- veriscript-1.5.0/tests/test_doc_data_mixed.py +133 -0
- veriscript-1.5.0/tests/test_doc_metrics_agreement.py +51 -0
- veriscript-1.5.0/tests/test_doc_metrics_boxes.py +177 -0
- veriscript-1.5.0/tests/test_doc_metrics_queue.py +55 -0
- veriscript-1.5.0/tests/test_doc_metrics_validity.py +133 -0
- veriscript-1.5.0/tests/test_document_confusions.py +74 -0
- veriscript-1.5.0/tests/test_document_corrections.py +196 -0
- veriscript-1.5.0/tests/test_document_date_flags.py +67 -0
- veriscript-1.5.0/tests/test_document_deva_lines.py +365 -0
- veriscript-1.5.0/tests/test_document_eval_smoke.py +221 -0
- veriscript-1.5.0/tests/test_document_export.py +150 -0
- veriscript-1.5.0/tests/test_document_export_unicode.py +106 -0
- veriscript-1.5.0/tests/test_document_flags.py +173 -0
- veriscript-1.5.0/tests/test_document_lang.py +78 -0
- veriscript-1.5.0/tests/test_document_layout.py +298 -0
- veriscript-1.5.0/tests/test_document_markdown.py +75 -0
- veriscript-1.5.0/tests/test_document_memory.py +237 -0
- veriscript-1.5.0/tests/test_document_multi_read.py +137 -0
- veriscript-1.5.0/tests/test_document_number_split.py +220 -0
- veriscript-1.5.0/tests/test_document_ocr_state.py +48 -0
- veriscript-1.5.0/tests/test_document_orientation.py +176 -0
- veriscript-1.5.0/tests/test_document_reconcile.py +98 -0
- veriscript-1.5.0/tests/test_document_restore.py +123 -0
- veriscript-1.5.0/tests/test_document_router.py +96 -0
- veriscript-1.5.0/tests/test_document_routing.py +95 -0
- veriscript-1.5.0/tests/test_document_verifier.py +166 -0
- veriscript-1.5.0/tests/test_error_taxonomy.py +62 -0
- veriscript-1.5.0/tests/test_eval_anchor.py +119 -0
- veriscript-1.5.0/tests/test_eval_freeze.py +108 -0
- veriscript-1.5.0/tests/test_eval_lines.py +99 -0
- veriscript-1.5.0/tests/test_eval_mixed.py +60 -0
- veriscript-1.5.0/tests/test_eval_sanity.py +79 -0
- veriscript-1.5.0/tests/test_frontend_shell.py +47 -0
- veriscript-1.5.0/tests/test_harvest_gate.py +92 -0
- veriscript-1.5.0/tests/test_harvest_textbooks.py +148 -0
- veriscript-1.5.0/tests/test_lexicon.py +78 -0
- veriscript-1.5.0/tests/test_lexicon_fetch.py +81 -0
- veriscript-1.5.0/tests/test_logging_setup.py +44 -0
- veriscript-1.5.0/tests/test_profile_textbook_scans.py +48 -0
- veriscript-1.5.0/tests/test_real_lines.py +205 -0
- veriscript-1.5.0/tests/test_smart_upscaler.py +152 -0
- veriscript-1.5.0/tests/test_sr_engine_tier_c_gate.py +52 -0
- veriscript-1.5.0/tests/test_vector_raster.py +197 -0
- veriscript-1.5.0/tests/test_webapp_api.py +496 -0
- veriscript-1.5.0/tests/test_webapp_security.py +196 -0
- veriscript-1.5.0/veriscript/__init__.py +9 -0
- veriscript-1.5.0/veriscript/__main__.py +4 -0
- veriscript-1.5.0/veriscript/branding.py +26 -0
- veriscript-1.5.0/veriscript/calibration.py +74 -0
- veriscript-1.5.0/veriscript/cli.py +473 -0
- veriscript-1.5.0/veriscript/core/__init__.py +1 -0
- veriscript-1.5.0/veriscript/core/data.py +1586 -0
- veriscript-1.5.0/veriscript/core/degradation.py +230 -0
- veriscript-1.5.0/veriscript/core/metrics.py +591 -0
- veriscript-1.5.0/veriscript/deva/__init__.py +1 -0
- veriscript-1.5.0/veriscript/deva/reader.py +218 -0
- veriscript-1.5.0/veriscript/document/__init__.py +1 -0
- veriscript-1.5.0/veriscript/document/confusions.py +142 -0
- veriscript-1.5.0/veriscript/document/corrections.py +378 -0
- veriscript-1.5.0/veriscript/document/export.py +388 -0
- veriscript-1.5.0/veriscript/document/layout.py +528 -0
- veriscript-1.5.0/veriscript/document/memory.py +561 -0
- veriscript-1.5.0/veriscript/document/ocr.py +1295 -0
- veriscript-1.5.0/veriscript/document/orientation.py +247 -0
- veriscript-1.5.0/veriscript/document/pipeline.py +367 -0
- veriscript-1.5.0/veriscript/document/reconcile.py +225 -0
- veriscript-1.5.0/veriscript/document/restore.py +222 -0
- veriscript-1.5.0/veriscript/document/router.py +76 -0
- veriscript-1.5.0/veriscript/document/verifier.py +132 -0
- veriscript-1.5.0/veriscript/lexicon.py +105 -0
- veriscript-1.5.0/veriscript/logging_setup.py +44 -0
- veriscript-1.5.0/veriscript/paths.py +13 -0
- veriscript-1.5.0/veriscript/photo/__init__.py +1 -0
- veriscript-1.5.0/veriscript/photo/hybrid.py +736 -0
- veriscript-1.5.0/veriscript/photo/rrdbnet.py +70 -0
- veriscript-1.5.0/veriscript/photo/sr_engine.py +390 -0
- veriscript-1.5.0/veriscript/photo/srvggnet.py +69 -0
- veriscript-1.5.0/veriscript/photo/tv_refinement.py +142 -0
- veriscript-1.5.0/veriscript/photo/upscaler.py +360 -0
- veriscript-1.5.0/veriscript.egg-info/PKG-INFO +342 -0
- veriscript-1.5.0/veriscript.egg-info/SOURCES.txt +111 -0
- veriscript-1.5.0/veriscript.egg-info/dependency_links.txt +1 -0
- veriscript-1.5.0/veriscript.egg-info/entry_points.txt +3 -0
- veriscript-1.5.0/veriscript.egg-info/requires.txt +30 -0
- veriscript-1.5.0/veriscript.egg-info/top_level.txt +2 -0
veriscript-1.5.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 DontHash
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,342 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: veriscript
|
|
3
|
+
Version: 1.5.0
|
|
4
|
+
Summary: Offline document restoration: photo/scan in, searchable PDF + honest review flags out
|
|
5
|
+
Author: DontHash
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2026 DontHash
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Project-URL: Homepage, https://github.com/DontHash/VeriScript
|
|
29
|
+
Project-URL: Live demo, https://veriscript.live
|
|
30
|
+
Project-URL: Issues, https://github.com/DontHash/VeriScript/issues
|
|
31
|
+
Keywords: ocr,devanagari,nepali,document-restore,searchable-pdf,offline
|
|
32
|
+
Classifier: Development Status :: 4 - Beta
|
|
33
|
+
Classifier: Environment :: Console
|
|
34
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
35
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
36
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
37
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
38
|
+
Classifier: Programming Language :: Python :: 3
|
|
39
|
+
Classifier: Topic :: Multimedia :: Graphics :: Graphics Conversion
|
|
40
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
41
|
+
Requires-Python: >=3.10
|
|
42
|
+
Description-Content-Type: text/markdown
|
|
43
|
+
License-File: LICENSE
|
|
44
|
+
Requires-Dist: numpy
|
|
45
|
+
Requires-Dist: opencv-python
|
|
46
|
+
Requires-Dist: scipy
|
|
47
|
+
Requires-Dist: scikit-image
|
|
48
|
+
Requires-Dist: onnxruntime
|
|
49
|
+
Requires-Dist: Pillow
|
|
50
|
+
Requires-Dist: jiwer
|
|
51
|
+
Requires-Dist: rapidocr==3.9.*
|
|
52
|
+
Requires-Dist: reportlab
|
|
53
|
+
Requires-Dist: pypdfium2
|
|
54
|
+
Provides-Extra: photo
|
|
55
|
+
Requires-Dist: torch; extra == "photo"
|
|
56
|
+
Provides-Extra: verifier
|
|
57
|
+
Requires-Dist: torch; extra == "verifier"
|
|
58
|
+
Requires-Dist: transformers; extra == "verifier"
|
|
59
|
+
Requires-Dist: huggingface_hub; extra == "verifier"
|
|
60
|
+
Provides-Extra: dev
|
|
61
|
+
Requires-Dist: pytest; extra == "dev"
|
|
62
|
+
Requires-Dist: httpx; extra == "dev"
|
|
63
|
+
Requires-Dist: matplotlib; extra == "dev"
|
|
64
|
+
Requires-Dist: scipy; extra == "dev"
|
|
65
|
+
Requires-Dist: torch; extra == "dev"
|
|
66
|
+
Requires-Dist: pyiqa; extra == "dev"
|
|
67
|
+
Requires-Dist: datasets; extra == "dev"
|
|
68
|
+
Requires-Dist: transformers; extra == "dev"
|
|
69
|
+
Requires-Dist: PySide6; extra == "dev"
|
|
70
|
+
Requires-Dist: google-genai; extra == "dev"
|
|
71
|
+
Dynamic: license-file
|
|
72
|
+
|
|
73
|
+
# VeriScript
|
|
74
|
+
|
|
75
|
+
**Offline document restoration.** A photo, scan or PDF goes in; a cleaned page
|
|
76
|
+
with a **searchable PDF**, overlay, transcript, **Markdown** and OCR JSON comes
|
|
77
|
+
out — with honest flags on the numbers it is not sure about. Devanagari
|
|
78
|
+
(Nepali/Hindi) first, English supported by the same pipeline.
|
|
79
|
+
|
|
80
|
+
Everything runs on your machine. No cloud calls, no telemetry, no account.
|
|
81
|
+
|
|
82
|
+
**Live demo:** <https://veriscript.live> — one page at
|
|
83
|
+
a time, no signup; demo uploads are deleted after the 60-minute retention
|
|
84
|
+
window.
|
|
85
|
+
|
|
86
|
+
```
|
|
87
|
+
photo/scan/PDF ──▶ orientation ──▶ layout ──▶ OCR ──▶ risk flags ──▶ searchable PDF
|
|
88
|
+
├─ overlay.png (review boxes)
|
|
89
|
+
├─ transcript.txt
|
|
90
|
+
├─ transcript.md (Markdown)
|
|
91
|
+
└─ ocr.json (tokens + queue)
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Before / after
|
|
95
|
+
|
|
96
|
+

|
|
97
|
+
|
|
98
|
+
Real pipeline output. **Green** = accepted, **amber** = uncertain, **red** =
|
|
99
|
+
digit conflict with the alternative reading kept. The letterpress page is from
|
|
100
|
+
heiDATA (doi:10.11588/data/EGOKEI, CC BY 4.0); the photo row is this
|
|
101
|
+
repository's own synthetic fixture under a simulated phone shadow. Regenerate
|
|
102
|
+
with `python scripts/make_demo_image.py`.
|
|
103
|
+
|
|
104
|
+
## Why it exists
|
|
105
|
+
|
|
106
|
+
- **It will not invent the numbers on your bill.** Digits are the highest-risk
|
|
107
|
+
tokens, so they are re-read, risk-ranked and put in a review queue instead of
|
|
108
|
+
being silently guessed. The queue is measured, not asserted
|
|
109
|
+
([EVALUATION.md](docs/EVALUATION.md)).
|
|
110
|
+
- **Devanagari is a first-class citizen, not a language pack.** Letterpress
|
|
111
|
+
books, modern government PDFs and phone photos each have frozen evaluation
|
|
112
|
+
sets with published numbers and confidence intervals.
|
|
113
|
+
- **Offline by default.** Privacy is the reason to use this instead of a cloud
|
|
114
|
+
lens: the pipeline runs with networking disabled.
|
|
115
|
+
|
|
116
|
+
## Install
|
|
117
|
+
|
|
118
|
+
From a source checkout (the package is not on PyPI yet):
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
pip install -r requirements.txt # full runtime (includes torch)
|
|
122
|
+
pip install -r requirements-dev.txt # + dev/eval tooling (optional)
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Or install the Python package itself — the document pipeline is torch-free
|
|
126
|
+
(`photo`/`verifier` are opt-in extras):
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
pip install -e . # document pipeline + web/CLI
|
|
130
|
+
pip install -e ".[photo]" # + photo upscaling (torch)
|
|
131
|
+
pip install -e ".[verifier]" # + the optional bodhan digit verifier
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Optional: [Tesseract 5](https://github.com/UB-Mannheim/tesseract/wiki) as a
|
|
135
|
+
second OCR backend (clean-scan fallback and the OSD rotation fallback).
|
|
136
|
+
|
|
137
|
+
## Use
|
|
138
|
+
|
|
139
|
+
### CLI
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
# Photo/scan in: searchable PDF + overlay + transcript + Markdown + JSON out
|
|
143
|
+
python cli.py --mode document --input page.jpg --output out/run1
|
|
144
|
+
|
|
145
|
+
# PDF in: first page by default; --max-pages 3 = first three pages
|
|
146
|
+
python cli.py --mode document --input scan.pdf --output out/run1 --max-pages 3
|
|
147
|
+
|
|
148
|
+
# PDF → Markdown: per-page .md always; --max-pages 0 adds one combined .md
|
|
149
|
+
python cli.py --mode document --input report.pdf --output out/report --max-pages 0
|
|
150
|
+
|
|
151
|
+
# Same CLI as a module / installed console script:
|
|
152
|
+
python -m veriscript --mode document --input report.pdf --output out/report
|
|
153
|
+
|
|
154
|
+
# Photo upscaling (shares the same CLI/app)
|
|
155
|
+
python cli.py --mode photo --input photos/ --output out/upscaled --scale 4
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Useful knobs (all default to the measured best configuration):
|
|
159
|
+
|
|
160
|
+
```
|
|
161
|
+
--ocr rapidocr|tesseract --lang en|ne|hi --deskew
|
|
162
|
+
--no-reading-order --rotate auto|off --repass-digits/--no-repass-digits
|
|
163
|
+
--digit-verifier off|bodhan (optional second-model digit check)
|
|
164
|
+
--no-pdf --no-overlay --no-txt --no-md
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
### Web app (the studio)
|
|
168
|
+
|
|
169
|
+
The single UI — FastAPI + SolidJS, running the same pipeline and outputs
|
|
170
|
+
(searchable PDF, transcript, **Markdown**, OCR JSON) as the CLI.
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
pip install -r webapp/requirements-web.txt # backend, on top of requirements.txt
|
|
174
|
+
cd webapp/frontend && npm install && npm run build && cd ../..
|
|
175
|
+
python webapp/server.py # http://127.0.0.1:8000
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
Hosted deployments (Docker, basic auth, HF Spaces/Render/Fly) are covered in
|
|
179
|
+
[docs/DEPLOY.md](docs/DEPLOY.md). The server binds to localhost by default;
|
|
180
|
+
the pipeline, models and any training artifacts stay server-side. The photo
|
|
181
|
+
upscaler is CLI-only (its weights are non-commercial); the web studio is
|
|
182
|
+
document-only — see [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md).
|
|
183
|
+
|
|
184
|
+
For JS/TS integrations there is a typed HTTP client in
|
|
185
|
+
[`packages/veriscript-client`](packages/veriscript-client) (npm:
|
|
186
|
+
`veriscript-client`; not published yet).
|
|
187
|
+
|
|
188
|
+
### Outputs (per page)
|
|
189
|
+
|
|
190
|
+
For an input `page.jpg`, every run writes:
|
|
191
|
+
|
|
192
|
+
| File | Contents |
|
|
193
|
+
|---|---|
|
|
194
|
+
| `page.pdf` | searchable PDF — cleaned image + invisible Unicode text layer (Devanagari included; bundled Mukta font), selectable/copyable |
|
|
195
|
+
| `page_overlay.png` | cleaned page with numbered review boxes for the flagged tokens |
|
|
196
|
+
| `page.txt` | reading-order plain text |
|
|
197
|
+
| `page.md` | **Markdown transcript** — the text plus the review queue as a table (this is the PDF → Markdown output) |
|
|
198
|
+
| `page.json` | tokens, risk flags, review queue, orientation/reading-order evidence |
|
|
199
|
+
| `report_pNNN.*` | per-page set for a multi-page PDF (`report_p000.pdf`, `report_p000.md`, …) |
|
|
200
|
+
| `report_combined.pdf` / `.txt` / `.md` | all pages joined with `--max-pages 0`; the combined Markdown has one `## Page N` section per page |
|
|
201
|
+
|
|
202
|
+
The web studio serves the same artifacts under canonical names
|
|
203
|
+
(`restored.png`, `searchable.pdf`, `transcript.txt`, `transcript.md`,
|
|
204
|
+
`ocr.json`, `combined.md`). `--no-md` (CLI) or the Markdown toggle (web)
|
|
205
|
+
skips the `.md` files.
|
|
206
|
+
|
|
207
|
+
## Measured behaviour
|
|
208
|
+
|
|
209
|
+
Numbers below come from hash-frozen evaluation sets with bootstrap 95% CIs
|
|
210
|
+
(full tables, provenance and rebuild commands: [docs/EVALUATION.md](docs/EVALUATION.md)).
|
|
211
|
+
No number in this repo is marketing.
|
|
212
|
+
|
|
213
|
+
| Capability | Measured |
|
|
214
|
+
|---|---|
|
|
215
|
+
| Orientation (EXIF + 0/90/180/270 via OCR evidence) | 16/16 synthetic; 0/30 false rotations on upright real scans; 90/90 rotated pages decided |
|
|
216
|
+
| Devanagari — modern government PDFs (raw engine) | CER **0.135 [0.097–0.186]**, exact-token recall 0.877 (n=41, Gemini-2.5-Pro anchor GT, engine-corroborated) |
|
|
217
|
+
| Devanagari — letterpress books (the honest bar) | CER 0.434 [0.387–0.481] raw; **0.373** under the default `auto` line reader (42/69 pages engaged, Appendix AE); exact-token recall 0.647 (n=69, human-corrected ALTO GT) |
|
|
218
|
+
| Devanagari — rendered fixtures | clean CER 0.030/0.028; degraded ne 0.094 (pass) / hi 0.170 (heavy fails) |
|
|
219
|
+
| Digit honesty — 2× re-pass (default ON for Devanagari) | review-queue digit recall@10 **0.88** letterpress / **0.38** PDFs (was 0.73 / 0.15), ≈ +0.4 s/page |
|
|
220
|
+
| Flags + calibration | `script_mismatch`, `invalid_sequence` (impossible combining sequence), isotonic `cal_conf`; ECE 0.82 → 0.30 on scans; optional `unknown_word` lexicon flag (needs `scripts/fetch_nepali_lexicon.py`) lifts the frozen-PDF review queue R@10 **0.148 → 0.175** with the digit queue unchanged |
|
|
221
|
+
| Optional digit verifier (`--digit-verifier bodhan`) | flags 0.71 recall / 0.81 precision; **opt-in** (cost gate failed: +8.3 s/page) |
|
|
222
|
+
| Multi-page PDF | `--max-pages 0` → every page + `<stem>_combined.pdf`/`.txt`; GUI "All pages" checkbox |
|
|
223
|
+
| Two-column reading order | arXiv set WER 0.870 → 0.268 (−69%) on split pages; >2 columns unsupported |
|
|
224
|
+
| Phone-photo proxy (not real photos) | CER 0.370 medium / 0.757 heavy (clean 0.135) — recognition, not honesty, is what breaks |
|
|
225
|
+
| Mixed-page router (opt-in) | restored text + untouched logos/photos (non-text PSNR 60–67 dB vs 18–23 dB naive); 1 invented token on photo texture → not auto-enabled |
|
|
226
|
+
| Detection on real pages | covers 97.5% of ALTO line boxes (area view) — recognition, not detection, is the bottleneck |
|
|
227
|
+
|
|
228
|
+
Known limitations are listed with their evidence in
|
|
229
|
+
[docs/EVALUATION.md](docs/EVALUATION.md) and the full research log in
|
|
230
|
+
[docs/PLAN.md](docs/PLAN.md). Since 2026-10-01 the image paths (CLI and web
|
|
231
|
+
studio) default to the `auto` line-reader gate — running text + aged paper
|
|
232
|
+
only. PDF inputs keep the engine reading, and the hosted document-only demo
|
|
233
|
+
ships no reader checkpoint, so it stays on the engine there.
|
|
234
|
+
|
|
235
|
+
## How it compares
|
|
236
|
+
|
|
237
|
+
Head-to-head on the frozen Devanagari hard-10 (the table-heavy hardest 10 of
|
|
238
|
+
the 41-page `nepali_pdf_v2` set, Gemini-2.5-Pro anchor GT; every arm scored by
|
|
239
|
+
the same metric code — canonical write-up in
|
|
240
|
+
[docs/BENCHMARK.md](docs/BENCHMARK.md), raw lab notes and reproduction
|
|
241
|
+
commands in [evals/bakeoff_results.md](evals/bakeoff_results.md)):
|
|
242
|
+
|
|
243
|
+
| Engine | CER (mean [95% CI]) | Median CER | bagCER | Speed (s/page) | Runs on CPU |
|
|
244
|
+
|---|---|---|---|---|---|
|
|
245
|
+
| Surya OCR 2 | 0.166 [0.100–0.258] | 0.107 | 0.360 | 172.0 | ✗ |
|
|
246
|
+
| Qwen3-VL-8B NF4 | 0.176 [0.129–0.235] | 0.141 | 0.369 | 152.7 | ✗ |
|
|
247
|
+
| **VeriScript pipeline (default)** | **0.255 [0.202–0.329]** | **0.227** | 0.376 | **7.3** | ✓ |
|
|
248
|
+
| raw RapidOCR engine | 0.530 [0.463–0.585] | 0.543 | 0.380 | 3.8 | ✓ |
|
|
249
|
+
| PaddleOCR 3.x full pipeline | 0.662 [0.585–0.713] | 0.687 | 0.400 | 21.8 | ✓ |
|
|
250
|
+
|
|
251
|
+
Speeds are per page on the harness hardware — ours, raw and Surya on an
|
|
252
|
+
RTX 2050 4 GB (Surya via llama.cpp/Vulkan), Qwen on a T4 (NF4), PaddleOCR on
|
|
253
|
+
**CPU**. Runs on CPU: ✓ = CPU-first (PaddleOCR was measured CPU-only here;
|
|
254
|
+
VeriScript and RapidOCR ship CPU-first); ✗ = GPU-class in practice — a GPU was
|
|
255
|
+
used and CPU would be far slower.
|
|
256
|
+
|
|
257
|
+
**Metrics.** **CER** (character error rate) is the edit distance between the
|
|
258
|
+
OCR output and the ground truth (insertions + deletions + substitutions)
|
|
259
|
+
divided by the number of ground-truth characters: order-sensitive, 0 is
|
|
260
|
+
perfect, lower is better. **bagCER** is the same character distance after
|
|
261
|
+
alphabetically sorting the tokens on both sides — it ignores reading order, so
|
|
262
|
+
it isolates glyph recognition from layout (a page with perfect words in the
|
|
263
|
+
wrong order still scores 0). **Median CER** is the middle page's CER, robust
|
|
264
|
+
to a single catastrophic page; **s/page** is seconds per page.
|
|
265
|
+
|
|
266
|
+
**Reading.** bagCER is tied across every arm (0.360–0.400): glyph recognition
|
|
267
|
+
is a commodity, and the page-CER spread is reading order, table structure and
|
|
268
|
+
digit handling — which is where this pipeline earns its keep (0.530 → 0.255
|
|
269
|
+
over the raw engine on the same frozen pages, while staying CPU-first,
|
|
270
|
+
offline and Apache-2.0). Surya 2 reads best by median at ~23× our page time
|
|
271
|
+
and OpenRAIL-M weights, and the GPU arms invent text the source never
|
|
272
|
+
contained (82–224 tokens over the 10 pages); only this stack ships an
|
|
273
|
+
alternative-reading digit review queue.
|
|
274
|
+
|
|
275
|
+
## Project layout
|
|
276
|
+
|
|
277
|
+
```
|
|
278
|
+
veriscript/ the product package
|
|
279
|
+
cli.py command-line entry point (document + photo)
|
|
280
|
+
core/ dataset IO + metrics (doc_data, doc_metrics), degradation
|
|
281
|
+
document/ pipeline: OCR, layout, orientation, restore, export, router, verifier
|
|
282
|
+
deva/ Devanagari line reader (CRNN+CTC), opt-in via --deva-lines
|
|
283
|
+
photo/ photo/vector stack (parked; CLI photo mode)
|
|
284
|
+
branding.py, logging_setup.py, lexicon.py, calibration.py
|
|
285
|
+
cli.py compatibility launcher for `python -m veriscript`
|
|
286
|
+
webapp/ the studio: FastAPI backend (vvweb/) + SolidJS frontend
|
|
287
|
+
deva_crnn/ training package for the line reader (repo tooling)
|
|
288
|
+
evals/harness/ evaluation harnesses (frozen sets, metrics, gates)
|
|
289
|
+
evals/manifests/ content-hash frozen evaluation sets
|
|
290
|
+
scripts/ data acquisition, training export, gates, release tooling
|
|
291
|
+
tests/ the test suite (pytest counts them; ~338)
|
|
292
|
+
docs/ architecture, evaluation, deployment, licensing, plan
|
|
293
|
+
```
|
|
294
|
+
|
|
295
|
+
## Docs
|
|
296
|
+
|
|
297
|
+
| Document | Contents |
|
|
298
|
+
|---|---|
|
|
299
|
+
| [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) | module map, data flow, extension points |
|
|
300
|
+
| [docs/EVALUATION.md](docs/EVALUATION.md) | frozen sets, metrics, how to reproduce every number |
|
|
301
|
+
| [docs/DEPLOY.md](docs/DEPLOY.md) | hosted web app (Docker, auth, spaces) |
|
|
302
|
+
| [docs/LICENSES.md](docs/LICENSES.md) | dependency + model license gate |
|
|
303
|
+
| [docs/RELEASE.md](docs/RELEASE.md) | release checklist, versioning, desktop packaging path |
|
|
304
|
+
| [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md) | web studio: audit, diet and phased plan |
|
|
305
|
+
| [docs/MARKET.md](docs/MARKET.md) | market landscape, competitors, positioning |
|
|
306
|
+
| [evals/bakeoff_results.md](evals/bakeoff_results.md) | head-to-head bake-off results on the frozen sets |
|
|
307
|
+
| [docs/PLAN.md](docs/PLAN.md) | the full research log and decision record |
|
|
308
|
+
|
|
309
|
+
## Tests
|
|
310
|
+
|
|
311
|
+
```bash
|
|
312
|
+
python -m pytest tests/ -q
|
|
313
|
+
```
|
|
314
|
+
|
|
315
|
+
The suite (≈336 collected) covers: OCR/export/routing/CLI/web, layout, orientation, language plumbing,
|
|
316
|
+
frozen-manifest and CI guards, GT-validity audits, error taxonomy, anchor
|
|
317
|
+
harness, queue metrics, Unicode-path IO, Devanagari line synthesis, CRNN
|
|
318
|
+
plumbing (height/width round-trip, beam search, cosine fine-tune), the
|
|
319
|
+
real-line mining gate, the opt-in line reader (merge geometry, table-rule
|
|
320
|
+
blocking, auto policy, graceful degradation, provenance), the lexicon
|
|
321
|
+
fetch/build + `unknown_word` flag mechanics, the textbook harvest parsers +
|
|
322
|
+
real-print scan probe, the CC-100 corpus builder + controlled sampler, the
|
|
323
|
+
Vertex wheel builder and the trainer GCS output path, the Devanagari
|
|
324
|
+
PDF text-layer extraction round-trips, the logging configuration +
|
|
325
|
+
server-error capture, and the Markdown export (per page and combined).
|
|
326
|
+
|
|
327
|
+
## Data, models and licensing
|
|
328
|
+
|
|
329
|
+
This repository contains **code and evaluation evidence only**: no training
|
|
330
|
+
datasets, no model checkpoints and no redistributed third-party corpora.
|
|
331
|
+
Evaluation manifests record hashes and provenance; the data itself stays local
|
|
332
|
+
(`data/`, `weights/` and `artifacts/` are git-ignored). The optional
|
|
333
|
+
Devanagari lexicon behind the `unknown_word` flag is built locally by
|
|
334
|
+
`scripts/fetch_nepali_lexicon.py` (Apache-2.0 + MIT sources) and is not
|
|
335
|
+
redistributed. Third-party license
|
|
336
|
+
obligations and the air-gap story are tracked in
|
|
337
|
+
[docs/LICENSES.md](docs/LICENSES.md).
|
|
338
|
+
|
|
339
|
+
MIT licensed — see [LICENSE](LICENSE). Third-party components keep their own
|
|
340
|
+
licenses (RapidOCR/PP-OCR Apache-2.0, reportlab/pypdfium2 permissive, PySide6
|
|
341
|
+
LGPL, UltraSharp weights CC-BY-NC-SA and therefore excluded from commercial
|
|
342
|
+
builds).
|
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
# VeriScript
|
|
2
|
+
|
|
3
|
+
**Offline document restoration.** A photo, scan or PDF goes in; a cleaned page
|
|
4
|
+
with a **searchable PDF**, overlay, transcript, **Markdown** and OCR JSON comes
|
|
5
|
+
out — with honest flags on the numbers it is not sure about. Devanagari
|
|
6
|
+
(Nepali/Hindi) first, English supported by the same pipeline.
|
|
7
|
+
|
|
8
|
+
Everything runs on your machine. No cloud calls, no telemetry, no account.
|
|
9
|
+
|
|
10
|
+
**Live demo:** <https://veriscript.live> — one page at
|
|
11
|
+
a time, no signup; demo uploads are deleted after the 60-minute retention
|
|
12
|
+
window.
|
|
13
|
+
|
|
14
|
+
```
|
|
15
|
+
photo/scan/PDF ──▶ orientation ──▶ layout ──▶ OCR ──▶ risk flags ──▶ searchable PDF
|
|
16
|
+
├─ overlay.png (review boxes)
|
|
17
|
+
├─ transcript.txt
|
|
18
|
+
├─ transcript.md (Markdown)
|
|
19
|
+
└─ ocr.json (tokens + queue)
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## Before / after
|
|
23
|
+
|
|
24
|
+

|
|
25
|
+
|
|
26
|
+
Real pipeline output. **Green** = accepted, **amber** = uncertain, **red** =
|
|
27
|
+
digit conflict with the alternative reading kept. The letterpress page is from
|
|
28
|
+
heiDATA (doi:10.11588/data/EGOKEI, CC BY 4.0); the photo row is this
|
|
29
|
+
repository's own synthetic fixture under a simulated phone shadow. Regenerate
|
|
30
|
+
with `python scripts/make_demo_image.py`.
|
|
31
|
+
|
|
32
|
+
## Why it exists
|
|
33
|
+
|
|
34
|
+
- **It will not invent the numbers on your bill.** Digits are the highest-risk
|
|
35
|
+
tokens, so they are re-read, risk-ranked and put in a review queue instead of
|
|
36
|
+
being silently guessed. The queue is measured, not asserted
|
|
37
|
+
([EVALUATION.md](docs/EVALUATION.md)).
|
|
38
|
+
- **Devanagari is a first-class citizen, not a language pack.** Letterpress
|
|
39
|
+
books, modern government PDFs and phone photos each have frozen evaluation
|
|
40
|
+
sets with published numbers and confidence intervals.
|
|
41
|
+
- **Offline by default.** Privacy is the reason to use this instead of a cloud
|
|
42
|
+
lens: the pipeline runs with networking disabled.
|
|
43
|
+
|
|
44
|
+
## Install
|
|
45
|
+
|
|
46
|
+
From a source checkout (the package is not on PyPI yet):
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install -r requirements.txt # full runtime (includes torch)
|
|
50
|
+
pip install -r requirements-dev.txt # + dev/eval tooling (optional)
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Or install the Python package itself — the document pipeline is torch-free
|
|
54
|
+
(`photo`/`verifier` are opt-in extras):
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pip install -e . # document pipeline + web/CLI
|
|
58
|
+
pip install -e ".[photo]" # + photo upscaling (torch)
|
|
59
|
+
pip install -e ".[verifier]" # + the optional bodhan digit verifier
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Optional: [Tesseract 5](https://github.com/UB-Mannheim/tesseract/wiki) as a
|
|
63
|
+
second OCR backend (clean-scan fallback and the OSD rotation fallback).
|
|
64
|
+
|
|
65
|
+
## Use
|
|
66
|
+
|
|
67
|
+
### CLI
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
# Photo/scan in: searchable PDF + overlay + transcript + Markdown + JSON out
|
|
71
|
+
python cli.py --mode document --input page.jpg --output out/run1
|
|
72
|
+
|
|
73
|
+
# PDF in: first page by default; --max-pages 3 = first three pages
|
|
74
|
+
python cli.py --mode document --input scan.pdf --output out/run1 --max-pages 3
|
|
75
|
+
|
|
76
|
+
# PDF → Markdown: per-page .md always; --max-pages 0 adds one combined .md
|
|
77
|
+
python cli.py --mode document --input report.pdf --output out/report --max-pages 0
|
|
78
|
+
|
|
79
|
+
# Same CLI as a module / installed console script:
|
|
80
|
+
python -m veriscript --mode document --input report.pdf --output out/report
|
|
81
|
+
|
|
82
|
+
# Photo upscaling (shares the same CLI/app)
|
|
83
|
+
python cli.py --mode photo --input photos/ --output out/upscaled --scale 4
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Useful knobs (all default to the measured best configuration):
|
|
87
|
+
|
|
88
|
+
```
|
|
89
|
+
--ocr rapidocr|tesseract --lang en|ne|hi --deskew
|
|
90
|
+
--no-reading-order --rotate auto|off --repass-digits/--no-repass-digits
|
|
91
|
+
--digit-verifier off|bodhan (optional second-model digit check)
|
|
92
|
+
--no-pdf --no-overlay --no-txt --no-md
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
### Web app (the studio)
|
|
96
|
+
|
|
97
|
+
The single UI — FastAPI + SolidJS, running the same pipeline and outputs
|
|
98
|
+
(searchable PDF, transcript, **Markdown**, OCR JSON) as the CLI.
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
pip install -r webapp/requirements-web.txt # backend, on top of requirements.txt
|
|
102
|
+
cd webapp/frontend && npm install && npm run build && cd ../..
|
|
103
|
+
python webapp/server.py # http://127.0.0.1:8000
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Hosted deployments (Docker, basic auth, HF Spaces/Render/Fly) are covered in
|
|
107
|
+
[docs/DEPLOY.md](docs/DEPLOY.md). The server binds to localhost by default;
|
|
108
|
+
the pipeline, models and any training artifacts stay server-side. The photo
|
|
109
|
+
upscaler is CLI-only (its weights are non-commercial); the web studio is
|
|
110
|
+
document-only — see [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md).
|
|
111
|
+
|
|
112
|
+
For JS/TS integrations there is a typed HTTP client in
|
|
113
|
+
[`packages/veriscript-client`](packages/veriscript-client) (npm:
|
|
114
|
+
`veriscript-client`; not published yet).
|
|
115
|
+
|
|
116
|
+
### Outputs (per page)
|
|
117
|
+
|
|
118
|
+
For an input `page.jpg`, every run writes:
|
|
119
|
+
|
|
120
|
+
| File | Contents |
|
|
121
|
+
|---|---|
|
|
122
|
+
| `page.pdf` | searchable PDF — cleaned image + invisible Unicode text layer (Devanagari included; bundled Mukta font), selectable/copyable |
|
|
123
|
+
| `page_overlay.png` | cleaned page with numbered review boxes for the flagged tokens |
|
|
124
|
+
| `page.txt` | reading-order plain text |
|
|
125
|
+
| `page.md` | **Markdown transcript** — the text plus the review queue as a table (this is the PDF → Markdown output) |
|
|
126
|
+
| `page.json` | tokens, risk flags, review queue, orientation/reading-order evidence |
|
|
127
|
+
| `report_pNNN.*` | per-page set for a multi-page PDF (`report_p000.pdf`, `report_p000.md`, …) |
|
|
128
|
+
| `report_combined.pdf` / `.txt` / `.md` | all pages joined with `--max-pages 0`; the combined Markdown has one `## Page N` section per page |
|
|
129
|
+
|
|
130
|
+
The web studio serves the same artifacts under canonical names
|
|
131
|
+
(`restored.png`, `searchable.pdf`, `transcript.txt`, `transcript.md`,
|
|
132
|
+
`ocr.json`, `combined.md`). `--no-md` (CLI) or the Markdown toggle (web)
|
|
133
|
+
skips the `.md` files.
|
|
134
|
+
|
|
135
|
+
## Measured behaviour
|
|
136
|
+
|
|
137
|
+
Numbers below come from hash-frozen evaluation sets with bootstrap 95% CIs
|
|
138
|
+
(full tables, provenance and rebuild commands: [docs/EVALUATION.md](docs/EVALUATION.md)).
|
|
139
|
+
No number in this repo is marketing.
|
|
140
|
+
|
|
141
|
+
| Capability | Measured |
|
|
142
|
+
|---|---|
|
|
143
|
+
| Orientation (EXIF + 0/90/180/270 via OCR evidence) | 16/16 synthetic; 0/30 false rotations on upright real scans; 90/90 rotated pages decided |
|
|
144
|
+
| Devanagari — modern government PDFs (raw engine) | CER **0.135 [0.097–0.186]**, exact-token recall 0.877 (n=41, Gemini-2.5-Pro anchor GT, engine-corroborated) |
|
|
145
|
+
| Devanagari — letterpress books (the honest bar) | CER 0.434 [0.387–0.481] raw; **0.373** under the default `auto` line reader (42/69 pages engaged, Appendix AE); exact-token recall 0.647 (n=69, human-corrected ALTO GT) |
|
|
146
|
+
| Devanagari — rendered fixtures | clean CER 0.030/0.028; degraded ne 0.094 (pass) / hi 0.170 (heavy fails) |
|
|
147
|
+
| Digit honesty — 2× re-pass (default ON for Devanagari) | review-queue digit recall@10 **0.88** letterpress / **0.38** PDFs (was 0.73 / 0.15), ≈ +0.4 s/page |
|
|
148
|
+
| Flags + calibration | `script_mismatch`, `invalid_sequence` (impossible combining sequence), isotonic `cal_conf`; ECE 0.82 → 0.30 on scans; optional `unknown_word` lexicon flag (needs `scripts/fetch_nepali_lexicon.py`) lifts the frozen-PDF review queue R@10 **0.148 → 0.175** with the digit queue unchanged |
|
|
149
|
+
| Optional digit verifier (`--digit-verifier bodhan`) | flags 0.71 recall / 0.81 precision; **opt-in** (cost gate failed: +8.3 s/page) |
|
|
150
|
+
| Multi-page PDF | `--max-pages 0` → every page + `<stem>_combined.pdf`/`.txt`; GUI "All pages" checkbox |
|
|
151
|
+
| Two-column reading order | arXiv set WER 0.870 → 0.268 (−69%) on split pages; >2 columns unsupported |
|
|
152
|
+
| Phone-photo proxy (not real photos) | CER 0.370 medium / 0.757 heavy (clean 0.135) — recognition, not honesty, is what breaks |
|
|
153
|
+
| Mixed-page router (opt-in) | restored text + untouched logos/photos (non-text PSNR 60–67 dB vs 18–23 dB naive); 1 invented token on photo texture → not auto-enabled |
|
|
154
|
+
| Detection on real pages | covers 97.5% of ALTO line boxes (area view) — recognition, not detection, is the bottleneck |
|
|
155
|
+
|
|
156
|
+
Known limitations are listed with their evidence in
|
|
157
|
+
[docs/EVALUATION.md](docs/EVALUATION.md) and the full research log in
|
|
158
|
+
[docs/PLAN.md](docs/PLAN.md). Since 2026-10-01 the image paths (CLI and web
|
|
159
|
+
studio) default to the `auto` line-reader gate — running text + aged paper
|
|
160
|
+
only. PDF inputs keep the engine reading, and the hosted document-only demo
|
|
161
|
+
ships no reader checkpoint, so it stays on the engine there.
|
|
162
|
+
|
|
163
|
+
## How it compares
|
|
164
|
+
|
|
165
|
+
Head-to-head on the frozen Devanagari hard-10 (the table-heavy hardest 10 of
|
|
166
|
+
the 41-page `nepali_pdf_v2` set, Gemini-2.5-Pro anchor GT; every arm scored by
|
|
167
|
+
the same metric code — canonical write-up in
|
|
168
|
+
[docs/BENCHMARK.md](docs/BENCHMARK.md), raw lab notes and reproduction
|
|
169
|
+
commands in [evals/bakeoff_results.md](evals/bakeoff_results.md)):
|
|
170
|
+
|
|
171
|
+
| Engine | CER (mean [95% CI]) | Median CER | bagCER | Speed (s/page) | Runs on CPU |
|
|
172
|
+
|---|---|---|---|---|---|
|
|
173
|
+
| Surya OCR 2 | 0.166 [0.100–0.258] | 0.107 | 0.360 | 172.0 | ✗ |
|
|
174
|
+
| Qwen3-VL-8B NF4 | 0.176 [0.129–0.235] | 0.141 | 0.369 | 152.7 | ✗ |
|
|
175
|
+
| **VeriScript pipeline (default)** | **0.255 [0.202–0.329]** | **0.227** | 0.376 | **7.3** | ✓ |
|
|
176
|
+
| raw RapidOCR engine | 0.530 [0.463–0.585] | 0.543 | 0.380 | 3.8 | ✓ |
|
|
177
|
+
| PaddleOCR 3.x full pipeline | 0.662 [0.585–0.713] | 0.687 | 0.400 | 21.8 | ✓ |
|
|
178
|
+
|
|
179
|
+
Speeds are per page on the harness hardware — ours, raw and Surya on an
|
|
180
|
+
RTX 2050 4 GB (Surya via llama.cpp/Vulkan), Qwen on a T4 (NF4), PaddleOCR on
|
|
181
|
+
**CPU**. Runs on CPU: ✓ = CPU-first (PaddleOCR was measured CPU-only here;
|
|
182
|
+
VeriScript and RapidOCR ship CPU-first); ✗ = GPU-class in practice — a GPU was
|
|
183
|
+
used and CPU would be far slower.
|
|
184
|
+
|
|
185
|
+
**Metrics.** **CER** (character error rate) is the edit distance between the
|
|
186
|
+
OCR output and the ground truth (insertions + deletions + substitutions)
|
|
187
|
+
divided by the number of ground-truth characters: order-sensitive, 0 is
|
|
188
|
+
perfect, lower is better. **bagCER** is the same character distance after
|
|
189
|
+
alphabetically sorting the tokens on both sides — it ignores reading order, so
|
|
190
|
+
it isolates glyph recognition from layout (a page with perfect words in the
|
|
191
|
+
wrong order still scores 0). **Median CER** is the middle page's CER, robust
|
|
192
|
+
to a single catastrophic page; **s/page** is seconds per page.
|
|
193
|
+
|
|
194
|
+
**Reading.** bagCER is tied across every arm (0.360–0.400): glyph recognition
|
|
195
|
+
is a commodity, and the page-CER spread is reading order, table structure and
|
|
196
|
+
digit handling — which is where this pipeline earns its keep (0.530 → 0.255
|
|
197
|
+
over the raw engine on the same frozen pages, while staying CPU-first,
|
|
198
|
+
offline and Apache-2.0). Surya 2 reads best by median at ~23× our page time
|
|
199
|
+
and OpenRAIL-M weights, and the GPU arms invent text the source never
|
|
200
|
+
contained (82–224 tokens over the 10 pages); only this stack ships an
|
|
201
|
+
alternative-reading digit review queue.
|
|
202
|
+
|
|
203
|
+
## Project layout
|
|
204
|
+
|
|
205
|
+
```
|
|
206
|
+
veriscript/ the product package
|
|
207
|
+
cli.py command-line entry point (document + photo)
|
|
208
|
+
core/ dataset IO + metrics (doc_data, doc_metrics), degradation
|
|
209
|
+
document/ pipeline: OCR, layout, orientation, restore, export, router, verifier
|
|
210
|
+
deva/ Devanagari line reader (CRNN+CTC), opt-in via --deva-lines
|
|
211
|
+
photo/ photo/vector stack (parked; CLI photo mode)
|
|
212
|
+
branding.py, logging_setup.py, lexicon.py, calibration.py
|
|
213
|
+
cli.py compatibility launcher for `python -m veriscript`
|
|
214
|
+
webapp/ the studio: FastAPI backend (vvweb/) + SolidJS frontend
|
|
215
|
+
deva_crnn/ training package for the line reader (repo tooling)
|
|
216
|
+
evals/harness/ evaluation harnesses (frozen sets, metrics, gates)
|
|
217
|
+
evals/manifests/ content-hash frozen evaluation sets
|
|
218
|
+
scripts/ data acquisition, training export, gates, release tooling
|
|
219
|
+
tests/ the test suite (pytest counts them; ~338)
|
|
220
|
+
docs/ architecture, evaluation, deployment, licensing, plan
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
## Docs
|
|
224
|
+
|
|
225
|
+
| Document | Contents |
|
|
226
|
+
|---|---|
|
|
227
|
+
| [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) | module map, data flow, extension points |
|
|
228
|
+
| [docs/EVALUATION.md](docs/EVALUATION.md) | frozen sets, metrics, how to reproduce every number |
|
|
229
|
+
| [docs/DEPLOY.md](docs/DEPLOY.md) | hosted web app (Docker, auth, spaces) |
|
|
230
|
+
| [docs/LICENSES.md](docs/LICENSES.md) | dependency + model license gate |
|
|
231
|
+
| [docs/RELEASE.md](docs/RELEASE.md) | release checklist, versioning, desktop packaging path |
|
|
232
|
+
| [docs/PLAN_WEB_FULL.md](docs/PLAN_WEB_FULL.md) | web studio: audit, diet and phased plan |
|
|
233
|
+
| [docs/MARKET.md](docs/MARKET.md) | market landscape, competitors, positioning |
|
|
234
|
+
| [evals/bakeoff_results.md](evals/bakeoff_results.md) | head-to-head bake-off results on the frozen sets |
|
|
235
|
+
| [docs/PLAN.md](docs/PLAN.md) | the full research log and decision record |
|
|
236
|
+
|
|
237
|
+
## Tests
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
python -m pytest tests/ -q
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
The suite (≈336 collected) covers: OCR/export/routing/CLI/web, layout, orientation, language plumbing,
|
|
244
|
+
frozen-manifest and CI guards, GT-validity audits, error taxonomy, anchor
|
|
245
|
+
harness, queue metrics, Unicode-path IO, Devanagari line synthesis, CRNN
|
|
246
|
+
plumbing (height/width round-trip, beam search, cosine fine-tune), the
|
|
247
|
+
real-line mining gate, the opt-in line reader (merge geometry, table-rule
|
|
248
|
+
blocking, auto policy, graceful degradation, provenance), the lexicon
|
|
249
|
+
fetch/build + `unknown_word` flag mechanics, the textbook harvest parsers +
|
|
250
|
+
real-print scan probe, the CC-100 corpus builder + controlled sampler, the
|
|
251
|
+
Vertex wheel builder and the trainer GCS output path, the Devanagari
|
|
252
|
+
PDF text-layer extraction round-trips, the logging configuration +
|
|
253
|
+
server-error capture, and the Markdown export (per page and combined).
|
|
254
|
+
|
|
255
|
+
## Data, models and licensing
|
|
256
|
+
|
|
257
|
+
This repository contains **code and evaluation evidence only**: no training
|
|
258
|
+
datasets, no model checkpoints and no redistributed third-party corpora.
|
|
259
|
+
Evaluation manifests record hashes and provenance; the data itself stays local
|
|
260
|
+
(`data/`, `weights/` and `artifacts/` are git-ignored). The optional
|
|
261
|
+
Devanagari lexicon behind the `unknown_word` flag is built locally by
|
|
262
|
+
`scripts/fetch_nepali_lexicon.py` (Apache-2.0 + MIT sources) and is not
|
|
263
|
+
redistributed. Third-party license
|
|
264
|
+
obligations and the air-gap story are tracked in
|
|
265
|
+
[docs/LICENSES.md](docs/LICENSES.md).
|
|
266
|
+
|
|
267
|
+
MIT licensed — see [LICENSE](LICENSE). Third-party components keep their own
|
|
268
|
+
licenses (RapidOCR/PP-OCR Apache-2.0, reportlab/pypdfium2 permissive, PySide6
|
|
269
|
+
LGPL, UltraSharp weights CC-BY-NC-SA and therefore excluded from commercial
|
|
270
|
+
builds).
|