paradox2 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- paradox2/__init__.py +7 -0
- paradox2/__main__.py +5 -0
- paradox2/api.py +320 -0
- paradox2/cli.py +113 -0
- paradox2/config.py +33 -0
- paradox2/doctor.py +74 -0
- paradox2/engines/__init__.py +0 -0
- paradox2/engines/_gpu_worker.py +214 -0
- paradox2/engines/_html_table.py +68 -0
- paradox2/engines/_lazy.py +53 -0
- paradox2/engines/_row_confidence.py +166 -0
- paradox2/engines/doc_classifier.py +67 -0
- paradox2/engines/formula_latex.py +38 -0
- paradox2/engines/handwriting_trocr.py +49 -0
- paradox2/engines/kie_heuristic.py +366 -0
- paradox2/engines/signature_detr.py +43 -0
- paradox2/engines/table_azure_di.py +161 -0
- paradox2/engines/table_detector.py +130 -0
- paradox2/engines/table_ppstructure.py +256 -0
- paradox2/engines/table_rapidai.py +178 -0
- paradox2/engines/table_vlm.py +207 -0
- paradox2/exporters/__init__.py +0 -0
- paradox2/exporters/markdown.py +99 -0
- paradox2/exporters/structured.py +129 -0
- paradox2/formats/__init__.py +0 -0
- paradox2/formats/archive_adapter.py +176 -0
- paradox2/formats/detector.py +150 -0
- paradox2/formats/doc_adapter.py +55 -0
- paradox2/formats/docx_adapter.py +38 -0
- paradox2/formats/eml_adapter.py +63 -0
- paradox2/formats/image_adapter.py +19 -0
- paradox2/formats/open_document_adapter.py +71 -0
- paradox2/formats/pptx_adapter.py +33 -0
- paradox2/formats/render_bundle.py +95 -0
- paradox2/formats/rtf_adapter.py +46 -0
- paradox2/formats/text_adapters.py +48 -0
- paradox2/formats/web_adapter.py +64 -0
- paradox2/formats/xls_adapter.py +43 -0
- paradox2/formats/xlsb_adapter.py +35 -0
- paradox2/formats/xlsx_adapter.py +41 -0
- paradox2/ir/__init__.py +0 -0
- paradox2/pipeline/__init__.py +0 -0
- paradox2/pipeline/annotations.py +192 -0
- paradox2/pipeline/classify.py +113 -0
- paradox2/pipeline/digital_fast.py +220 -0
- paradox2/pipeline/feature_flags.py +78 -0
- paradox2/pipeline/language_detect.py +43 -0
- paradox2/pipeline/links.py +103 -0
- paradox2/pipeline/orientation.py +87 -0
- paradox2/pipeline/reading_order.py +124 -0
- paradox2/pipeline/scanned.py +488 -0
- paradox2/runtime/__init__.py +0 -0
- paradox2/runtime/backend_detect.py +49 -0
- paradox2/runtime/guardrails.py +28 -0
- paradox2/runtime/machine_state.py +380 -0
- paradox2/runtime/page_pool.py +159 -0
- paradox2/runtime/resource_detect.py +339 -0
- paradox2/service/__init__.py +0 -0
- paradox2/service/client.py +73 -0
- paradox2/service/protocol.py +61 -0
- paradox2/service/server.py +81 -0
- paradox2/tables/__init__.py +0 -0
- paradox2/tables/cv_tables.py +106 -0
- paradox2/tables/geometry.py +84 -0
- paradox2/tables/line_tables.py +320 -0
- paradox2/tables/scanned.py +322 -0
- paradox2/tables/scanned_router.py +574 -0
- paradox2/tables/structure.py +146 -0
- paradox2-0.1.0.dist-info/METADATA +227 -0
- paradox2-0.1.0.dist-info/RECORD +73 -0
- paradox2-0.1.0.dist-info/WHEEL +4 -0
- paradox2-0.1.0.dist-info/entry_points.txt +2 -0
- paradox2-0.1.0.dist-info/licenses/LICENSE +34 -0
paradox2/__init__.py
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Public package surface for the from-scratch extractor."""
|
|
2
|
+
|
|
3
|
+
from paradox2.api import extract, extract_kie, extract_tables, extract_text, read
|
|
4
|
+
from paradox2.doctor import doctor
|
|
5
|
+
from paradox2.runtime.machine_state import warmup
|
|
6
|
+
|
|
7
|
+
__all__ = ["extract", "extract_kie", "extract_tables", "extract_text", "read", "doctor", "warmup"]
|
paradox2/__main__.py
ADDED
paradox2/api.py
ADDED
|
@@ -0,0 +1,320 @@
|
|
|
1
|
+
"""Public facade — dispatch only, no business logic.
|
|
2
|
+
|
|
3
|
+
The old api.py grew to 2,783 lines by letting business logic (add-on
|
|
4
|
+
dispatch, backend resolution details, overlay rendering) live in the
|
|
5
|
+
facade itself. This file must stay thin: routing to pipeline.* and
|
|
6
|
+
attaching tables. If this file starts approaching 100+ lines, that's a
|
|
7
|
+
signal work leaked out of pipeline/ and should be pushed back down
|
|
8
|
+
(see docs/PLAN.md Phase 2).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import pymupdf
|
|
14
|
+
import re
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from collections.abc import Sequence
|
|
17
|
+
|
|
18
|
+
from paradox2.pipeline import digital_fast, scanned
|
|
19
|
+
from paradox2.pipeline.digital_fast import PageResult
|
|
20
|
+
from paradox2.tables.line_tables import detect_table_bboxes
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _page_selection(pages) -> set[int] | None:
|
|
24
|
+
if pages is None:
|
|
25
|
+
return None
|
|
26
|
+
if isinstance(pages, int):
|
|
27
|
+
return {pages - 1}
|
|
28
|
+
if isinstance(pages, str):
|
|
29
|
+
selected: set[int] = set()
|
|
30
|
+
for part in pages.replace(" ", "").split(","):
|
|
31
|
+
if not part:
|
|
32
|
+
continue
|
|
33
|
+
if "-" in part:
|
|
34
|
+
start, end = part.split("-", 1)
|
|
35
|
+
selected.update(range(int(start) - 1, int(end)))
|
|
36
|
+
else:
|
|
37
|
+
selected.add(int(part) - 1)
|
|
38
|
+
return selected
|
|
39
|
+
return {int(page) - 1 for page in pages}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def extract(
|
|
43
|
+
path: str,
|
|
44
|
+
backend: str = "auto",
|
|
45
|
+
language: str = "latin",
|
|
46
|
+
*,
|
|
47
|
+
pages: Sequence[int] | str | int | None = None,
|
|
48
|
+
feature: str | list[str] | None = None,
|
|
49
|
+
fields: bool = False,
|
|
50
|
+
include_links: bool = False,
|
|
51
|
+
include_annotations: bool = False,
|
|
52
|
+
detect_language: bool = False,
|
|
53
|
+
output_format: str = "dict",
|
|
54
|
+
mode: str | None = None,
|
|
55
|
+
config=None,
|
|
56
|
+
classify: bool = False,
|
|
57
|
+
handwriting: bool = False,
|
|
58
|
+
signatures: bool = False,
|
|
59
|
+
formulas: bool = False,
|
|
60
|
+
marks: bool = False,
|
|
61
|
+
detect_formulas: bool = False,
|
|
62
|
+
detect_signatures: bool = False,
|
|
63
|
+
page_workers: int = 1,
|
|
64
|
+
):
|
|
65
|
+
"""Extract every page of a PDF. Digital pages use the zero-ML fast path
|
|
66
|
+
(with vector-line table detection attached); scanned pages route to the
|
|
67
|
+
GPU OCR pipeline. Each page decides independently — a mixed digital+
|
|
68
|
+
scanned document is handled correctly per-page.
|
|
69
|
+
|
|
70
|
+
page_workers>1 parallelizes only the scanned pages across a
|
|
71
|
+
ProcessPoolExecutor (see pipeline.scanned.extract_scanned_parallel for
|
|
72
|
+
the measured numbers: ~1.9x on GPU, ~2x on CPU vs sequential, on a
|
|
73
|
+
12-page real document). Digital pages stay on the sequential zero-ML
|
|
74
|
+
path regardless — they're already fast enough that process-spawn
|
|
75
|
+
overhead would not pay for itself."""
|
|
76
|
+
min_native_text_chars = digital_fast._MIN_NATIVE_TEXT_CHARS
|
|
77
|
+
scanned_dpi = 200 # matches scanned.py's own _page_to_image default
|
|
78
|
+
if config is not None:
|
|
79
|
+
# Config is an explicit convenience object; direct keyword arguments
|
|
80
|
+
# remain the most local override for callers that need one-off changes.
|
|
81
|
+
backend = getattr(config, "backend", backend)
|
|
82
|
+
language = getattr(config, "language", language)
|
|
83
|
+
include_links = include_links or bool(getattr(config, "include_links", False))
|
|
84
|
+
include_annotations = include_annotations or bool(getattr(config, "include_annotations", False))
|
|
85
|
+
detect_language = detect_language or bool(getattr(config, "detect_language", False))
|
|
86
|
+
# Paradox-pdf 0.9.2 parity knob (was "scan_text_threshold") - only
|
|
87
|
+
# overrides the digital/scanned routing decision when a caller
|
|
88
|
+
# explicitly sets it via ExtractConfig; every existing caller keeps
|
|
89
|
+
# today's behavior unchanged (default matches the module constant).
|
|
90
|
+
min_native_text_chars = getattr(config, "scan_text_threshold", min_native_text_chars)
|
|
91
|
+
# ExtractConfig.dpi existed but was never read here - only affects scanned-OCR render.
|
|
92
|
+
scanned_dpi = getattr(config, "dpi", scanned_dpi)
|
|
93
|
+
|
|
94
|
+
# Keep non-PDF adapters behind the same public facade. They are already
|
|
95
|
+
# normalized to PageResult by formats.detector and never pay PDF/OCR costs.
|
|
96
|
+
if Path(path).suffix.lower() != ".pdf":
|
|
97
|
+
from paradox2.formats.detector import detect_and_extract
|
|
98
|
+
from paradox2.exporters.structured import select_feature, serialize
|
|
99
|
+
|
|
100
|
+
result = detect_and_extract(path, backend=backend, language=language)
|
|
101
|
+
if feature is None and output_format == "dict":
|
|
102
|
+
return result
|
|
103
|
+
value = select_feature(result, feature or "all")
|
|
104
|
+
return serialize(value, output_format)
|
|
105
|
+
|
|
106
|
+
from paradox2.pipeline.feature_flags import resolve_requested_features
|
|
107
|
+
|
|
108
|
+
flags = resolve_requested_features(
|
|
109
|
+
feature=feature, fields=fields, include_links=include_links,
|
|
110
|
+
include_annotations=include_annotations, detect_language=detect_language,
|
|
111
|
+
mode=mode, classify=classify, handwriting=handwriting, signatures=signatures,
|
|
112
|
+
formulas=formulas, marks=marks, detect_formulas=detect_formulas,
|
|
113
|
+
detect_signatures=detect_signatures,
|
|
114
|
+
)
|
|
115
|
+
selected = _page_selection(pages)
|
|
116
|
+
|
|
117
|
+
doc = pymupdf.open(path)
|
|
118
|
+
digital_fast.reject_if_encrypted(doc, path)
|
|
119
|
+
try:
|
|
120
|
+
precomputed_scanned: dict[int, PageResult] = {}
|
|
121
|
+
page_dicts: dict[int, dict] = {}
|
|
122
|
+
if page_workers > 1:
|
|
123
|
+
precomputed_scanned, page_dicts = scanned.precompute_scanned_pages(
|
|
124
|
+
path, doc, selected, backend, language, page_workers, min_native_text_chars, scanned_dpi
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
results: list[PageResult] = []
|
|
128
|
+
for page in doc:
|
|
129
|
+
if selected is not None and page.number not in selected:
|
|
130
|
+
continue
|
|
131
|
+
# Single get_text("dict") call, reused for both the has-text
|
|
132
|
+
# routing check and (on the digital path) the real extraction —
|
|
133
|
+
# was two separate PyMuPDF text-extraction passes per digital
|
|
134
|
+
# page (has_text_layer's get_text("words") + extract_page's own
|
|
135
|
+
# get_text("dict")), pure duplicate work with no benefit. Reuse
|
|
136
|
+
# the pre-pass's cached dict when page_workers>1 already parsed
|
|
137
|
+
# this page above, instead of parsing it a second time.
|
|
138
|
+
raw = page_dicts.get(page.number)
|
|
139
|
+
if raw is None:
|
|
140
|
+
raw = page.get_text("dict")
|
|
141
|
+
has_text = digital_fast.raw_has_text(raw, min_native_text_chars)
|
|
142
|
+
if has_text:
|
|
143
|
+
result = digital_fast.extract_page(page, raw=raw)
|
|
144
|
+
if flags.tables:
|
|
145
|
+
# Stroke-line extraction is expensive on vector-dense
|
|
146
|
+
# pages (verified: ~20s/call on a real page with
|
|
147
|
+
# 29,121 drawings) - compute once and share it between
|
|
148
|
+
# bbox detection and each bbox's structure extraction,
|
|
149
|
+
# instead of recomputing per call.
|
|
150
|
+
from paradox2.tables.line_tables import _extract_stroke_lines_from_drawings
|
|
151
|
+
|
|
152
|
+
horiz_segs, vert_segs = _extract_stroke_lines_from_drawings(page)
|
|
153
|
+
result.tables = detect_table_bboxes(page, horiz=horiz_segs, vert=vert_segs)
|
|
154
|
+
if result.tables:
|
|
155
|
+
from paradox2.tables.structure import extract_table_structured
|
|
156
|
+
|
|
157
|
+
for bbox in result.tables:
|
|
158
|
+
result.structured_tables.extend(
|
|
159
|
+
extract_table_structured(
|
|
160
|
+
page, bbox, horiz_segs=horiz_segs, vert_segs=vert_segs
|
|
161
|
+
)
|
|
162
|
+
)
|
|
163
|
+
else:
|
|
164
|
+
if page.number in precomputed_scanned:
|
|
165
|
+
result = precomputed_scanned[page.number]
|
|
166
|
+
else:
|
|
167
|
+
result = scanned.extract_page(page, backend=backend, language=language, dpi=scanned_dpi)
|
|
168
|
+
if flags.tables:
|
|
169
|
+
from paradox2.tables.scanned_router import route_scanned_tables
|
|
170
|
+
|
|
171
|
+
result.structured_tables = route_scanned_tables(page, result.blocks)
|
|
172
|
+
|
|
173
|
+
if flags.fields:
|
|
174
|
+
from paradox2.engines.kie_heuristic import (
|
|
175
|
+
detect_grid_regions,
|
|
176
|
+
extract_key_values,
|
|
177
|
+
extract_key_values_spatial,
|
|
178
|
+
looks_like_form,
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
if has_text:
|
|
182
|
+
words = page.get_text("words")
|
|
183
|
+
lines = page.get_text().splitlines()
|
|
184
|
+
if looks_like_form(lines) or any(":" in line for line in lines):
|
|
185
|
+
result.fields = extract_key_values_spatial(
|
|
186
|
+
words, table_boxes=detect_grid_regions(words)
|
|
187
|
+
)
|
|
188
|
+
else:
|
|
189
|
+
result.fields = extract_key_values([b.text for b in result.blocks])
|
|
190
|
+
|
|
191
|
+
if flags.links:
|
|
192
|
+
from paradox2.pipeline.links import extract_page_hyperlinks
|
|
193
|
+
|
|
194
|
+
result.links = extract_page_hyperlinks(page)
|
|
195
|
+
if flags.annotations:
|
|
196
|
+
from paradox2.pipeline.annotations import extract_page_annotations
|
|
197
|
+
|
|
198
|
+
result.annotations = extract_page_annotations(page)
|
|
199
|
+
if flags.handwriting or flags.signatures or flags.formulas:
|
|
200
|
+
_apply_specialists(
|
|
201
|
+
page, result,
|
|
202
|
+
handwriting=flags.handwriting,
|
|
203
|
+
signatures=flags.signatures,
|
|
204
|
+
formulas=flags.formulas,
|
|
205
|
+
)
|
|
206
|
+
if flags.marks:
|
|
207
|
+
result.marks = [
|
|
208
|
+
{"bbox": list(block.bbox), "text": block.text,
|
|
209
|
+
"bold": block.bold, "italic": block.italic}
|
|
210
|
+
for block in result.blocks if block.bold or block.italic
|
|
211
|
+
]
|
|
212
|
+
if flags.language:
|
|
213
|
+
from paradox2.pipeline.language_detect import detect_text_language
|
|
214
|
+
|
|
215
|
+
result.language = detect_text_language(result.text)
|
|
216
|
+
result.metadata = {"page_number": page.number + 1, "source": str(path)}
|
|
217
|
+
results.append(result)
|
|
218
|
+
if flags.classify and results:
|
|
219
|
+
from paradox2.engines.doc_classifier import get_engine
|
|
220
|
+
|
|
221
|
+
label, confidence = get_engine().classify("\n".join(p.text for p in results))
|
|
222
|
+
classification = {"label": label, "confidence": confidence}
|
|
223
|
+
for result in results:
|
|
224
|
+
result.classification = classification
|
|
225
|
+
# Default remains the efficient, geometry-preserving IR used by the
|
|
226
|
+
# service and existing callers. Feature/output requests are pure
|
|
227
|
+
# projections: extraction is never repeated.
|
|
228
|
+
if feature is None:
|
|
229
|
+
if output_format == "dict":
|
|
230
|
+
return results
|
|
231
|
+
if output_format == "markdown":
|
|
232
|
+
from paradox2.exporters.structured import serialize
|
|
233
|
+
|
|
234
|
+
return serialize(results, "markdown")
|
|
235
|
+
from paradox2.exporters.structured import select_feature, serialize
|
|
236
|
+
|
|
237
|
+
value = select_feature(results, feature or "all")
|
|
238
|
+
return serialize(value, output_format)
|
|
239
|
+
finally:
|
|
240
|
+
doc.close()
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def extract_kie(path: str) -> list[list[dict]]:
|
|
244
|
+
"""Per-page key-value extraction. Only meaningful on digital, form-like
|
|
245
|
+
pages (invoices/forms) - scanned pages and prose pages return []. Kept
|
|
246
|
+
separate from extract() rather than a feature= flag: KIE's output shape
|
|
247
|
+
(list of key-value dicts) doesn't fit PageResult, forcing it in would
|
|
248
|
+
be the wrong abstraction just to have one entry point."""
|
|
249
|
+
from paradox2.engines.kie_heuristic import (
|
|
250
|
+
detect_grid_regions,
|
|
251
|
+
extract_key_values_spatial,
|
|
252
|
+
looks_like_form,
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
doc = pymupdf.open(path)
|
|
256
|
+
try:
|
|
257
|
+
results: list[list[dict]] = []
|
|
258
|
+
for page in doc:
|
|
259
|
+
if not digital_fast.has_text_layer(page):
|
|
260
|
+
results.append([])
|
|
261
|
+
continue
|
|
262
|
+
words = page.get_text("words")
|
|
263
|
+
lines = page.get_text().splitlines()
|
|
264
|
+
if not looks_like_form(lines):
|
|
265
|
+
results.append([])
|
|
266
|
+
continue
|
|
267
|
+
regions = detect_grid_regions(words)
|
|
268
|
+
results.append(extract_key_values_spatial(words, table_boxes=regions))
|
|
269
|
+
return results
|
|
270
|
+
finally:
|
|
271
|
+
doc.close()
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _render_crop(page, bbox=None, dpi: int = 180):
|
|
275
|
+
"""Render a page/crop once for opt-in vision specialists."""
|
|
276
|
+
pix = page.get_pixmap(dpi=dpi, clip=pymupdf.Rect(bbox) if bbox else None, alpha=False)
|
|
277
|
+
from PIL import Image
|
|
278
|
+
return Image.frombytes("RGB", [pix.width, pix.height], pix.samples)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _apply_specialists(page, result: PageResult, *, handwriting: bool, signatures: bool, formulas: bool):
|
|
282
|
+
"""Run explicitly requested heavy models; never touched on the hot path."""
|
|
283
|
+
scale = 72.0 / 180.0
|
|
284
|
+
if signatures:
|
|
285
|
+
from paradox2.engines.signature_detr import get_engine
|
|
286
|
+
image = _render_crop(page)
|
|
287
|
+
for item in get_engine().detect(image):
|
|
288
|
+
item["bbox"] = [round(v * scale, 2) for v in item["bbox"]]
|
|
289
|
+
item["page"] = result.page_number
|
|
290
|
+
result.signatures.append(item)
|
|
291
|
+
if handwriting:
|
|
292
|
+
from paradox2.engines.handwriting_trocr import get_engine
|
|
293
|
+
engine = get_engine()
|
|
294
|
+
# TrOCR is line-oriented; use extracted blocks as tight crops and cap
|
|
295
|
+
# work so an accidental request cannot turn into an unbounded scan.
|
|
296
|
+
for block in result.blocks[:64]:
|
|
297
|
+
text = engine.process(_render_crop(page, block.bbox))
|
|
298
|
+
if text:
|
|
299
|
+
result.handwriting.append({"bbox": list(block.bbox), "text": text})
|
|
300
|
+
if formulas:
|
|
301
|
+
from paradox2.engines.formula_latex import image_to_latex
|
|
302
|
+
for block in result.blocks:
|
|
303
|
+
if not re.search(r"(?:=|∑|∫|√|\^|_[{\d])", block.text):
|
|
304
|
+
continue
|
|
305
|
+
latex = image_to_latex(_render_crop(page, block.bbox))
|
|
306
|
+
if latex:
|
|
307
|
+
result.formulas.append({"bbox": list(block.bbox), "text": block.text, "latex": latex})
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def read(path: str, **kwargs):
|
|
311
|
+
"""Compatibility alias for universal multi-format dispatch."""
|
|
312
|
+
return extract(path, **kwargs)
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def extract_text(path: str, **kwargs) -> str:
|
|
316
|
+
return extract(path, feature="text", **kwargs)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def extract_tables(path: str, **kwargs):
|
|
320
|
+
return extract(path, feature="tables", **kwargs)
|
paradox2/cli.py
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""paradox2 CLI: one-shot extraction, and serve/client subcommands wrapping
|
|
2
|
+
service/ (the persistent worker). argparse, stdlib only — no new deps for
|
|
3
|
+
a CLI wrapper around already-built pipeline/service code.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import argparse
|
|
9
|
+
import json
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
from paradox2.service.protocol import DEFAULT_SOCKET_PATH
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _cmd_extract(args: argparse.Namespace) -> int:
|
|
16
|
+
from paradox2.formats.detector import detect_and_extract
|
|
17
|
+
from paradox2.exporters.structured import page_to_dict, select_feature, serialize
|
|
18
|
+
|
|
19
|
+
pages = detect_and_extract(args.path, backend=args.backend, language=args.language)
|
|
20
|
+
|
|
21
|
+
if args.feature:
|
|
22
|
+
value = select_feature(pages, args.feature.split(","))
|
|
23
|
+
print(serialize(value, args.format))
|
|
24
|
+
elif args.format == "json":
|
|
25
|
+
print(json.dumps([page_to_dict(p) for p in pages], ensure_ascii=False))
|
|
26
|
+
else:
|
|
27
|
+
from paradox2.exporters.markdown import to_markdown
|
|
28
|
+
|
|
29
|
+
print(to_markdown(pages))
|
|
30
|
+
return 0
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _cmd_serve(args: argparse.Namespace) -> int:
|
|
34
|
+
from paradox2.service.server import serve
|
|
35
|
+
|
|
36
|
+
serve(args.socket)
|
|
37
|
+
return 0
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _cmd_client(args: argparse.Namespace) -> int:
|
|
41
|
+
from paradox2.service.client import extract, is_server_running
|
|
42
|
+
|
|
43
|
+
if not is_server_running(args.socket):
|
|
44
|
+
print(f"error: no paradox2 service running at {args.socket} "
|
|
45
|
+
f"(start one with 'paradox2 serve')", file=sys.stderr)
|
|
46
|
+
return 1
|
|
47
|
+
|
|
48
|
+
pages = extract(args.path, backend=args.backend, language=args.language, socket_path=args.socket)
|
|
49
|
+
for p in pages:
|
|
50
|
+
print(p.text)
|
|
51
|
+
return 0
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _cmd_doctor(args: argparse.Namespace) -> int:
|
|
55
|
+
from paradox2.doctor import diagnose
|
|
56
|
+
|
|
57
|
+
report = diagnose()
|
|
58
|
+
if args.format == "json":
|
|
59
|
+
print(json.dumps(report, sort_keys=True))
|
|
60
|
+
else:
|
|
61
|
+
for key, value in report.items():
|
|
62
|
+
print(f"{key}: {value}")
|
|
63
|
+
return 0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
67
|
+
parser = argparse.ArgumentParser(prog="paradox2")
|
|
68
|
+
sub = parser.add_subparsers(dest="command")
|
|
69
|
+
|
|
70
|
+
p_extract = sub.add_parser("extract", help="one-shot extraction (default if no subcommand given)")
|
|
71
|
+
p_extract.add_argument("path")
|
|
72
|
+
p_extract.add_argument("--backend", default="auto", choices=["auto", "gpu", "cpu"])
|
|
73
|
+
p_extract.add_argument("--language", default="latin")
|
|
74
|
+
p_extract.add_argument("--format", default="markdown", choices=["markdown", "json"])
|
|
75
|
+
p_extract.add_argument("--feature", help="comma-separated projection (text,fields,tables,...)")
|
|
76
|
+
p_extract.set_defaults(func=_cmd_extract)
|
|
77
|
+
|
|
78
|
+
p_serve = sub.add_parser("serve", help="start the persistent worker")
|
|
79
|
+
p_serve.add_argument("--socket", default=DEFAULT_SOCKET_PATH)
|
|
80
|
+
p_serve.set_defaults(func=_cmd_serve)
|
|
81
|
+
|
|
82
|
+
p_client = sub.add_parser("client", help="extract via a running service (fast, warm)")
|
|
83
|
+
p_client.add_argument("path")
|
|
84
|
+
p_client.add_argument("--socket", default=DEFAULT_SOCKET_PATH)
|
|
85
|
+
p_client.add_argument("--backend", default="auto", choices=["auto", "gpu", "cpu"])
|
|
86
|
+
p_client.add_argument("--language", default="latin")
|
|
87
|
+
p_client.set_defaults(func=_cmd_client)
|
|
88
|
+
|
|
89
|
+
p_doctor = sub.add_parser("doctor", help="report PDF/OCR runtime health")
|
|
90
|
+
p_doctor.add_argument("--format", choices=["text", "json"], default="text")
|
|
91
|
+
p_doctor.set_defaults(func=_cmd_doctor)
|
|
92
|
+
|
|
93
|
+
return parser
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def main(argv: list[str] | None = None) -> int:
|
|
97
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
98
|
+
|
|
99
|
+
# Bare `paradox2 <path>` (no subcommand) defaults to extract, for
|
|
100
|
+
# convenience — matches the "1-second script" ergonomics goal.
|
|
101
|
+
if argv and argv[0] not in ("extract", "serve", "client", "doctor", "-h", "--help"):
|
|
102
|
+
argv = ["extract", *argv]
|
|
103
|
+
|
|
104
|
+
parser = build_parser()
|
|
105
|
+
args = parser.parse_args(argv)
|
|
106
|
+
if not getattr(args, "func", None):
|
|
107
|
+
parser.print_help()
|
|
108
|
+
return 1
|
|
109
|
+
return args.func(args)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
if __name__ == "__main__":
|
|
113
|
+
sys.exit(main())
|
paradox2/config.py
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
"""Small explicit configuration surface for stable routing knobs."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, replace
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True, slots=True)
|
|
9
|
+
class ExtractConfig:
|
|
10
|
+
backend: str = "auto"
|
|
11
|
+
language: str = "latin"
|
|
12
|
+
dpi: int = 200
|
|
13
|
+
include_links: bool = False
|
|
14
|
+
include_annotations: bool = False
|
|
15
|
+
detect_language: bool = False
|
|
16
|
+
# Paradox-pdf 0.9.2 called this "scan_text_threshold" (same value, 50) -
|
|
17
|
+
# kept as a distinct name here since digital_fast.py's own constant is
|
|
18
|
+
# named _MIN_NATIVE_TEXT_CHARS; this field is the public override knob
|
|
19
|
+
# for it (api.py wires it through raw_has_text's min_native_text_chars).
|
|
20
|
+
scan_text_threshold: int = 50
|
|
21
|
+
|
|
22
|
+
def with_overrides(self, **values) -> "ExtractConfig":
|
|
23
|
+
allowed = {
|
|
24
|
+
"backend", "language", "dpi", "include_links",
|
|
25
|
+
"include_annotations", "detect_language", "scan_text_threshold",
|
|
26
|
+
}
|
|
27
|
+
unknown = set(values) - allowed
|
|
28
|
+
if unknown:
|
|
29
|
+
raise TypeError(f"unknown extraction options: {sorted(unknown)}")
|
|
30
|
+
return replace(self, **values)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
__all__ = ["ExtractConfig"]
|
paradox2/doctor.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Runtime health checks; never hide a backend mismatch."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from paradox2.runtime import machine_state
|
|
6
|
+
from paradox2.runtime.backend_detect import has_cuda
|
|
7
|
+
from paradox2.runtime.guardrails import BackendMismatch, check_backend
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def diagnose() -> dict[str, object]:
|
|
11
|
+
"""Return a side-effect-free report suitable for CI and the CLI."""
|
|
12
|
+
report: dict[str, object] = {"cuda": has_cuda()}
|
|
13
|
+
try:
|
|
14
|
+
import pymupdf
|
|
15
|
+
|
|
16
|
+
report["pymupdf"] = getattr(pymupdf, "__version__", "installed")
|
|
17
|
+
except Exception as exc:
|
|
18
|
+
report["pymupdf_error"] = str(exc)
|
|
19
|
+
try:
|
|
20
|
+
import paddle
|
|
21
|
+
|
|
22
|
+
report["paddle"] = getattr(paddle, "__version__", "unknown")
|
|
23
|
+
report["paddle_cuda_compiled"] = bool(paddle.device.is_compiled_with_cuda())
|
|
24
|
+
report["paddle_devices"] = int(paddle.device.cuda.device_count())
|
|
25
|
+
except Exception as exc:
|
|
26
|
+
report["paddle_error"] = str(exc)
|
|
27
|
+
|
|
28
|
+
torch_cuda = False
|
|
29
|
+
torch_mps = False
|
|
30
|
+
try:
|
|
31
|
+
import torch
|
|
32
|
+
|
|
33
|
+
torch_cuda = bool(torch.cuda.is_available())
|
|
34
|
+
# ROCm-built torch reports itself through this SAME torch.cuda API
|
|
35
|
+
# (confusingly named) - True here does not necessarily mean NVIDIA.
|
|
36
|
+
mps_backend = getattr(torch.backends, "mps", None)
|
|
37
|
+
torch_mps = bool(mps_backend and mps_backend.is_available())
|
|
38
|
+
except Exception:
|
|
39
|
+
pass
|
|
40
|
+
report["torch_cuda"] = torch_cuda
|
|
41
|
+
report["torch_mps"] = torch_mps
|
|
42
|
+
|
|
43
|
+
# cuda=False (paddle has no usable NVIDIA device) previously looked
|
|
44
|
+
# IDENTICAL whether the machine had no GPU at all, or had a real
|
|
45
|
+
# AMD/Apple GPU torch can see but paddle's mainline wheel cannot (no
|
|
46
|
+
# ROCm/Metal build) - confirmed 2026-09-24, no such distinction existed
|
|
47
|
+
# anywhere in this codebase. Surfacing it matters: the core OCR
|
|
48
|
+
# pipeline (paddle-only) is correctly CPU-bound either way, but opt-in
|
|
49
|
+
# torch/ultralytics-based specialists (e.g. table_detector.py's YOLO26)
|
|
50
|
+
# may still be able to use a GPU paddle can't.
|
|
51
|
+
report["gpu_unsupported_by_paddle"] = bool(not report["cuda"] and (torch_cuda or torch_mps))
|
|
52
|
+
if report["gpu_unsupported_by_paddle"]:
|
|
53
|
+
report["gpu_note"] = (
|
|
54
|
+
f"A GPU is visible to torch (cuda={torch_cuda}, mps={torch_mps}) but not "
|
|
55
|
+
"to paddle - mainline paddlepaddle has no AMD ROCm or Apple Metal build, "
|
|
56
|
+
"so the core OCR pipeline will run on CPU. Opt-in torch-based specialists "
|
|
57
|
+
"(e.g. YOLO26 table detection) may still be able to use this GPU."
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
# Per-machine cached verdicts (see machine_state.py) - what this
|
|
61
|
+
# machine has already learned about itself, and how many times a
|
|
62
|
+
# timeout-guarded engine call has fallen back this process. Operators
|
|
63
|
+
# can use this to see WHY a tier is silently skipped without digging
|
|
64
|
+
# through logs.
|
|
65
|
+
report["cached_engine_status"] = machine_state.load_state().get("engines", {})
|
|
66
|
+
report["timeout_events"] = machine_state.get_timeout_stats()
|
|
67
|
+
return report
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def doctor() -> dict[str, object]:
|
|
71
|
+
return diagnose()
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
__all__ = ["BackendMismatch", "check_backend", "diagnose", "doctor"]
|
|
File without changes
|