paradox2 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. paradox2/__init__.py +7 -0
  2. paradox2/__main__.py +5 -0
  3. paradox2/api.py +320 -0
  4. paradox2/cli.py +113 -0
  5. paradox2/config.py +33 -0
  6. paradox2/doctor.py +74 -0
  7. paradox2/engines/__init__.py +0 -0
  8. paradox2/engines/_gpu_worker.py +214 -0
  9. paradox2/engines/_html_table.py +68 -0
  10. paradox2/engines/_lazy.py +53 -0
  11. paradox2/engines/_row_confidence.py +166 -0
  12. paradox2/engines/doc_classifier.py +67 -0
  13. paradox2/engines/formula_latex.py +38 -0
  14. paradox2/engines/handwriting_trocr.py +49 -0
  15. paradox2/engines/kie_heuristic.py +366 -0
  16. paradox2/engines/signature_detr.py +43 -0
  17. paradox2/engines/table_azure_di.py +161 -0
  18. paradox2/engines/table_detector.py +130 -0
  19. paradox2/engines/table_ppstructure.py +256 -0
  20. paradox2/engines/table_rapidai.py +178 -0
  21. paradox2/engines/table_vlm.py +207 -0
  22. paradox2/exporters/__init__.py +0 -0
  23. paradox2/exporters/markdown.py +99 -0
  24. paradox2/exporters/structured.py +129 -0
  25. paradox2/formats/__init__.py +0 -0
  26. paradox2/formats/archive_adapter.py +176 -0
  27. paradox2/formats/detector.py +150 -0
  28. paradox2/formats/doc_adapter.py +55 -0
  29. paradox2/formats/docx_adapter.py +38 -0
  30. paradox2/formats/eml_adapter.py +63 -0
  31. paradox2/formats/image_adapter.py +19 -0
  32. paradox2/formats/open_document_adapter.py +71 -0
  33. paradox2/formats/pptx_adapter.py +33 -0
  34. paradox2/formats/render_bundle.py +95 -0
  35. paradox2/formats/rtf_adapter.py +46 -0
  36. paradox2/formats/text_adapters.py +48 -0
  37. paradox2/formats/web_adapter.py +64 -0
  38. paradox2/formats/xls_adapter.py +43 -0
  39. paradox2/formats/xlsb_adapter.py +35 -0
  40. paradox2/formats/xlsx_adapter.py +41 -0
  41. paradox2/ir/__init__.py +0 -0
  42. paradox2/pipeline/__init__.py +0 -0
  43. paradox2/pipeline/annotations.py +192 -0
  44. paradox2/pipeline/classify.py +113 -0
  45. paradox2/pipeline/digital_fast.py +220 -0
  46. paradox2/pipeline/feature_flags.py +78 -0
  47. paradox2/pipeline/language_detect.py +43 -0
  48. paradox2/pipeline/links.py +103 -0
  49. paradox2/pipeline/orientation.py +87 -0
  50. paradox2/pipeline/reading_order.py +124 -0
  51. paradox2/pipeline/scanned.py +488 -0
  52. paradox2/runtime/__init__.py +0 -0
  53. paradox2/runtime/backend_detect.py +49 -0
  54. paradox2/runtime/guardrails.py +28 -0
  55. paradox2/runtime/machine_state.py +380 -0
  56. paradox2/runtime/page_pool.py +159 -0
  57. paradox2/runtime/resource_detect.py +339 -0
  58. paradox2/service/__init__.py +0 -0
  59. paradox2/service/client.py +73 -0
  60. paradox2/service/protocol.py +61 -0
  61. paradox2/service/server.py +81 -0
  62. paradox2/tables/__init__.py +0 -0
  63. paradox2/tables/cv_tables.py +106 -0
  64. paradox2/tables/geometry.py +84 -0
  65. paradox2/tables/line_tables.py +320 -0
  66. paradox2/tables/scanned.py +322 -0
  67. paradox2/tables/scanned_router.py +574 -0
  68. paradox2/tables/structure.py +146 -0
  69. paradox2-0.1.0.dist-info/METADATA +227 -0
  70. paradox2-0.1.0.dist-info/RECORD +73 -0
  71. paradox2-0.1.0.dist-info/WHEEL +4 -0
  72. paradox2-0.1.0.dist-info/entry_points.txt +2 -0
  73. paradox2-0.1.0.dist-info/licenses/LICENSE +34 -0
paradox2/__init__.py ADDED
@@ -0,0 +1,7 @@
1
+ """Public package surface for the from-scratch extractor."""
2
+
3
+ from paradox2.api import extract, extract_kie, extract_tables, extract_text, read
4
+ from paradox2.doctor import doctor
5
+ from paradox2.runtime.machine_state import warmup
6
+
7
+ __all__ = ["extract", "extract_kie", "extract_tables", "extract_text", "read", "doctor", "warmup"]
paradox2/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ """Allow ``python -m paradox2`` as well as the installed console script."""
2
+
3
+ from paradox2.cli import main
4
+
5
+ raise SystemExit(main())
paradox2/api.py ADDED
@@ -0,0 +1,320 @@
1
+ """Public facade — dispatch only, no business logic.
2
+
3
+ The old api.py grew to 2,783 lines by letting business logic (add-on
4
+ dispatch, backend resolution details, overlay rendering) live in the
5
+ facade itself. This file must stay thin: routing to pipeline.* and
6
+ attaching tables. If this file starts approaching 100+ lines, that's a
7
+ signal work leaked out of pipeline/ and should be pushed back down
8
+ (see docs/PLAN.md Phase 2).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import pymupdf
14
+ import re
15
+ from pathlib import Path
16
+ from collections.abc import Sequence
17
+
18
+ from paradox2.pipeline import digital_fast, scanned
19
+ from paradox2.pipeline.digital_fast import PageResult
20
+ from paradox2.tables.line_tables import detect_table_bboxes
21
+
22
+
23
+ def _page_selection(pages) -> set[int] | None:
24
+ if pages is None:
25
+ return None
26
+ if isinstance(pages, int):
27
+ return {pages - 1}
28
+ if isinstance(pages, str):
29
+ selected: set[int] = set()
30
+ for part in pages.replace(" ", "").split(","):
31
+ if not part:
32
+ continue
33
+ if "-" in part:
34
+ start, end = part.split("-", 1)
35
+ selected.update(range(int(start) - 1, int(end)))
36
+ else:
37
+ selected.add(int(part) - 1)
38
+ return selected
39
+ return {int(page) - 1 for page in pages}
40
+
41
+
42
+ def extract(
43
+ path: str,
44
+ backend: str = "auto",
45
+ language: str = "latin",
46
+ *,
47
+ pages: Sequence[int] | str | int | None = None,
48
+ feature: str | list[str] | None = None,
49
+ fields: bool = False,
50
+ include_links: bool = False,
51
+ include_annotations: bool = False,
52
+ detect_language: bool = False,
53
+ output_format: str = "dict",
54
+ mode: str | None = None,
55
+ config=None,
56
+ classify: bool = False,
57
+ handwriting: bool = False,
58
+ signatures: bool = False,
59
+ formulas: bool = False,
60
+ marks: bool = False,
61
+ detect_formulas: bool = False,
62
+ detect_signatures: bool = False,
63
+ page_workers: int = 1,
64
+ ):
65
+ """Extract every page of a PDF. Digital pages use the zero-ML fast path
66
+ (with vector-line table detection attached); scanned pages route to the
67
+ GPU OCR pipeline. Each page decides independently — a mixed digital+
68
+ scanned document is handled correctly per-page.
69
+
70
+ page_workers>1 parallelizes only the scanned pages across a
71
+ ProcessPoolExecutor (see pipeline.scanned.extract_scanned_parallel for
72
+ the measured numbers: ~1.9x on GPU, ~2x on CPU vs sequential, on a
73
+ 12-page real document). Digital pages stay on the sequential zero-ML
74
+ path regardless — they're already fast enough that process-spawn
75
+ overhead would not pay for itself."""
76
+ min_native_text_chars = digital_fast._MIN_NATIVE_TEXT_CHARS
77
+ scanned_dpi = 200 # matches scanned.py's own _page_to_image default
78
+ if config is not None:
79
+ # Config is an explicit convenience object; direct keyword arguments
80
+ # remain the most local override for callers that need one-off changes.
81
+ backend = getattr(config, "backend", backend)
82
+ language = getattr(config, "language", language)
83
+ include_links = include_links or bool(getattr(config, "include_links", False))
84
+ include_annotations = include_annotations or bool(getattr(config, "include_annotations", False))
85
+ detect_language = detect_language or bool(getattr(config, "detect_language", False))
86
+ # Paradox-pdf 0.9.2 parity knob (was "scan_text_threshold") - only
87
+ # overrides the digital/scanned routing decision when a caller
88
+ # explicitly sets it via ExtractConfig; every existing caller keeps
89
+ # today's behavior unchanged (default matches the module constant).
90
+ min_native_text_chars = getattr(config, "scan_text_threshold", min_native_text_chars)
91
+ # ExtractConfig.dpi existed but was never read here - only affects scanned-OCR render.
92
+ scanned_dpi = getattr(config, "dpi", scanned_dpi)
93
+
94
+ # Keep non-PDF adapters behind the same public facade. They are already
95
+ # normalized to PageResult by formats.detector and never pay PDF/OCR costs.
96
+ if Path(path).suffix.lower() != ".pdf":
97
+ from paradox2.formats.detector import detect_and_extract
98
+ from paradox2.exporters.structured import select_feature, serialize
99
+
100
+ result = detect_and_extract(path, backend=backend, language=language)
101
+ if feature is None and output_format == "dict":
102
+ return result
103
+ value = select_feature(result, feature or "all")
104
+ return serialize(value, output_format)
105
+
106
+ from paradox2.pipeline.feature_flags import resolve_requested_features
107
+
108
+ flags = resolve_requested_features(
109
+ feature=feature, fields=fields, include_links=include_links,
110
+ include_annotations=include_annotations, detect_language=detect_language,
111
+ mode=mode, classify=classify, handwriting=handwriting, signatures=signatures,
112
+ formulas=formulas, marks=marks, detect_formulas=detect_formulas,
113
+ detect_signatures=detect_signatures,
114
+ )
115
+ selected = _page_selection(pages)
116
+
117
+ doc = pymupdf.open(path)
118
+ digital_fast.reject_if_encrypted(doc, path)
119
+ try:
120
+ precomputed_scanned: dict[int, PageResult] = {}
121
+ page_dicts: dict[int, dict] = {}
122
+ if page_workers > 1:
123
+ precomputed_scanned, page_dicts = scanned.precompute_scanned_pages(
124
+ path, doc, selected, backend, language, page_workers, min_native_text_chars, scanned_dpi
125
+ )
126
+
127
+ results: list[PageResult] = []
128
+ for page in doc:
129
+ if selected is not None and page.number not in selected:
130
+ continue
131
+ # Single get_text("dict") call, reused for both the has-text
132
+ # routing check and (on the digital path) the real extraction —
133
+ # was two separate PyMuPDF text-extraction passes per digital
134
+ # page (has_text_layer's get_text("words") + extract_page's own
135
+ # get_text("dict")), pure duplicate work with no benefit. Reuse
136
+ # the pre-pass's cached dict when page_workers>1 already parsed
137
+ # this page above, instead of parsing it a second time.
138
+ raw = page_dicts.get(page.number)
139
+ if raw is None:
140
+ raw = page.get_text("dict")
141
+ has_text = digital_fast.raw_has_text(raw, min_native_text_chars)
142
+ if has_text:
143
+ result = digital_fast.extract_page(page, raw=raw)
144
+ if flags.tables:
145
+ # Stroke-line extraction is expensive on vector-dense
146
+ # pages (verified: ~20s/call on a real page with
147
+ # 29,121 drawings) - compute once and share it between
148
+ # bbox detection and each bbox's structure extraction,
149
+ # instead of recomputing per call.
150
+ from paradox2.tables.line_tables import _extract_stroke_lines_from_drawings
151
+
152
+ horiz_segs, vert_segs = _extract_stroke_lines_from_drawings(page)
153
+ result.tables = detect_table_bboxes(page, horiz=horiz_segs, vert=vert_segs)
154
+ if result.tables:
155
+ from paradox2.tables.structure import extract_table_structured
156
+
157
+ for bbox in result.tables:
158
+ result.structured_tables.extend(
159
+ extract_table_structured(
160
+ page, bbox, horiz_segs=horiz_segs, vert_segs=vert_segs
161
+ )
162
+ )
163
+ else:
164
+ if page.number in precomputed_scanned:
165
+ result = precomputed_scanned[page.number]
166
+ else:
167
+ result = scanned.extract_page(page, backend=backend, language=language, dpi=scanned_dpi)
168
+ if flags.tables:
169
+ from paradox2.tables.scanned_router import route_scanned_tables
170
+
171
+ result.structured_tables = route_scanned_tables(page, result.blocks)
172
+
173
+ if flags.fields:
174
+ from paradox2.engines.kie_heuristic import (
175
+ detect_grid_regions,
176
+ extract_key_values,
177
+ extract_key_values_spatial,
178
+ looks_like_form,
179
+ )
180
+
181
+ if has_text:
182
+ words = page.get_text("words")
183
+ lines = page.get_text().splitlines()
184
+ if looks_like_form(lines) or any(":" in line for line in lines):
185
+ result.fields = extract_key_values_spatial(
186
+ words, table_boxes=detect_grid_regions(words)
187
+ )
188
+ else:
189
+ result.fields = extract_key_values([b.text for b in result.blocks])
190
+
191
+ if flags.links:
192
+ from paradox2.pipeline.links import extract_page_hyperlinks
193
+
194
+ result.links = extract_page_hyperlinks(page)
195
+ if flags.annotations:
196
+ from paradox2.pipeline.annotations import extract_page_annotations
197
+
198
+ result.annotations = extract_page_annotations(page)
199
+ if flags.handwriting or flags.signatures or flags.formulas:
200
+ _apply_specialists(
201
+ page, result,
202
+ handwriting=flags.handwriting,
203
+ signatures=flags.signatures,
204
+ formulas=flags.formulas,
205
+ )
206
+ if flags.marks:
207
+ result.marks = [
208
+ {"bbox": list(block.bbox), "text": block.text,
209
+ "bold": block.bold, "italic": block.italic}
210
+ for block in result.blocks if block.bold or block.italic
211
+ ]
212
+ if flags.language:
213
+ from paradox2.pipeline.language_detect import detect_text_language
214
+
215
+ result.language = detect_text_language(result.text)
216
+ result.metadata = {"page_number": page.number + 1, "source": str(path)}
217
+ results.append(result)
218
+ if flags.classify and results:
219
+ from paradox2.engines.doc_classifier import get_engine
220
+
221
+ label, confidence = get_engine().classify("\n".join(p.text for p in results))
222
+ classification = {"label": label, "confidence": confidence}
223
+ for result in results:
224
+ result.classification = classification
225
+ # Default remains the efficient, geometry-preserving IR used by the
226
+ # service and existing callers. Feature/output requests are pure
227
+ # projections: extraction is never repeated.
228
+ if feature is None:
229
+ if output_format == "dict":
230
+ return results
231
+ if output_format == "markdown":
232
+ from paradox2.exporters.structured import serialize
233
+
234
+ return serialize(results, "markdown")
235
+ from paradox2.exporters.structured import select_feature, serialize
236
+
237
+ value = select_feature(results, feature or "all")
238
+ return serialize(value, output_format)
239
+ finally:
240
+ doc.close()
241
+
242
+
243
+ def extract_kie(path: str) -> list[list[dict]]:
244
+ """Per-page key-value extraction. Only meaningful on digital, form-like
245
+ pages (invoices/forms) - scanned pages and prose pages return []. Kept
246
+ separate from extract() rather than a feature= flag: KIE's output shape
247
+ (list of key-value dicts) doesn't fit PageResult, forcing it in would
248
+ be the wrong abstraction just to have one entry point."""
249
+ from paradox2.engines.kie_heuristic import (
250
+ detect_grid_regions,
251
+ extract_key_values_spatial,
252
+ looks_like_form,
253
+ )
254
+
255
+ doc = pymupdf.open(path)
256
+ try:
257
+ results: list[list[dict]] = []
258
+ for page in doc:
259
+ if not digital_fast.has_text_layer(page):
260
+ results.append([])
261
+ continue
262
+ words = page.get_text("words")
263
+ lines = page.get_text().splitlines()
264
+ if not looks_like_form(lines):
265
+ results.append([])
266
+ continue
267
+ regions = detect_grid_regions(words)
268
+ results.append(extract_key_values_spatial(words, table_boxes=regions))
269
+ return results
270
+ finally:
271
+ doc.close()
272
+
273
+
274
+ def _render_crop(page, bbox=None, dpi: int = 180):
275
+ """Render a page/crop once for opt-in vision specialists."""
276
+ pix = page.get_pixmap(dpi=dpi, clip=pymupdf.Rect(bbox) if bbox else None, alpha=False)
277
+ from PIL import Image
278
+ return Image.frombytes("RGB", [pix.width, pix.height], pix.samples)
279
+
280
+
281
+ def _apply_specialists(page, result: PageResult, *, handwriting: bool, signatures: bool, formulas: bool):
282
+ """Run explicitly requested heavy models; never touched on the hot path."""
283
+ scale = 72.0 / 180.0
284
+ if signatures:
285
+ from paradox2.engines.signature_detr import get_engine
286
+ image = _render_crop(page)
287
+ for item in get_engine().detect(image):
288
+ item["bbox"] = [round(v * scale, 2) for v in item["bbox"]]
289
+ item["page"] = result.page_number
290
+ result.signatures.append(item)
291
+ if handwriting:
292
+ from paradox2.engines.handwriting_trocr import get_engine
293
+ engine = get_engine()
294
+ # TrOCR is line-oriented; use extracted blocks as tight crops and cap
295
+ # work so an accidental request cannot turn into an unbounded scan.
296
+ for block in result.blocks[:64]:
297
+ text = engine.process(_render_crop(page, block.bbox))
298
+ if text:
299
+ result.handwriting.append({"bbox": list(block.bbox), "text": text})
300
+ if formulas:
301
+ from paradox2.engines.formula_latex import image_to_latex
302
+ for block in result.blocks:
303
+ if not re.search(r"(?:=|∑|∫|√|\^|_[{\d])", block.text):
304
+ continue
305
+ latex = image_to_latex(_render_crop(page, block.bbox))
306
+ if latex:
307
+ result.formulas.append({"bbox": list(block.bbox), "text": block.text, "latex": latex})
308
+
309
+
310
+ def read(path: str, **kwargs):
311
+ """Compatibility alias for universal multi-format dispatch."""
312
+ return extract(path, **kwargs)
313
+
314
+
315
+ def extract_text(path: str, **kwargs) -> str:
316
+ return extract(path, feature="text", **kwargs)
317
+
318
+
319
+ def extract_tables(path: str, **kwargs):
320
+ return extract(path, feature="tables", **kwargs)
paradox2/cli.py ADDED
@@ -0,0 +1,113 @@
1
+ """paradox2 CLI: one-shot extraction, and serve/client subcommands wrapping
2
+ service/ (the persistent worker). argparse, stdlib only — no new deps for
3
+ a CLI wrapper around already-built pipeline/service code.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import argparse
9
+ import json
10
+ import sys
11
+
12
+ from paradox2.service.protocol import DEFAULT_SOCKET_PATH
13
+
14
+
15
+ def _cmd_extract(args: argparse.Namespace) -> int:
16
+ from paradox2.formats.detector import detect_and_extract
17
+ from paradox2.exporters.structured import page_to_dict, select_feature, serialize
18
+
19
+ pages = detect_and_extract(args.path, backend=args.backend, language=args.language)
20
+
21
+ if args.feature:
22
+ value = select_feature(pages, args.feature.split(","))
23
+ print(serialize(value, args.format))
24
+ elif args.format == "json":
25
+ print(json.dumps([page_to_dict(p) for p in pages], ensure_ascii=False))
26
+ else:
27
+ from paradox2.exporters.markdown import to_markdown
28
+
29
+ print(to_markdown(pages))
30
+ return 0
31
+
32
+
33
+ def _cmd_serve(args: argparse.Namespace) -> int:
34
+ from paradox2.service.server import serve
35
+
36
+ serve(args.socket)
37
+ return 0
38
+
39
+
40
+ def _cmd_client(args: argparse.Namespace) -> int:
41
+ from paradox2.service.client import extract, is_server_running
42
+
43
+ if not is_server_running(args.socket):
44
+ print(f"error: no paradox2 service running at {args.socket} "
45
+ f"(start one with 'paradox2 serve')", file=sys.stderr)
46
+ return 1
47
+
48
+ pages = extract(args.path, backend=args.backend, language=args.language, socket_path=args.socket)
49
+ for p in pages:
50
+ print(p.text)
51
+ return 0
52
+
53
+
54
+ def _cmd_doctor(args: argparse.Namespace) -> int:
55
+ from paradox2.doctor import diagnose
56
+
57
+ report = diagnose()
58
+ if args.format == "json":
59
+ print(json.dumps(report, sort_keys=True))
60
+ else:
61
+ for key, value in report.items():
62
+ print(f"{key}: {value}")
63
+ return 0
64
+
65
+
66
+ def build_parser() -> argparse.ArgumentParser:
67
+ parser = argparse.ArgumentParser(prog="paradox2")
68
+ sub = parser.add_subparsers(dest="command")
69
+
70
+ p_extract = sub.add_parser("extract", help="one-shot extraction (default if no subcommand given)")
71
+ p_extract.add_argument("path")
72
+ p_extract.add_argument("--backend", default="auto", choices=["auto", "gpu", "cpu"])
73
+ p_extract.add_argument("--language", default="latin")
74
+ p_extract.add_argument("--format", default="markdown", choices=["markdown", "json"])
75
+ p_extract.add_argument("--feature", help="comma-separated projection (text,fields,tables,...)")
76
+ p_extract.set_defaults(func=_cmd_extract)
77
+
78
+ p_serve = sub.add_parser("serve", help="start the persistent worker")
79
+ p_serve.add_argument("--socket", default=DEFAULT_SOCKET_PATH)
80
+ p_serve.set_defaults(func=_cmd_serve)
81
+
82
+ p_client = sub.add_parser("client", help="extract via a running service (fast, warm)")
83
+ p_client.add_argument("path")
84
+ p_client.add_argument("--socket", default=DEFAULT_SOCKET_PATH)
85
+ p_client.add_argument("--backend", default="auto", choices=["auto", "gpu", "cpu"])
86
+ p_client.add_argument("--language", default="latin")
87
+ p_client.set_defaults(func=_cmd_client)
88
+
89
+ p_doctor = sub.add_parser("doctor", help="report PDF/OCR runtime health")
90
+ p_doctor.add_argument("--format", choices=["text", "json"], default="text")
91
+ p_doctor.set_defaults(func=_cmd_doctor)
92
+
93
+ return parser
94
+
95
+
96
+ def main(argv: list[str] | None = None) -> int:
97
+ argv = sys.argv[1:] if argv is None else argv
98
+
99
+ # Bare `paradox2 <path>` (no subcommand) defaults to extract, for
100
+ # convenience — matches the "1-second script" ergonomics goal.
101
+ if argv and argv[0] not in ("extract", "serve", "client", "doctor", "-h", "--help"):
102
+ argv = ["extract", *argv]
103
+
104
+ parser = build_parser()
105
+ args = parser.parse_args(argv)
106
+ if not getattr(args, "func", None):
107
+ parser.print_help()
108
+ return 1
109
+ return args.func(args)
110
+
111
+
112
+ if __name__ == "__main__":
113
+ sys.exit(main())
paradox2/config.py ADDED
@@ -0,0 +1,33 @@
1
+ """Small explicit configuration surface for stable routing knobs."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass, replace
6
+
7
+
8
+ @dataclass(frozen=True, slots=True)
9
+ class ExtractConfig:
10
+ backend: str = "auto"
11
+ language: str = "latin"
12
+ dpi: int = 200
13
+ include_links: bool = False
14
+ include_annotations: bool = False
15
+ detect_language: bool = False
16
+ # Paradox-pdf 0.9.2 called this "scan_text_threshold" (same value, 50) -
17
+ # kept as a distinct name here since digital_fast.py's own constant is
18
+ # named _MIN_NATIVE_TEXT_CHARS; this field is the public override knob
19
+ # for it (api.py wires it through raw_has_text's min_native_text_chars).
20
+ scan_text_threshold: int = 50
21
+
22
+ def with_overrides(self, **values) -> "ExtractConfig":
23
+ allowed = {
24
+ "backend", "language", "dpi", "include_links",
25
+ "include_annotations", "detect_language", "scan_text_threshold",
26
+ }
27
+ unknown = set(values) - allowed
28
+ if unknown:
29
+ raise TypeError(f"unknown extraction options: {sorted(unknown)}")
30
+ return replace(self, **values)
31
+
32
+
33
+ __all__ = ["ExtractConfig"]
paradox2/doctor.py ADDED
@@ -0,0 +1,74 @@
1
+ """Runtime health checks; never hide a backend mismatch."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from paradox2.runtime import machine_state
6
+ from paradox2.runtime.backend_detect import has_cuda
7
+ from paradox2.runtime.guardrails import BackendMismatch, check_backend
8
+
9
+
10
+ def diagnose() -> dict[str, object]:
11
+ """Return a side-effect-free report suitable for CI and the CLI."""
12
+ report: dict[str, object] = {"cuda": has_cuda()}
13
+ try:
14
+ import pymupdf
15
+
16
+ report["pymupdf"] = getattr(pymupdf, "__version__", "installed")
17
+ except Exception as exc:
18
+ report["pymupdf_error"] = str(exc)
19
+ try:
20
+ import paddle
21
+
22
+ report["paddle"] = getattr(paddle, "__version__", "unknown")
23
+ report["paddle_cuda_compiled"] = bool(paddle.device.is_compiled_with_cuda())
24
+ report["paddle_devices"] = int(paddle.device.cuda.device_count())
25
+ except Exception as exc:
26
+ report["paddle_error"] = str(exc)
27
+
28
+ torch_cuda = False
29
+ torch_mps = False
30
+ try:
31
+ import torch
32
+
33
+ torch_cuda = bool(torch.cuda.is_available())
34
+ # ROCm-built torch reports itself through this SAME torch.cuda API
35
+ # (confusingly named) - True here does not necessarily mean NVIDIA.
36
+ mps_backend = getattr(torch.backends, "mps", None)
37
+ torch_mps = bool(mps_backend and mps_backend.is_available())
38
+ except Exception:
39
+ pass
40
+ report["torch_cuda"] = torch_cuda
41
+ report["torch_mps"] = torch_mps
42
+
43
+ # cuda=False (paddle has no usable NVIDIA device) previously looked
44
+ # IDENTICAL whether the machine had no GPU at all, or had a real
45
+ # AMD/Apple GPU torch can see but paddle's mainline wheel cannot (no
46
+ # ROCm/Metal build) - confirmed 2026-09-24, no such distinction existed
47
+ # anywhere in this codebase. Surfacing it matters: the core OCR
48
+ # pipeline (paddle-only) is correctly CPU-bound either way, but opt-in
49
+ # torch/ultralytics-based specialists (e.g. table_detector.py's YOLO26)
50
+ # may still be able to use a GPU paddle can't.
51
+ report["gpu_unsupported_by_paddle"] = bool(not report["cuda"] and (torch_cuda or torch_mps))
52
+ if report["gpu_unsupported_by_paddle"]:
53
+ report["gpu_note"] = (
54
+ f"A GPU is visible to torch (cuda={torch_cuda}, mps={torch_mps}) but not "
55
+ "to paddle - mainline paddlepaddle has no AMD ROCm or Apple Metal build, "
56
+ "so the core OCR pipeline will run on CPU. Opt-in torch-based specialists "
57
+ "(e.g. YOLO26 table detection) may still be able to use this GPU."
58
+ )
59
+
60
+ # Per-machine cached verdicts (see machine_state.py) - what this
61
+ # machine has already learned about itself, and how many times a
62
+ # timeout-guarded engine call has fallen back this process. Operators
63
+ # can use this to see WHY a tier is silently skipped without digging
64
+ # through logs.
65
+ report["cached_engine_status"] = machine_state.load_state().get("engines", {})
66
+ report["timeout_events"] = machine_state.get_timeout_stats()
67
+ return report
68
+
69
+
70
+ def doctor() -> dict[str, object]:
71
+ return diagnose()
72
+
73
+
74
+ __all__ = ["BackendMismatch", "check_backend", "diagnose", "doctor"]
File without changes