universal-doc-parser 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
- universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
- universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
- universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
- universal_parser/__init__.py +28 -0
- universal_parser/adaptive/__init__.py +17 -0
- universal_parser/adaptive/config_cache.py +127 -0
- universal_parser/adaptive/fingerprint.py +133 -0
- universal_parser/adaptive/tuner.py +97 -0
- universal_parser/core/__init__.py +0 -0
- universal_parser/core/engine.py +55 -0
- universal_parser/core/router.py +38 -0
- universal_parser/core/schema.py +58 -0
- universal_parser/core/sniffer.py +125 -0
- universal_parser/enrichment/__init__.py +0 -0
- universal_parser/enrichment/vlm_enricher.py +0 -0
- universal_parser/exports/__init__.py +1 -0
- universal_parser/exports/to_chunks.py +108 -0
- universal_parser/exports/to_graph.py +114 -0
- universal_parser/exports/to_markdown.py +31 -0
- universal_parser/extractors/__init__.py +0 -0
- universal_parser/extractors/base.py +39 -0
- universal_parser/extractors/images/__init__.py +0 -0
- universal_parser/extractors/images/scan_extractor.py +79 -0
- universal_parser/extractors/mail/__init__.py +0 -0
- universal_parser/extractors/mail/mail_extractor.py +206 -0
- universal_parser/extractors/office/__init__.py +0 -0
- universal_parser/extractors/office/docx_extractor.py +131 -0
- universal_parser/extractors/office/legacy_extractor.py +112 -0
- universal_parser/extractors/office/pptx_extractor.py +109 -0
- universal_parser/extractors/office/xlsx_extractor.py +93 -0
- universal_parser/extractors/pdf/__init__.py +0 -0
- universal_parser/extractors/pdf/native.py +331 -0
- universal_parser/extractors/pdf/tables.py +155 -0
- universal_parser/extractors/pdf/visual_onnx.py +0 -0
- universal_parser/extractors/structured/__init__.py +0 -0
- universal_parser/extractors/structured/csv_extractor.py +86 -0
- universal_parser/extractors/structured/json_xml_extractor.py +104 -0
- universal_parser/extractors/structured/parquet_extractor.py +59 -0
- universal_parser/extractors/web/__init__.py +1 -0
- universal_parser/extractors/web/epub_extractor.py +81 -0
- universal_parser/extractors/web/html_extractor.py +111 -0
- universal_parser/mcp/__init__.py +1 -0
- universal_parser/mcp/server.py +61 -0
- universal_parser/observability/__init__.py +12 -0
- universal_parser/observability/dashboard.py +231 -0
- universal_parser/observability/logger.py +0 -0
- universal_parser/observability/metrics.py +99 -0
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
universal_parser/__init__.py,sha256=aXTv6p_QoWsK5yyuRATxtnPkoTL3M_ju7AetSDRhbSY,1337
|
|
2
|
+
universal_parser/adaptive/__init__.py,sha256=-y0YQaC-GrsGJejduofpE4Cm7lzcZeogpk-THo49RAc,476
|
|
3
|
+
universal_parser/adaptive/config_cache.py,sha256=U4MQkO7SrNESpiR7UWw5dSulRqtUd7TEUj0we8ytmF4,4698
|
|
4
|
+
universal_parser/adaptive/fingerprint.py,sha256=hwnk85msBHCtRW4wekH5bFH2pEaqLxH1v1FJEJJ1eEQ,4720
|
|
5
|
+
universal_parser/adaptive/tuner.py,sha256=zT50xD2fs5nkYppI9DgzK5hzCzCwrU17P6Cx09uPdpY,3791
|
|
6
|
+
universal_parser/core/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
universal_parser/core/engine.py,sha256=ZLDVP03MMxCEhUfUVBNf4CSIvHq4fwVi2p9pXhZjvEg,1725
|
|
8
|
+
universal_parser/core/router.py,sha256=jJOE4qT3s9J-KlHjhuYZjT6bRKyHgeSLnM2QWYW1lk8,1387
|
|
9
|
+
universal_parser/core/schema.py,sha256=vyB5MKwrFrRxAuIH7g4BAq3X3PavJ2fAr1gthkLqvJI,1787
|
|
10
|
+
universal_parser/core/sniffer.py,sha256=GqUHRTJt-Fc8cXr-zm5ZwsGuUXFyf2oLCYwipCea2SE,3700
|
|
11
|
+
universal_parser/enrichment/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
12
|
+
universal_parser/enrichment/vlm_enricher.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
13
|
+
universal_parser/exports/__init__.py,sha256=-1b3-EAbnEgjgglQVPG0jPrzrAwabmMxzXIHBG62YrY,17
|
|
14
|
+
universal_parser/exports/to_chunks.py,sha256=wnEh0jD5YPrUeB9bwY-8ac1G1lc0wD8teHUDLjzSMak,3687
|
|
15
|
+
universal_parser/exports/to_graph.py,sha256=AuqrtDkd270VL_leWWIJ9_Y22IUGWnW5GlZbUMHH6lw,3262
|
|
16
|
+
universal_parser/exports/to_markdown.py,sha256=1_rgNL2GD3SQNdZXBtUI98w7ZxnMcYojiZ0gH0pLaig,985
|
|
17
|
+
universal_parser/extractors/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
18
|
+
universal_parser/extractors/base.py,sha256=Vslbmb0fdd2e2a5j-e8GZX7pZBXMFq4yogQLhW0kf2c,1400
|
|
19
|
+
universal_parser/extractors/images/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
20
|
+
universal_parser/extractors/images/scan_extractor.py,sha256=ftD2uitFlrAL7YmK8xyhJHeuMI0UWhwLfb2jUV3fZ8Y,2601
|
|
21
|
+
universal_parser/extractors/mail/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
22
|
+
universal_parser/extractors/mail/mail_extractor.py,sha256=dpeDkdLERWFdjCCU4tkglO26iLdXbVDyLzPXuDr04R0,7530
|
|
23
|
+
universal_parser/extractors/office/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
24
|
+
universal_parser/extractors/office/docx_extractor.py,sha256=pRfQjTd4P-W3iy5SNwf8HeECsQTthTf4_mbJQk4hHaI,4458
|
|
25
|
+
universal_parser/extractors/office/legacy_extractor.py,sha256=Dq0c0oCZMoENie2qND1QoI_Qv1yUzijbcRTxYCa7rZY,3843
|
|
26
|
+
universal_parser/extractors/office/pptx_extractor.py,sha256=7WxDmpQAxK8uYPe1Hwpkjhbun40nYiW4ehHaSOuNw6c,4074
|
|
27
|
+
universal_parser/extractors/office/xlsx_extractor.py,sha256=Z06ayb0Z37lJVlgglanhmiadCogwwpKEBCN_5bewd7c,3438
|
|
28
|
+
universal_parser/extractors/pdf/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
29
|
+
universal_parser/extractors/pdf/native.py,sha256=B36CvFYA22MYCQa9xmdnKsxsu43gxOTh2lLjxbgT28c,13352
|
|
30
|
+
universal_parser/extractors/pdf/tables.py,sha256=Xqhcfq1uA1Q3sV2oXzVmUwDBxsStdtPbmMIRJfBBBTM,5501
|
|
31
|
+
universal_parser/extractors/pdf/visual_onnx.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
32
|
+
universal_parser/extractors/structured/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
33
|
+
universal_parser/extractors/structured/csv_extractor.py,sha256=YKZgjyfOJdl-x8W5Q4lKam82QtwFRCu92XxSSBTnXUs,2999
|
|
34
|
+
universal_parser/extractors/structured/json_xml_extractor.py,sha256=MNVOnJxLoWuf4s20PPRzi4aPOC6bar-EONsQX32xFb4,3835
|
|
35
|
+
universal_parser/extractors/structured/parquet_extractor.py,sha256=RT6WUi1aMmtGYusV5F7yd7b4Y3vxGduykVrKvLN-YCw,1912
|
|
36
|
+
universal_parser/extractors/web/__init__.py,sha256=qgz0BE4177sP3FVQqDifYgNKejYD-D1Dbq-YTY0we4w,17
|
|
37
|
+
universal_parser/extractors/web/epub_extractor.py,sha256=9H5CI_Y7k046_ho38PojRi-iqxEr3b4XQ-w1Ii3_Dd8,2836
|
|
38
|
+
universal_parser/extractors/web/html_extractor.py,sha256=7Erx1oncwCLmKx3Iz1gb2X8zh7mx5IqpQN5d_rMllHM,3977
|
|
39
|
+
universal_parser/mcp/__init__.py,sha256=IjzqjjmYctuyTAGB4hZpd93bNmu0EMlHLTimYnIm42M,14
|
|
40
|
+
universal_parser/mcp/server.py,sha256=ccyr3Sd2F2GLf7_lKwygZsR0HICQoesALt4sZGXqpq4,1848
|
|
41
|
+
universal_parser/observability/__init__.py,sha256=6LpvpSKaK_W-UtEIfKIYF3VMWyjSlhByBIMW4WM-i3w,312
|
|
42
|
+
universal_parser/observability/dashboard.py,sha256=vNpTAhDfaQf_tk3-5i81dMJggzh9IH-OrCho6InoRMU,7393
|
|
43
|
+
universal_parser/observability/logger.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
44
|
+
universal_parser/observability/metrics.py,sha256=NQEFOXDKVH2QkHV4lrhebu7qLRQgQK9dgah7PHIvJn4,3084
|
|
45
|
+
universal_doc_parser-1.0.0.dist-info/METADATA,sha256=Q5dxXh6tz_XdGPkmbCLl6NP-X_U6iv3AFJmaQ1C7giw,26771
|
|
46
|
+
universal_doc_parser-1.0.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
47
|
+
universal_doc_parser-1.0.0.dist-info/licenses/LICENSE,sha256=TBBq_QxVzFx6N4MBDqm1ipgIA_nmeP0DJlLugIUK1ws,1069
|
|
48
|
+
universal_doc_parser-1.0.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Karan Shelar
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
__version__ = "1.0.0"
|
|
4
|
+
|
|
5
|
+
import universal_parser.extractors.images.scan_extractor
|
|
6
|
+
import universal_parser.extractors.mail.mail_extractor
|
|
7
|
+
import universal_parser.extractors.office.docx_extractor
|
|
8
|
+
import universal_parser.extractors.office.legacy_extractor
|
|
9
|
+
import universal_parser.extractors.office.pptx_extractor
|
|
10
|
+
import universal_parser.extractors.office.xlsx_extractor
|
|
11
|
+
|
|
12
|
+
# Import all extractors to trigger their @register decorators.
|
|
13
|
+
# Without this, the router registry remains empty until the files are imported.
|
|
14
|
+
import universal_parser.extractors.pdf.native
|
|
15
|
+
import universal_parser.extractors.structured.csv_extractor
|
|
16
|
+
import universal_parser.extractors.structured.json_xml_extractor
|
|
17
|
+
import universal_parser.extractors.structured.parquet_extractor
|
|
18
|
+
import universal_parser.extractors.web.epub_extractor
|
|
19
|
+
import universal_parser.extractors.web.html_extractor # noqa: F401
|
|
20
|
+
|
|
21
|
+
# Import the main entry point to expose it at the root of the package
|
|
22
|
+
from universal_parser.core.engine import parse
|
|
23
|
+
from universal_parser.exports.to_chunks import to_chunks
|
|
24
|
+
from universal_parser.exports.to_graph import to_graph
|
|
25
|
+
from universal_parser.exports.to_markdown import to_markdown
|
|
26
|
+
|
|
27
|
+
# Define what is exposed when doing: from universal_parser import *
|
|
28
|
+
__all__ = ["__version__", "parse", "to_chunks", "to_graph", "to_markdown"]
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
from universal_parser.adaptive.config_cache import ParserConfig, TemplateConfigCache
|
|
2
|
+
from universal_parser.adaptive.fingerprint import (
|
|
3
|
+
LayoutFingerprint,
|
|
4
|
+
compute_fingerprint,
|
|
5
|
+
fingerprint_similarity,
|
|
6
|
+
)
|
|
7
|
+
from universal_parser.adaptive.tuner import AdaptiveTuner, FeedbackSignal
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"AdaptiveTuner",
|
|
11
|
+
"FeedbackSignal",
|
|
12
|
+
"LayoutFingerprint",
|
|
13
|
+
"ParserConfig",
|
|
14
|
+
"TemplateConfigCache",
|
|
15
|
+
"compute_fingerprint",
|
|
16
|
+
"fingerprint_similarity",
|
|
17
|
+
]
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections import OrderedDict
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from threading import Lock
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from pydantic import BaseModel, Field
|
|
10
|
+
|
|
11
|
+
from universal_parser.adaptive.fingerprint import (
|
|
12
|
+
LayoutFingerprint,
|
|
13
|
+
fingerprint_similarity,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class ParserConfig(BaseModel):
|
|
18
|
+
"""Hyperparameters and configuration tuned for a specific document layout template."""
|
|
19
|
+
|
|
20
|
+
column_gap_threshold: float = 30.0
|
|
21
|
+
heading_p95_ratio: float = 1.30
|
|
22
|
+
heading_p85_ratio: float = 1.15
|
|
23
|
+
table_detection_mode: str = "auto" # "lattice", "stream", "auto"
|
|
24
|
+
ocr_dpi: int = 150
|
|
25
|
+
min_table_confidence: float = 0.60
|
|
26
|
+
custom_overrides: dict[str, Any] = Field(default_factory=dict)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class TemplateConfigCache:
|
|
30
|
+
"""Thread-safe, LRU cache for template fingerprints and parser configurations."""
|
|
31
|
+
|
|
32
|
+
def __init__(self, capacity: int = 1000) -> None:
|
|
33
|
+
self.capacity = capacity
|
|
34
|
+
self._cache: OrderedDict[str, tuple[LayoutFingerprint, ParserConfig]] = OrderedDict()
|
|
35
|
+
self._lock = Lock()
|
|
36
|
+
|
|
37
|
+
def get_exact(self, hash_digest: str) -> ParserConfig | None:
|
|
38
|
+
"""Retrieves matching ParserConfig strictly by exact SHA-256 hash."""
|
|
39
|
+
with self._lock:
|
|
40
|
+
if hash_digest in self._cache:
|
|
41
|
+
self._cache.move_to_end(hash_digest)
|
|
42
|
+
_, config = self._cache[hash_digest]
|
|
43
|
+
return config.model_copy(deep=True)
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
def get(
|
|
47
|
+
self, fingerprint: LayoutFingerprint, similarity_threshold: float = 0.85
|
|
48
|
+
) -> tuple[str, ParserConfig] | None:
|
|
49
|
+
"""Retrieves matching ParserConfig either by exact hash or best fuzzy match >= similarity_threshold."""
|
|
50
|
+
with self._lock:
|
|
51
|
+
# 1. Exact match
|
|
52
|
+
if fingerprint.hash_digest in self._cache:
|
|
53
|
+
self._cache.move_to_end(fingerprint.hash_digest)
|
|
54
|
+
_, config = self._cache[fingerprint.hash_digest]
|
|
55
|
+
return fingerprint.hash_digest, config.model_copy(deep=True)
|
|
56
|
+
|
|
57
|
+
# 2. Fuzzy similarity search
|
|
58
|
+
best_match_key: str | None = None
|
|
59
|
+
best_score = 0.0
|
|
60
|
+
best_config: ParserConfig | None = None
|
|
61
|
+
|
|
62
|
+
for key, (cached_fp, cached_cfg) in self._cache.items():
|
|
63
|
+
sim = fingerprint_similarity(fingerprint, cached_fp)
|
|
64
|
+
if sim > best_score:
|
|
65
|
+
best_score = sim
|
|
66
|
+
best_match_key = key
|
|
67
|
+
best_config = cached_cfg
|
|
68
|
+
|
|
69
|
+
if (
|
|
70
|
+
best_match_key is not None
|
|
71
|
+
and best_score >= similarity_threshold
|
|
72
|
+
and best_config is not None
|
|
73
|
+
):
|
|
74
|
+
self._cache.move_to_end(best_match_key)
|
|
75
|
+
return best_match_key, best_config.model_copy(deep=True)
|
|
76
|
+
|
|
77
|
+
return None
|
|
78
|
+
|
|
79
|
+
def set(self, fingerprint: LayoutFingerprint, config: ParserConfig) -> None:
|
|
80
|
+
"""Saves or updates a template fingerprint and its tuned configuration."""
|
|
81
|
+
with self._lock:
|
|
82
|
+
if fingerprint.hash_digest in self._cache:
|
|
83
|
+
self._cache.move_to_end(fingerprint.hash_digest)
|
|
84
|
+
self._cache[fingerprint.hash_digest] = (
|
|
85
|
+
fingerprint,
|
|
86
|
+
config.model_copy(deep=True),
|
|
87
|
+
)
|
|
88
|
+
if len(self._cache) > self.capacity:
|
|
89
|
+
self._cache.popitem(last=False)
|
|
90
|
+
|
|
91
|
+
def save(self, file_path: str | Path) -> None:
|
|
92
|
+
"""Serializes cached template configurations to a JSON file."""
|
|
93
|
+
path = Path(file_path)
|
|
94
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
95
|
+
with self._lock:
|
|
96
|
+
data = {}
|
|
97
|
+
for digest, (fp, cfg) in self._cache.items():
|
|
98
|
+
data[digest] = {
|
|
99
|
+
"fingerprint": fp.model_dump(),
|
|
100
|
+
"config": cfg.model_dump(),
|
|
101
|
+
}
|
|
102
|
+
with path.open("w", encoding="utf-8") as f:
|
|
103
|
+
json.dump(data, f, indent=2)
|
|
104
|
+
|
|
105
|
+
def load(self, file_path: str | Path) -> None:
|
|
106
|
+
"""Loads template configurations from a JSON file."""
|
|
107
|
+
path = Path(file_path)
|
|
108
|
+
if not path.is_file():
|
|
109
|
+
return
|
|
110
|
+
|
|
111
|
+
with path.open("r", encoding="utf-8") as f:
|
|
112
|
+
data = json.load(f)
|
|
113
|
+
|
|
114
|
+
with self._lock:
|
|
115
|
+
self._cache.clear()
|
|
116
|
+
for digest, payload in data.items():
|
|
117
|
+
fp = LayoutFingerprint.model_validate(payload["fingerprint"])
|
|
118
|
+
cfg = ParserConfig.model_validate(payload["config"])
|
|
119
|
+
self._cache[digest] = (fp, cfg)
|
|
120
|
+
|
|
121
|
+
def clear(self) -> None:
|
|
122
|
+
with self._lock:
|
|
123
|
+
self._cache.clear()
|
|
124
|
+
|
|
125
|
+
def __len__(self) -> int:
|
|
126
|
+
with self._lock:
|
|
127
|
+
return len(self._cache)
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import math
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from pydantic import BaseModel, Field
|
|
9
|
+
|
|
10
|
+
from universal_parser.core.schema import Document
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class LayoutFingerprint(BaseModel):
|
|
14
|
+
"""Structural layout fingerprint of a document or page template."""
|
|
15
|
+
|
|
16
|
+
hash_digest: str
|
|
17
|
+
spatial_grid: list[list[float]] = Field(default_factory=list)
|
|
18
|
+
type_distribution: dict[str, float] = Field(default_factory=dict)
|
|
19
|
+
page_count: int = 1
|
|
20
|
+
avg_elements_per_page: float = 0.0
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def compute_fingerprint(doc: Document, grid_size: int = 10) -> LayoutFingerprint:
|
|
24
|
+
"""Computes a content-agnostic structural layout fingerprint from a Document.
|
|
25
|
+
Quantizes spatial bounding boxes into an N x N normalized occupancy matrix
|
|
26
|
+
and aggregates element type proportions.
|
|
27
|
+
"""
|
|
28
|
+
grid = [[0.0 for _ in range(grid_size)] for _ in range(grid_size)]
|
|
29
|
+
type_counts: dict[str, int] = {}
|
|
30
|
+
total_elements = len(doc.content_tree)
|
|
31
|
+
|
|
32
|
+
page_count = doc.metadata.page_count or 1
|
|
33
|
+
if page_count <= 0:
|
|
34
|
+
page_count = 1
|
|
35
|
+
|
|
36
|
+
# Standard document dimension reference (Letter / A4 approx: 612 x 792 pt)
|
|
37
|
+
ref_w = 612.0
|
|
38
|
+
ref_h = 792.0
|
|
39
|
+
|
|
40
|
+
for idx, elem in enumerate(doc.content_tree):
|
|
41
|
+
t = elem.type
|
|
42
|
+
type_counts[t] = type_counts.get(t, 0) + 1
|
|
43
|
+
|
|
44
|
+
if elem.bbox is not None:
|
|
45
|
+
# Normalize coordinates to [0.0, 1.0]
|
|
46
|
+
norm_x0 = max(0.0, min(1.0, elem.bbox.x0 / ref_w))
|
|
47
|
+
norm_y0 = max(0.0, min(1.0, elem.bbox.y0 / ref_h))
|
|
48
|
+
norm_x1 = max(0.0, min(1.0, elem.bbox.x1 / ref_w))
|
|
49
|
+
norm_y1 = max(0.0, min(1.0, elem.bbox.y1 / ref_h))
|
|
50
|
+
|
|
51
|
+
# Discretize into grid buckets
|
|
52
|
+
gx0 = min(grid_size - 1, int(norm_x0 * grid_size))
|
|
53
|
+
gy0 = min(grid_size - 1, int(norm_y0 * grid_size))
|
|
54
|
+
gx1 = min(grid_size - 1, int(norm_x1 * grid_size))
|
|
55
|
+
gy1 = min(grid_size - 1, int(norm_y1 * grid_size))
|
|
56
|
+
|
|
57
|
+
for r in range(gy0, gy1 + 1):
|
|
58
|
+
for c in range(gx0, gx1 + 1):
|
|
59
|
+
grid[r][c] += 1.0
|
|
60
|
+
|
|
61
|
+
else:
|
|
62
|
+
# Fallback for bbox-less elements: sequential normalized bucket
|
|
63
|
+
pos_ratio = idx / max(1, total_elements)
|
|
64
|
+
r = min(grid_size - 1, int(pos_ratio * grid_size))
|
|
65
|
+
c = 0
|
|
66
|
+
grid[r][c] += 1.0
|
|
67
|
+
|
|
68
|
+
# Normalize grid density to sum to 1.0 (if non-empty)
|
|
69
|
+
grid_sum = sum(sum(row) for row in grid)
|
|
70
|
+
if grid_sum > 0:
|
|
71
|
+
grid = [[round(val / grid_sum, 4) for val in row] for row in grid]
|
|
72
|
+
|
|
73
|
+
# Normalize type distribution
|
|
74
|
+
type_dist = (
|
|
75
|
+
{k: round(v / total_elements, 4) for k, v in sorted(type_counts.items())}
|
|
76
|
+
if total_elements > 0
|
|
77
|
+
else {}
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
avg_elements = round(total_elements / page_count, 2)
|
|
81
|
+
|
|
82
|
+
# Compute deterministic SHA-256 hash digest
|
|
83
|
+
payload: dict[str, Any] = {
|
|
84
|
+
"grid": grid,
|
|
85
|
+
"types": type_dist,
|
|
86
|
+
"page_count": page_count,
|
|
87
|
+
}
|
|
88
|
+
raw_bytes = json.dumps(payload, sort_keys=True).encode("utf-8")
|
|
89
|
+
hash_digest = hashlib.sha256(raw_bytes).hexdigest()
|
|
90
|
+
|
|
91
|
+
return LayoutFingerprint(
|
|
92
|
+
hash_digest=hash_digest,
|
|
93
|
+
spatial_grid=grid,
|
|
94
|
+
type_distribution=type_dist,
|
|
95
|
+
page_count=page_count,
|
|
96
|
+
avg_elements_per_page=avg_elements,
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _vector_cosine(v1: list[float], v2: list[float]) -> float:
|
|
101
|
+
"""Computes cosine similarity between two flat float vectors."""
|
|
102
|
+
dot = sum(a * b for a, b in zip(v1, v2))
|
|
103
|
+
norm1 = math.sqrt(sum(a * a for a in v1))
|
|
104
|
+
norm2 = math.sqrt(sum(b * b for b in v2))
|
|
105
|
+
if norm1 == 0.0 or norm2 == 0.0:
|
|
106
|
+
return 1.0 if norm1 == norm2 else 0.0
|
|
107
|
+
return max(0.0, min(1.0, dot / (norm1 * norm2)))
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def fingerprint_similarity(fp1: LayoutFingerprint, fp2: LayoutFingerprint) -> float:
|
|
111
|
+
"""Calculates a normalized similarity score in [0.0, 1.0] between two fingerprints.
|
|
112
|
+
Combines spatial layout alignment (70% weight) and element type distribution (30% weight).
|
|
113
|
+
"""
|
|
114
|
+
if fp1.hash_digest == fp2.hash_digest:
|
|
115
|
+
return 1.0
|
|
116
|
+
|
|
117
|
+
# 1. Flatten spatial grid
|
|
118
|
+
flat_grid1 = [val for row in fp1.spatial_grid for val in row]
|
|
119
|
+
flat_grid2 = [val for row in fp2.spatial_grid for val in row]
|
|
120
|
+
spatial_sim = _vector_cosine(flat_grid1, flat_grid2)
|
|
121
|
+
|
|
122
|
+
# 2. Type distribution similarity
|
|
123
|
+
all_keys = sorted(set(fp1.type_distribution.keys()) | set(fp2.type_distribution.keys()))
|
|
124
|
+
if not all_keys:
|
|
125
|
+
type_sim = 1.0
|
|
126
|
+
else:
|
|
127
|
+
v1 = [fp1.type_distribution.get(k, 0.0) for k in all_keys]
|
|
128
|
+
v2 = [fp2.type_distribution.get(k, 0.0) for k in all_keys]
|
|
129
|
+
type_sim = _vector_cosine(v1, v2)
|
|
130
|
+
|
|
131
|
+
# 3. Weighted total score
|
|
132
|
+
score = 0.7 * spatial_sim + 0.3 * type_sim
|
|
133
|
+
return round(max(0.0, min(1.0, score)), 4)
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Any, Literal
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel, Field
|
|
6
|
+
|
|
7
|
+
from universal_parser.adaptive.config_cache import ParserConfig
|
|
8
|
+
|
|
9
|
+
SignalType = Literal[
|
|
10
|
+
"missed_heading",
|
|
11
|
+
"false_heading",
|
|
12
|
+
"missed_table",
|
|
13
|
+
"false_table",
|
|
14
|
+
"merged_columns",
|
|
15
|
+
"split_columns",
|
|
16
|
+
"ocr_quality",
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class FeedbackSignal(BaseModel):
|
|
21
|
+
"""User-flagged or auto-detected parsing error signal."""
|
|
22
|
+
|
|
23
|
+
signal_type: SignalType
|
|
24
|
+
severity: float = 1.0 # Learning step multiplier (0.1 - 2.0)
|
|
25
|
+
details: dict[str, Any] = Field(default_factory=dict)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# Hard mathematical boundaries to prevent degenerate parameter drift
|
|
29
|
+
GUARD_RAILS: dict[str, tuple[float, float]] = {
|
|
30
|
+
"column_gap_threshold": (10.0, 120.0),
|
|
31
|
+
"heading_p95_ratio": (1.05, 3.00),
|
|
32
|
+
"heading_p85_ratio": (1.01, 2.50),
|
|
33
|
+
"min_table_confidence": (0.30, 0.95),
|
|
34
|
+
"ocr_dpi": (100.0, 300.0),
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class AdaptiveTuner:
|
|
39
|
+
"""Heuristic optimizer that adjusts ParserConfig parameters based on feedback signals."""
|
|
40
|
+
|
|
41
|
+
@staticmethod
|
|
42
|
+
def _clamp(param_name: str, value: float) -> float:
|
|
43
|
+
if param_name in GUARD_RAILS:
|
|
44
|
+
min_val, max_val = GUARD_RAILS[param_name]
|
|
45
|
+
return max(min_val, min(max_val, value))
|
|
46
|
+
return value
|
|
47
|
+
|
|
48
|
+
def tune(
|
|
49
|
+
self,
|
|
50
|
+
current_config: ParserConfig,
|
|
51
|
+
signals: list[FeedbackSignal],
|
|
52
|
+
base_step: float = 0.05,
|
|
53
|
+
) -> ParserConfig:
|
|
54
|
+
"""Calculates parameter adjustments from feedback signals while strictly respecting guardrails."""
|
|
55
|
+
cfg = current_config.model_copy(deep=True)
|
|
56
|
+
|
|
57
|
+
for sig in signals:
|
|
58
|
+
step = base_step * max(0.1, min(2.0, sig.severity))
|
|
59
|
+
|
|
60
|
+
if sig.signal_type == "missed_heading":
|
|
61
|
+
# Lower heading thresholds to make heading detection more sensitive
|
|
62
|
+
cfg.heading_p95_ratio -= step * 0.5
|
|
63
|
+
cfg.heading_p85_ratio -= step * 0.4
|
|
64
|
+
elif sig.signal_type == "false_heading":
|
|
65
|
+
# Raise heading thresholds to require larger font difference
|
|
66
|
+
cfg.heading_p95_ratio += step * 0.5
|
|
67
|
+
cfg.heading_p85_ratio += step * 0.4
|
|
68
|
+
elif sig.signal_type == "missed_table":
|
|
69
|
+
# Lower table confidence requirement or switch to stream mode
|
|
70
|
+
cfg.min_table_confidence -= step * 0.5
|
|
71
|
+
if cfg.min_table_confidence < 0.5:
|
|
72
|
+
cfg.table_detection_mode = "stream"
|
|
73
|
+
elif sig.signal_type == "false_table":
|
|
74
|
+
# Increase table confidence threshold
|
|
75
|
+
cfg.min_table_confidence += step * 0.5
|
|
76
|
+
elif sig.signal_type == "merged_columns":
|
|
77
|
+
# Columns were incorrectly merged -> reduce gap threshold to split columns more aggressively
|
|
78
|
+
cfg.column_gap_threshold -= step * 20.0
|
|
79
|
+
elif sig.signal_type == "split_columns":
|
|
80
|
+
# Single column was falsely split into multi-column -> increase gap threshold
|
|
81
|
+
cfg.column_gap_threshold += step * 20.0
|
|
82
|
+
elif sig.signal_type == "ocr_quality":
|
|
83
|
+
# Higher resolution for OCR rasterization
|
|
84
|
+
cfg.ocr_dpi = int(min(300, cfg.ocr_dpi + 50))
|
|
85
|
+
|
|
86
|
+
# Apply guardrail clamping and rounding
|
|
87
|
+
cfg.column_gap_threshold = round(
|
|
88
|
+
self._clamp("column_gap_threshold", cfg.column_gap_threshold), 2
|
|
89
|
+
)
|
|
90
|
+
cfg.heading_p95_ratio = round(self._clamp("heading_p95_ratio", cfg.heading_p95_ratio), 3)
|
|
91
|
+
cfg.heading_p85_ratio = round(self._clamp("heading_p85_ratio", cfg.heading_p85_ratio), 3)
|
|
92
|
+
cfg.min_table_confidence = round(
|
|
93
|
+
self._clamp("min_table_confidence", cfg.min_table_confidence), 3
|
|
94
|
+
)
|
|
95
|
+
cfg.ocr_dpi = int(self._clamp("ocr_dpi", float(cfg.ocr_dpi)))
|
|
96
|
+
|
|
97
|
+
return cfg
|
|
File without changes
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from universal_parser.core.router import get_extractor
|
|
6
|
+
from universal_parser.core.schema import Document, DocumentMetadata
|
|
7
|
+
from universal_parser.core.sniffer import sniff
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def parse(path: str | Path) -> Document:
|
|
11
|
+
"""
|
|
12
|
+
Parse any supported document into a Document object.
|
|
13
|
+
This is the only function external code needs to call.
|
|
14
|
+
Internally it:
|
|
15
|
+
1. Sniffs the file type via magic bytes
|
|
16
|
+
2. Looks up the registered extractor for that type
|
|
17
|
+
3. Streams Elements from the extractor into content_tree
|
|
18
|
+
4. Returns a validated Document
|
|
19
|
+
Args:
|
|
20
|
+
path: path to the document to parse
|
|
21
|
+
Returns:
|
|
22
|
+
Document — fully validated Pydantic model
|
|
23
|
+
Raises:
|
|
24
|
+
FileNotFoundError: if the file does not exist
|
|
25
|
+
ValueError: if the file type is unsupported (no extractor registered)
|
|
26
|
+
"""
|
|
27
|
+
path = Path(path)
|
|
28
|
+
|
|
29
|
+
if not path.exists():
|
|
30
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
31
|
+
|
|
32
|
+
# Step 1 — What is this file?
|
|
33
|
+
file_type = sniff(path)
|
|
34
|
+
|
|
35
|
+
# Step 2 — Do we have an extractor for it?
|
|
36
|
+
extractor = get_extractor(file_type)
|
|
37
|
+
if extractor is None:
|
|
38
|
+
raise ValueError(
|
|
39
|
+
f"Unsupported file type: {file_type.name} ({path.suffix}). "
|
|
40
|
+
f"No extractor registered for this format yet."
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
# Step 3 — Build the document shell
|
|
44
|
+
doc = Document(
|
|
45
|
+
metadata=DocumentMetadata(
|
|
46
|
+
file_name=path.name,
|
|
47
|
+
file_type=file_type.name.lower(),
|
|
48
|
+
)
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
# Step 4 — Stream elements from the extractor into content_tree
|
|
52
|
+
for element in extractor.stream(path):
|
|
53
|
+
doc.content_tree.append(element)
|
|
54
|
+
|
|
55
|
+
return doc
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from universal_parser.core.sniffer import FileType
|
|
4
|
+
from universal_parser.extractors.base import BaseExtractor
|
|
5
|
+
|
|
6
|
+
# Central registry: FileType -> Extractor class
|
|
7
|
+
# Extractors are added here as they are built, phase by phase.
|
|
8
|
+
# engine.py reads this — it never imports an extractor directly.
|
|
9
|
+
|
|
10
|
+
FORMAT_REGISTRY: dict[FileType, type[BaseExtractor]] = {}
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def register(extractor_cls: type[BaseExtractor]) -> type[BaseExtractor]:
|
|
14
|
+
"""
|
|
15
|
+
Decorator to register an extractor class into FORMAT_REGISTRY.
|
|
16
|
+
Usage — put this on any extractor class:
|
|
17
|
+
@register
|
|
18
|
+
class PDFExtractor(BaseExtractor):
|
|
19
|
+
supported_types = [FileType.PDF]
|
|
20
|
+
...
|
|
21
|
+
This automatically maps FileType.PDF -> PDFExtractor in the registry.
|
|
22
|
+
No manual entry needed in this file when adding a new format.
|
|
23
|
+
"""
|
|
24
|
+
for file_type in extractor_cls.supported_types:
|
|
25
|
+
FORMAT_REGISTRY[file_type] = extractor_cls
|
|
26
|
+
return extractor_cls
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def get_extractor(file_type: FileType) -> BaseExtractor | None:
|
|
30
|
+
"""
|
|
31
|
+
Look up and return an instantiated extractor for the given FileType.
|
|
32
|
+
Returns None if no extractor is registered for this type.
|
|
33
|
+
engine.py handles the None case — it never crashes here.
|
|
34
|
+
"""
|
|
35
|
+
extractor_cls = FORMAT_REGISTRY.get(file_type)
|
|
36
|
+
if extractor_cls is None:
|
|
37
|
+
return None
|
|
38
|
+
return extractor_cls()
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import uuid
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
from pydantic import BaseModel, Field
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class BBox(BaseModel):
|
|
10
|
+
"""Bounding box of an element on a page. Coordinates in points (PDF units)."""
|
|
11
|
+
|
|
12
|
+
x0: float
|
|
13
|
+
y0: float
|
|
14
|
+
x1: float
|
|
15
|
+
y1: float
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TableData(BaseModel):
|
|
19
|
+
"""Structured representation of a table."""
|
|
20
|
+
|
|
21
|
+
headers: list[str]
|
|
22
|
+
rows: list[list[str]]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class Element(BaseModel):
|
|
26
|
+
"""A single unit of content extracted from a document."""
|
|
27
|
+
|
|
28
|
+
element_id: str = Field(default_factory=lambda: str(uuid.uuid4()))
|
|
29
|
+
type: Literal["heading", "paragraph", "table", "figure", "list_item", "code_block"]
|
|
30
|
+
level: int | None = None # heading level: 1, 2, 3 etc. None for non-headings
|
|
31
|
+
text: str | None = None
|
|
32
|
+
page: int | None = None # 1-indexed page number
|
|
33
|
+
bbox: BBox | None = None
|
|
34
|
+
parent_id: str | None = (
|
|
35
|
+
None # links to a parent element (e.g. heading this paragraph belongs to)
|
|
36
|
+
)
|
|
37
|
+
data: TableData | None = None # only for type="table"
|
|
38
|
+
markdown_repr: str | None = None # pre-rendered markdown string of this element
|
|
39
|
+
confidence: float | None = None # 0.0 to 1.0, used for tables and OCR output
|
|
40
|
+
vlm_description: str | None = None # Phase 8 — optional VLM-generated caption for figures
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class DocumentMetadata(BaseModel):
|
|
44
|
+
"""Facts about the source file — not its content."""
|
|
45
|
+
|
|
46
|
+
file_name: str
|
|
47
|
+
file_type: str
|
|
48
|
+
page_count: int | None = None
|
|
49
|
+
has_scanned_pages: bool = False
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Document(BaseModel):
|
|
53
|
+
"""The top-level output object. This is what parse() returns."""
|
|
54
|
+
|
|
55
|
+
schema_version: str = "1.0"
|
|
56
|
+
doc_id: str = Field(default_factory=lambda: str(uuid.uuid4()))
|
|
57
|
+
metadata: DocumentMetadata
|
|
58
|
+
content_tree: list[Element] = Field(default_factory=list)
|