universal-doc-parser 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. universal_doc_parser-1.0.0.dist-info/METADATA +692 -0
  2. universal_doc_parser-1.0.0.dist-info/RECORD +48 -0
  3. universal_doc_parser-1.0.0.dist-info/WHEEL +4 -0
  4. universal_doc_parser-1.0.0.dist-info/licenses/LICENSE +21 -0
  5. universal_parser/__init__.py +28 -0
  6. universal_parser/adaptive/__init__.py +17 -0
  7. universal_parser/adaptive/config_cache.py +127 -0
  8. universal_parser/adaptive/fingerprint.py +133 -0
  9. universal_parser/adaptive/tuner.py +97 -0
  10. universal_parser/core/__init__.py +0 -0
  11. universal_parser/core/engine.py +55 -0
  12. universal_parser/core/router.py +38 -0
  13. universal_parser/core/schema.py +58 -0
  14. universal_parser/core/sniffer.py +125 -0
  15. universal_parser/enrichment/__init__.py +0 -0
  16. universal_parser/enrichment/vlm_enricher.py +0 -0
  17. universal_parser/exports/__init__.py +1 -0
  18. universal_parser/exports/to_chunks.py +108 -0
  19. universal_parser/exports/to_graph.py +114 -0
  20. universal_parser/exports/to_markdown.py +31 -0
  21. universal_parser/extractors/__init__.py +0 -0
  22. universal_parser/extractors/base.py +39 -0
  23. universal_parser/extractors/images/__init__.py +0 -0
  24. universal_parser/extractors/images/scan_extractor.py +79 -0
  25. universal_parser/extractors/mail/__init__.py +0 -0
  26. universal_parser/extractors/mail/mail_extractor.py +206 -0
  27. universal_parser/extractors/office/__init__.py +0 -0
  28. universal_parser/extractors/office/docx_extractor.py +131 -0
  29. universal_parser/extractors/office/legacy_extractor.py +112 -0
  30. universal_parser/extractors/office/pptx_extractor.py +109 -0
  31. universal_parser/extractors/office/xlsx_extractor.py +93 -0
  32. universal_parser/extractors/pdf/__init__.py +0 -0
  33. universal_parser/extractors/pdf/native.py +331 -0
  34. universal_parser/extractors/pdf/tables.py +155 -0
  35. universal_parser/extractors/pdf/visual_onnx.py +0 -0
  36. universal_parser/extractors/structured/__init__.py +0 -0
  37. universal_parser/extractors/structured/csv_extractor.py +86 -0
  38. universal_parser/extractors/structured/json_xml_extractor.py +104 -0
  39. universal_parser/extractors/structured/parquet_extractor.py +59 -0
  40. universal_parser/extractors/web/__init__.py +1 -0
  41. universal_parser/extractors/web/epub_extractor.py +81 -0
  42. universal_parser/extractors/web/html_extractor.py +111 -0
  43. universal_parser/mcp/__init__.py +1 -0
  44. universal_parser/mcp/server.py +61 -0
  45. universal_parser/observability/__init__.py +12 -0
  46. universal_parser/observability/dashboard.py +231 -0
  47. universal_parser/observability/logger.py +0 -0
  48. universal_parser/observability/metrics.py +99 -0
@@ -0,0 +1,48 @@
1
+ universal_parser/__init__.py,sha256=aXTv6p_QoWsK5yyuRATxtnPkoTL3M_ju7AetSDRhbSY,1337
2
+ universal_parser/adaptive/__init__.py,sha256=-y0YQaC-GrsGJejduofpE4Cm7lzcZeogpk-THo49RAc,476
3
+ universal_parser/adaptive/config_cache.py,sha256=U4MQkO7SrNESpiR7UWw5dSulRqtUd7TEUj0we8ytmF4,4698
4
+ universal_parser/adaptive/fingerprint.py,sha256=hwnk85msBHCtRW4wekH5bFH2pEaqLxH1v1FJEJJ1eEQ,4720
5
+ universal_parser/adaptive/tuner.py,sha256=zT50xD2fs5nkYppI9DgzK5hzCzCwrU17P6Cx09uPdpY,3791
6
+ universal_parser/core/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ universal_parser/core/engine.py,sha256=ZLDVP03MMxCEhUfUVBNf4CSIvHq4fwVi2p9pXhZjvEg,1725
8
+ universal_parser/core/router.py,sha256=jJOE4qT3s9J-KlHjhuYZjT6bRKyHgeSLnM2QWYW1lk8,1387
9
+ universal_parser/core/schema.py,sha256=vyB5MKwrFrRxAuIH7g4BAq3X3PavJ2fAr1gthkLqvJI,1787
10
+ universal_parser/core/sniffer.py,sha256=GqUHRTJt-Fc8cXr-zm5ZwsGuUXFyf2oLCYwipCea2SE,3700
11
+ universal_parser/enrichment/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
12
+ universal_parser/enrichment/vlm_enricher.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
13
+ universal_parser/exports/__init__.py,sha256=-1b3-EAbnEgjgglQVPG0jPrzrAwabmMxzXIHBG62YrY,17
14
+ universal_parser/exports/to_chunks.py,sha256=wnEh0jD5YPrUeB9bwY-8ac1G1lc0wD8teHUDLjzSMak,3687
15
+ universal_parser/exports/to_graph.py,sha256=AuqrtDkd270VL_leWWIJ9_Y22IUGWnW5GlZbUMHH6lw,3262
16
+ universal_parser/exports/to_markdown.py,sha256=1_rgNL2GD3SQNdZXBtUI98w7ZxnMcYojiZ0gH0pLaig,985
17
+ universal_parser/extractors/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
18
+ universal_parser/extractors/base.py,sha256=Vslbmb0fdd2e2a5j-e8GZX7pZBXMFq4yogQLhW0kf2c,1400
19
+ universal_parser/extractors/images/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
20
+ universal_parser/extractors/images/scan_extractor.py,sha256=ftD2uitFlrAL7YmK8xyhJHeuMI0UWhwLfb2jUV3fZ8Y,2601
21
+ universal_parser/extractors/mail/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
22
+ universal_parser/extractors/mail/mail_extractor.py,sha256=dpeDkdLERWFdjCCU4tkglO26iLdXbVDyLzPXuDr04R0,7530
23
+ universal_parser/extractors/office/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
24
+ universal_parser/extractors/office/docx_extractor.py,sha256=pRfQjTd4P-W3iy5SNwf8HeECsQTthTf4_mbJQk4hHaI,4458
25
+ universal_parser/extractors/office/legacy_extractor.py,sha256=Dq0c0oCZMoENie2qND1QoI_Qv1yUzijbcRTxYCa7rZY,3843
26
+ universal_parser/extractors/office/pptx_extractor.py,sha256=7WxDmpQAxK8uYPe1Hwpkjhbun40nYiW4ehHaSOuNw6c,4074
27
+ universal_parser/extractors/office/xlsx_extractor.py,sha256=Z06ayb0Z37lJVlgglanhmiadCogwwpKEBCN_5bewd7c,3438
28
+ universal_parser/extractors/pdf/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
29
+ universal_parser/extractors/pdf/native.py,sha256=B36CvFYA22MYCQa9xmdnKsxsu43gxOTh2lLjxbgT28c,13352
30
+ universal_parser/extractors/pdf/tables.py,sha256=Xqhcfq1uA1Q3sV2oXzVmUwDBxsStdtPbmMIRJfBBBTM,5501
31
+ universal_parser/extractors/pdf/visual_onnx.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
32
+ universal_parser/extractors/structured/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
33
+ universal_parser/extractors/structured/csv_extractor.py,sha256=YKZgjyfOJdl-x8W5Q4lKam82QtwFRCu92XxSSBTnXUs,2999
34
+ universal_parser/extractors/structured/json_xml_extractor.py,sha256=MNVOnJxLoWuf4s20PPRzi4aPOC6bar-EONsQX32xFb4,3835
35
+ universal_parser/extractors/structured/parquet_extractor.py,sha256=RT6WUi1aMmtGYusV5F7yd7b4Y3vxGduykVrKvLN-YCw,1912
36
+ universal_parser/extractors/web/__init__.py,sha256=qgz0BE4177sP3FVQqDifYgNKejYD-D1Dbq-YTY0we4w,17
37
+ universal_parser/extractors/web/epub_extractor.py,sha256=9H5CI_Y7k046_ho38PojRi-iqxEr3b4XQ-w1Ii3_Dd8,2836
38
+ universal_parser/extractors/web/html_extractor.py,sha256=7Erx1oncwCLmKx3Iz1gb2X8zh7mx5IqpQN5d_rMllHM,3977
39
+ universal_parser/mcp/__init__.py,sha256=IjzqjjmYctuyTAGB4hZpd93bNmu0EMlHLTimYnIm42M,14
40
+ universal_parser/mcp/server.py,sha256=ccyr3Sd2F2GLf7_lKwygZsR0HICQoesALt4sZGXqpq4,1848
41
+ universal_parser/observability/__init__.py,sha256=6LpvpSKaK_W-UtEIfKIYF3VMWyjSlhByBIMW4WM-i3w,312
42
+ universal_parser/observability/dashboard.py,sha256=vNpTAhDfaQf_tk3-5i81dMJggzh9IH-OrCho6InoRMU,7393
43
+ universal_parser/observability/logger.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
44
+ universal_parser/observability/metrics.py,sha256=NQEFOXDKVH2QkHV4lrhebu7qLRQgQK9dgah7PHIvJn4,3084
45
+ universal_doc_parser-1.0.0.dist-info/METADATA,sha256=Q5dxXh6tz_XdGPkmbCLl6NP-X_U6iv3AFJmaQ1C7giw,26771
46
+ universal_doc_parser-1.0.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
47
+ universal_doc_parser-1.0.0.dist-info/licenses/LICENSE,sha256=TBBq_QxVzFx6N4MBDqm1ipgIA_nmeP0DJlLugIUK1ws,1069
48
+ universal_doc_parser-1.0.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Karan Shelar
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,28 @@
1
+ from __future__ import annotations
2
+
3
+ __version__ = "1.0.0"
4
+
5
+ import universal_parser.extractors.images.scan_extractor
6
+ import universal_parser.extractors.mail.mail_extractor
7
+ import universal_parser.extractors.office.docx_extractor
8
+ import universal_parser.extractors.office.legacy_extractor
9
+ import universal_parser.extractors.office.pptx_extractor
10
+ import universal_parser.extractors.office.xlsx_extractor
11
+
12
+ # Import all extractors to trigger their @register decorators.
13
+ # Without this, the router registry remains empty until the files are imported.
14
+ import universal_parser.extractors.pdf.native
15
+ import universal_parser.extractors.structured.csv_extractor
16
+ import universal_parser.extractors.structured.json_xml_extractor
17
+ import universal_parser.extractors.structured.parquet_extractor
18
+ import universal_parser.extractors.web.epub_extractor
19
+ import universal_parser.extractors.web.html_extractor # noqa: F401
20
+
21
+ # Import the main entry point to expose it at the root of the package
22
+ from universal_parser.core.engine import parse
23
+ from universal_parser.exports.to_chunks import to_chunks
24
+ from universal_parser.exports.to_graph import to_graph
25
+ from universal_parser.exports.to_markdown import to_markdown
26
+
27
+ # Define what is exposed when doing: from universal_parser import *
28
+ __all__ = ["__version__", "parse", "to_chunks", "to_graph", "to_markdown"]
@@ -0,0 +1,17 @@
1
+ from universal_parser.adaptive.config_cache import ParserConfig, TemplateConfigCache
2
+ from universal_parser.adaptive.fingerprint import (
3
+ LayoutFingerprint,
4
+ compute_fingerprint,
5
+ fingerprint_similarity,
6
+ )
7
+ from universal_parser.adaptive.tuner import AdaptiveTuner, FeedbackSignal
8
+
9
+ __all__ = [
10
+ "AdaptiveTuner",
11
+ "FeedbackSignal",
12
+ "LayoutFingerprint",
13
+ "ParserConfig",
14
+ "TemplateConfigCache",
15
+ "compute_fingerprint",
16
+ "fingerprint_similarity",
17
+ ]
@@ -0,0 +1,127 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from collections import OrderedDict
5
+ from pathlib import Path
6
+ from threading import Lock
7
+ from typing import Any
8
+
9
+ from pydantic import BaseModel, Field
10
+
11
+ from universal_parser.adaptive.fingerprint import (
12
+ LayoutFingerprint,
13
+ fingerprint_similarity,
14
+ )
15
+
16
+
17
+ class ParserConfig(BaseModel):
18
+ """Hyperparameters and configuration tuned for a specific document layout template."""
19
+
20
+ column_gap_threshold: float = 30.0
21
+ heading_p95_ratio: float = 1.30
22
+ heading_p85_ratio: float = 1.15
23
+ table_detection_mode: str = "auto" # "lattice", "stream", "auto"
24
+ ocr_dpi: int = 150
25
+ min_table_confidence: float = 0.60
26
+ custom_overrides: dict[str, Any] = Field(default_factory=dict)
27
+
28
+
29
+ class TemplateConfigCache:
30
+ """Thread-safe, LRU cache for template fingerprints and parser configurations."""
31
+
32
+ def __init__(self, capacity: int = 1000) -> None:
33
+ self.capacity = capacity
34
+ self._cache: OrderedDict[str, tuple[LayoutFingerprint, ParserConfig]] = OrderedDict()
35
+ self._lock = Lock()
36
+
37
+ def get_exact(self, hash_digest: str) -> ParserConfig | None:
38
+ """Retrieves matching ParserConfig strictly by exact SHA-256 hash."""
39
+ with self._lock:
40
+ if hash_digest in self._cache:
41
+ self._cache.move_to_end(hash_digest)
42
+ _, config = self._cache[hash_digest]
43
+ return config.model_copy(deep=True)
44
+ return None
45
+
46
+ def get(
47
+ self, fingerprint: LayoutFingerprint, similarity_threshold: float = 0.85
48
+ ) -> tuple[str, ParserConfig] | None:
49
+ """Retrieves matching ParserConfig either by exact hash or best fuzzy match >= similarity_threshold."""
50
+ with self._lock:
51
+ # 1. Exact match
52
+ if fingerprint.hash_digest in self._cache:
53
+ self._cache.move_to_end(fingerprint.hash_digest)
54
+ _, config = self._cache[fingerprint.hash_digest]
55
+ return fingerprint.hash_digest, config.model_copy(deep=True)
56
+
57
+ # 2. Fuzzy similarity search
58
+ best_match_key: str | None = None
59
+ best_score = 0.0
60
+ best_config: ParserConfig | None = None
61
+
62
+ for key, (cached_fp, cached_cfg) in self._cache.items():
63
+ sim = fingerprint_similarity(fingerprint, cached_fp)
64
+ if sim > best_score:
65
+ best_score = sim
66
+ best_match_key = key
67
+ best_config = cached_cfg
68
+
69
+ if (
70
+ best_match_key is not None
71
+ and best_score >= similarity_threshold
72
+ and best_config is not None
73
+ ):
74
+ self._cache.move_to_end(best_match_key)
75
+ return best_match_key, best_config.model_copy(deep=True)
76
+
77
+ return None
78
+
79
+ def set(self, fingerprint: LayoutFingerprint, config: ParserConfig) -> None:
80
+ """Saves or updates a template fingerprint and its tuned configuration."""
81
+ with self._lock:
82
+ if fingerprint.hash_digest in self._cache:
83
+ self._cache.move_to_end(fingerprint.hash_digest)
84
+ self._cache[fingerprint.hash_digest] = (
85
+ fingerprint,
86
+ config.model_copy(deep=True),
87
+ )
88
+ if len(self._cache) > self.capacity:
89
+ self._cache.popitem(last=False)
90
+
91
+ def save(self, file_path: str | Path) -> None:
92
+ """Serializes cached template configurations to a JSON file."""
93
+ path = Path(file_path)
94
+ path.parent.mkdir(parents=True, exist_ok=True)
95
+ with self._lock:
96
+ data = {}
97
+ for digest, (fp, cfg) in self._cache.items():
98
+ data[digest] = {
99
+ "fingerprint": fp.model_dump(),
100
+ "config": cfg.model_dump(),
101
+ }
102
+ with path.open("w", encoding="utf-8") as f:
103
+ json.dump(data, f, indent=2)
104
+
105
+ def load(self, file_path: str | Path) -> None:
106
+ """Loads template configurations from a JSON file."""
107
+ path = Path(file_path)
108
+ if not path.is_file():
109
+ return
110
+
111
+ with path.open("r", encoding="utf-8") as f:
112
+ data = json.load(f)
113
+
114
+ with self._lock:
115
+ self._cache.clear()
116
+ for digest, payload in data.items():
117
+ fp = LayoutFingerprint.model_validate(payload["fingerprint"])
118
+ cfg = ParserConfig.model_validate(payload["config"])
119
+ self._cache[digest] = (fp, cfg)
120
+
121
+ def clear(self) -> None:
122
+ with self._lock:
123
+ self._cache.clear()
124
+
125
+ def __len__(self) -> int:
126
+ with self._lock:
127
+ return len(self._cache)
@@ -0,0 +1,133 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ import math
6
+ from typing import Any
7
+
8
+ from pydantic import BaseModel, Field
9
+
10
+ from universal_parser.core.schema import Document
11
+
12
+
13
+ class LayoutFingerprint(BaseModel):
14
+ """Structural layout fingerprint of a document or page template."""
15
+
16
+ hash_digest: str
17
+ spatial_grid: list[list[float]] = Field(default_factory=list)
18
+ type_distribution: dict[str, float] = Field(default_factory=dict)
19
+ page_count: int = 1
20
+ avg_elements_per_page: float = 0.0
21
+
22
+
23
+ def compute_fingerprint(doc: Document, grid_size: int = 10) -> LayoutFingerprint:
24
+ """Computes a content-agnostic structural layout fingerprint from a Document.
25
+ Quantizes spatial bounding boxes into an N x N normalized occupancy matrix
26
+ and aggregates element type proportions.
27
+ """
28
+ grid = [[0.0 for _ in range(grid_size)] for _ in range(grid_size)]
29
+ type_counts: dict[str, int] = {}
30
+ total_elements = len(doc.content_tree)
31
+
32
+ page_count = doc.metadata.page_count or 1
33
+ if page_count <= 0:
34
+ page_count = 1
35
+
36
+ # Standard document dimension reference (Letter / A4 approx: 612 x 792 pt)
37
+ ref_w = 612.0
38
+ ref_h = 792.0
39
+
40
+ for idx, elem in enumerate(doc.content_tree):
41
+ t = elem.type
42
+ type_counts[t] = type_counts.get(t, 0) + 1
43
+
44
+ if elem.bbox is not None:
45
+ # Normalize coordinates to [0.0, 1.0]
46
+ norm_x0 = max(0.0, min(1.0, elem.bbox.x0 / ref_w))
47
+ norm_y0 = max(0.0, min(1.0, elem.bbox.y0 / ref_h))
48
+ norm_x1 = max(0.0, min(1.0, elem.bbox.x1 / ref_w))
49
+ norm_y1 = max(0.0, min(1.0, elem.bbox.y1 / ref_h))
50
+
51
+ # Discretize into grid buckets
52
+ gx0 = min(grid_size - 1, int(norm_x0 * grid_size))
53
+ gy0 = min(grid_size - 1, int(norm_y0 * grid_size))
54
+ gx1 = min(grid_size - 1, int(norm_x1 * grid_size))
55
+ gy1 = min(grid_size - 1, int(norm_y1 * grid_size))
56
+
57
+ for r in range(gy0, gy1 + 1):
58
+ for c in range(gx0, gx1 + 1):
59
+ grid[r][c] += 1.0
60
+
61
+ else:
62
+ # Fallback for bbox-less elements: sequential normalized bucket
63
+ pos_ratio = idx / max(1, total_elements)
64
+ r = min(grid_size - 1, int(pos_ratio * grid_size))
65
+ c = 0
66
+ grid[r][c] += 1.0
67
+
68
+ # Normalize grid density to sum to 1.0 (if non-empty)
69
+ grid_sum = sum(sum(row) for row in grid)
70
+ if grid_sum > 0:
71
+ grid = [[round(val / grid_sum, 4) for val in row] for row in grid]
72
+
73
+ # Normalize type distribution
74
+ type_dist = (
75
+ {k: round(v / total_elements, 4) for k, v in sorted(type_counts.items())}
76
+ if total_elements > 0
77
+ else {}
78
+ )
79
+
80
+ avg_elements = round(total_elements / page_count, 2)
81
+
82
+ # Compute deterministic SHA-256 hash digest
83
+ payload: dict[str, Any] = {
84
+ "grid": grid,
85
+ "types": type_dist,
86
+ "page_count": page_count,
87
+ }
88
+ raw_bytes = json.dumps(payload, sort_keys=True).encode("utf-8")
89
+ hash_digest = hashlib.sha256(raw_bytes).hexdigest()
90
+
91
+ return LayoutFingerprint(
92
+ hash_digest=hash_digest,
93
+ spatial_grid=grid,
94
+ type_distribution=type_dist,
95
+ page_count=page_count,
96
+ avg_elements_per_page=avg_elements,
97
+ )
98
+
99
+
100
+ def _vector_cosine(v1: list[float], v2: list[float]) -> float:
101
+ """Computes cosine similarity between two flat float vectors."""
102
+ dot = sum(a * b for a, b in zip(v1, v2))
103
+ norm1 = math.sqrt(sum(a * a for a in v1))
104
+ norm2 = math.sqrt(sum(b * b for b in v2))
105
+ if norm1 == 0.0 or norm2 == 0.0:
106
+ return 1.0 if norm1 == norm2 else 0.0
107
+ return max(0.0, min(1.0, dot / (norm1 * norm2)))
108
+
109
+
110
+ def fingerprint_similarity(fp1: LayoutFingerprint, fp2: LayoutFingerprint) -> float:
111
+ """Calculates a normalized similarity score in [0.0, 1.0] between two fingerprints.
112
+ Combines spatial layout alignment (70% weight) and element type distribution (30% weight).
113
+ """
114
+ if fp1.hash_digest == fp2.hash_digest:
115
+ return 1.0
116
+
117
+ # 1. Flatten spatial grid
118
+ flat_grid1 = [val for row in fp1.spatial_grid for val in row]
119
+ flat_grid2 = [val for row in fp2.spatial_grid for val in row]
120
+ spatial_sim = _vector_cosine(flat_grid1, flat_grid2)
121
+
122
+ # 2. Type distribution similarity
123
+ all_keys = sorted(set(fp1.type_distribution.keys()) | set(fp2.type_distribution.keys()))
124
+ if not all_keys:
125
+ type_sim = 1.0
126
+ else:
127
+ v1 = [fp1.type_distribution.get(k, 0.0) for k in all_keys]
128
+ v2 = [fp2.type_distribution.get(k, 0.0) for k in all_keys]
129
+ type_sim = _vector_cosine(v1, v2)
130
+
131
+ # 3. Weighted total score
132
+ score = 0.7 * spatial_sim + 0.3 * type_sim
133
+ return round(max(0.0, min(1.0, score)), 4)
@@ -0,0 +1,97 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any, Literal
4
+
5
+ from pydantic import BaseModel, Field
6
+
7
+ from universal_parser.adaptive.config_cache import ParserConfig
8
+
9
+ SignalType = Literal[
10
+ "missed_heading",
11
+ "false_heading",
12
+ "missed_table",
13
+ "false_table",
14
+ "merged_columns",
15
+ "split_columns",
16
+ "ocr_quality",
17
+ ]
18
+
19
+
20
+ class FeedbackSignal(BaseModel):
21
+ """User-flagged or auto-detected parsing error signal."""
22
+
23
+ signal_type: SignalType
24
+ severity: float = 1.0 # Learning step multiplier (0.1 - 2.0)
25
+ details: dict[str, Any] = Field(default_factory=dict)
26
+
27
+
28
+ # Hard mathematical boundaries to prevent degenerate parameter drift
29
+ GUARD_RAILS: dict[str, tuple[float, float]] = {
30
+ "column_gap_threshold": (10.0, 120.0),
31
+ "heading_p95_ratio": (1.05, 3.00),
32
+ "heading_p85_ratio": (1.01, 2.50),
33
+ "min_table_confidence": (0.30, 0.95),
34
+ "ocr_dpi": (100.0, 300.0),
35
+ }
36
+
37
+
38
+ class AdaptiveTuner:
39
+ """Heuristic optimizer that adjusts ParserConfig parameters based on feedback signals."""
40
+
41
+ @staticmethod
42
+ def _clamp(param_name: str, value: float) -> float:
43
+ if param_name in GUARD_RAILS:
44
+ min_val, max_val = GUARD_RAILS[param_name]
45
+ return max(min_val, min(max_val, value))
46
+ return value
47
+
48
+ def tune(
49
+ self,
50
+ current_config: ParserConfig,
51
+ signals: list[FeedbackSignal],
52
+ base_step: float = 0.05,
53
+ ) -> ParserConfig:
54
+ """Calculates parameter adjustments from feedback signals while strictly respecting guardrails."""
55
+ cfg = current_config.model_copy(deep=True)
56
+
57
+ for sig in signals:
58
+ step = base_step * max(0.1, min(2.0, sig.severity))
59
+
60
+ if sig.signal_type == "missed_heading":
61
+ # Lower heading thresholds to make heading detection more sensitive
62
+ cfg.heading_p95_ratio -= step * 0.5
63
+ cfg.heading_p85_ratio -= step * 0.4
64
+ elif sig.signal_type == "false_heading":
65
+ # Raise heading thresholds to require larger font difference
66
+ cfg.heading_p95_ratio += step * 0.5
67
+ cfg.heading_p85_ratio += step * 0.4
68
+ elif sig.signal_type == "missed_table":
69
+ # Lower table confidence requirement or switch to stream mode
70
+ cfg.min_table_confidence -= step * 0.5
71
+ if cfg.min_table_confidence < 0.5:
72
+ cfg.table_detection_mode = "stream"
73
+ elif sig.signal_type == "false_table":
74
+ # Increase table confidence threshold
75
+ cfg.min_table_confidence += step * 0.5
76
+ elif sig.signal_type == "merged_columns":
77
+ # Columns were incorrectly merged -> reduce gap threshold to split columns more aggressively
78
+ cfg.column_gap_threshold -= step * 20.0
79
+ elif sig.signal_type == "split_columns":
80
+ # Single column was falsely split into multi-column -> increase gap threshold
81
+ cfg.column_gap_threshold += step * 20.0
82
+ elif sig.signal_type == "ocr_quality":
83
+ # Higher resolution for OCR rasterization
84
+ cfg.ocr_dpi = int(min(300, cfg.ocr_dpi + 50))
85
+
86
+ # Apply guardrail clamping and rounding
87
+ cfg.column_gap_threshold = round(
88
+ self._clamp("column_gap_threshold", cfg.column_gap_threshold), 2
89
+ )
90
+ cfg.heading_p95_ratio = round(self._clamp("heading_p95_ratio", cfg.heading_p95_ratio), 3)
91
+ cfg.heading_p85_ratio = round(self._clamp("heading_p85_ratio", cfg.heading_p85_ratio), 3)
92
+ cfg.min_table_confidence = round(
93
+ self._clamp("min_table_confidence", cfg.min_table_confidence), 3
94
+ )
95
+ cfg.ocr_dpi = int(self._clamp("ocr_dpi", float(cfg.ocr_dpi)))
96
+
97
+ return cfg
File without changes
@@ -0,0 +1,55 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+
5
+ from universal_parser.core.router import get_extractor
6
+ from universal_parser.core.schema import Document, DocumentMetadata
7
+ from universal_parser.core.sniffer import sniff
8
+
9
+
10
+ def parse(path: str | Path) -> Document:
11
+ """
12
+ Parse any supported document into a Document object.
13
+ This is the only function external code needs to call.
14
+ Internally it:
15
+ 1. Sniffs the file type via magic bytes
16
+ 2. Looks up the registered extractor for that type
17
+ 3. Streams Elements from the extractor into content_tree
18
+ 4. Returns a validated Document
19
+ Args:
20
+ path: path to the document to parse
21
+ Returns:
22
+ Document — fully validated Pydantic model
23
+ Raises:
24
+ FileNotFoundError: if the file does not exist
25
+ ValueError: if the file type is unsupported (no extractor registered)
26
+ """
27
+ path = Path(path)
28
+
29
+ if not path.exists():
30
+ raise FileNotFoundError(f"File not found: {path}")
31
+
32
+ # Step 1 — What is this file?
33
+ file_type = sniff(path)
34
+
35
+ # Step 2 — Do we have an extractor for it?
36
+ extractor = get_extractor(file_type)
37
+ if extractor is None:
38
+ raise ValueError(
39
+ f"Unsupported file type: {file_type.name} ({path.suffix}). "
40
+ f"No extractor registered for this format yet."
41
+ )
42
+
43
+ # Step 3 — Build the document shell
44
+ doc = Document(
45
+ metadata=DocumentMetadata(
46
+ file_name=path.name,
47
+ file_type=file_type.name.lower(),
48
+ )
49
+ )
50
+
51
+ # Step 4 — Stream elements from the extractor into content_tree
52
+ for element in extractor.stream(path):
53
+ doc.content_tree.append(element)
54
+
55
+ return doc
@@ -0,0 +1,38 @@
1
+ from __future__ import annotations
2
+
3
+ from universal_parser.core.sniffer import FileType
4
+ from universal_parser.extractors.base import BaseExtractor
5
+
6
+ # Central registry: FileType -> Extractor class
7
+ # Extractors are added here as they are built, phase by phase.
8
+ # engine.py reads this — it never imports an extractor directly.
9
+
10
+ FORMAT_REGISTRY: dict[FileType, type[BaseExtractor]] = {}
11
+
12
+
13
+ def register(extractor_cls: type[BaseExtractor]) -> type[BaseExtractor]:
14
+ """
15
+ Decorator to register an extractor class into FORMAT_REGISTRY.
16
+ Usage — put this on any extractor class:
17
+ @register
18
+ class PDFExtractor(BaseExtractor):
19
+ supported_types = [FileType.PDF]
20
+ ...
21
+ This automatically maps FileType.PDF -> PDFExtractor in the registry.
22
+ No manual entry needed in this file when adding a new format.
23
+ """
24
+ for file_type in extractor_cls.supported_types:
25
+ FORMAT_REGISTRY[file_type] = extractor_cls
26
+ return extractor_cls
27
+
28
+
29
+ def get_extractor(file_type: FileType) -> BaseExtractor | None:
30
+ """
31
+ Look up and return an instantiated extractor for the given FileType.
32
+ Returns None if no extractor is registered for this type.
33
+ engine.py handles the None case — it never crashes here.
34
+ """
35
+ extractor_cls = FORMAT_REGISTRY.get(file_type)
36
+ if extractor_cls is None:
37
+ return None
38
+ return extractor_cls()
@@ -0,0 +1,58 @@
1
+ from __future__ import annotations
2
+
3
+ import uuid
4
+ from typing import Literal
5
+
6
+ from pydantic import BaseModel, Field
7
+
8
+
9
+ class BBox(BaseModel):
10
+ """Bounding box of an element on a page. Coordinates in points (PDF units)."""
11
+
12
+ x0: float
13
+ y0: float
14
+ x1: float
15
+ y1: float
16
+
17
+
18
+ class TableData(BaseModel):
19
+ """Structured representation of a table."""
20
+
21
+ headers: list[str]
22
+ rows: list[list[str]]
23
+
24
+
25
+ class Element(BaseModel):
26
+ """A single unit of content extracted from a document."""
27
+
28
+ element_id: str = Field(default_factory=lambda: str(uuid.uuid4()))
29
+ type: Literal["heading", "paragraph", "table", "figure", "list_item", "code_block"]
30
+ level: int | None = None # heading level: 1, 2, 3 etc. None for non-headings
31
+ text: str | None = None
32
+ page: int | None = None # 1-indexed page number
33
+ bbox: BBox | None = None
34
+ parent_id: str | None = (
35
+ None # links to a parent element (e.g. heading this paragraph belongs to)
36
+ )
37
+ data: TableData | None = None # only for type="table"
38
+ markdown_repr: str | None = None # pre-rendered markdown string of this element
39
+ confidence: float | None = None # 0.0 to 1.0, used for tables and OCR output
40
+ vlm_description: str | None = None # Phase 8 — optional VLM-generated caption for figures
41
+
42
+
43
+ class DocumentMetadata(BaseModel):
44
+ """Facts about the source file — not its content."""
45
+
46
+ file_name: str
47
+ file_type: str
48
+ page_count: int | None = None
49
+ has_scanned_pages: bool = False
50
+
51
+
52
+ class Document(BaseModel):
53
+ """The top-level output object. This is what parse() returns."""
54
+
55
+ schema_version: str = "1.0"
56
+ doc_id: str = Field(default_factory=lambda: str(uuid.uuid4()))
57
+ metadata: DocumentMetadata
58
+ content_tree: list[Element] = Field(default_factory=list)