parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,109 @@
1
+ """Provider for YOLO-DocLayNet layout detection."""
2
+
3
+ from typing import Any
4
+
5
+ from parse_bench.inference.providers.base import ProviderPermanentError
6
+ from parse_bench.inference.providers.layoutdet.base import HFLayoutDetProvider
7
+ from parse_bench.inference.providers.registry import register_provider
8
+ from parse_bench.schemas.layout_detection_output import (
9
+ LayoutDetectionModel,
10
+ LayoutOutput,
11
+ LayoutPrediction,
12
+ YoloLabel,
13
+ )
14
+ from parse_bench.schemas.pipeline_io import InferenceResult, RawInferenceResult
15
+ from parse_bench.schemas.product import ProductType
16
+
17
+
18
+ @register_provider("yolo_layout")
19
+ class YoloLayoutProvider(HFLayoutDetProvider):
20
+ """
21
+ Provider for YOLO-DocLayNet layout detection model.
22
+
23
+ This provider uses the YOLO model trained on DocLayNet served on HuggingFace
24
+ inference endpoints for detecting document layout regions.
25
+
26
+ Response format:
27
+ {
28
+ "pred_boxes": [[x1, y1, x2, y2], ...],
29
+ "pred_classes": [class_id, ...],
30
+ "scores": [score, ...]
31
+ }
32
+ """
33
+
34
+ endpoint_url = "https://exqoktya7l52qu1r.us-east-1.aws.endpoints.huggingface.cloud"
35
+ model_type = LayoutDetectionModel.YOLO_DOCLAYNET
36
+
37
+ def __init__(
38
+ self,
39
+ provider_name: str,
40
+ base_config: dict[str, Any] | None = None,
41
+ ):
42
+ """Initialize the YOLO-DocLayNet layout detection provider."""
43
+ super().__init__(provider_name, base_config)
44
+
45
+ def _parse_response(self, response: dict[str, Any]) -> list[LayoutPrediction]:
46
+ """
47
+ Parse YOLO-DocLayNet response into layout predictions.
48
+
49
+ :param response: Raw JSON response with pred_boxes, pred_classes, scores
50
+ :return: List of unified LayoutPrediction objects
51
+ """
52
+ predictions: list[LayoutPrediction] = []
53
+
54
+ boxes = response.get("pred_boxes", [])
55
+ classes = response.get("pred_classes", [])
56
+ scores = response.get("scores", [])
57
+
58
+ for bbox, class_id, score in zip(boxes, classes, scores, strict=False):
59
+ # Convert class_id to YoloLabel enum
60
+ # Model outputs 0-indexed labels (0-10) that match YoloLabel directly
61
+ label = YoloLabel(class_id)
62
+ predictions.append(
63
+ LayoutPrediction(
64
+ bbox=bbox,
65
+ score=score,
66
+ label=str(int(label)),
67
+ provider_metadata={"label_name": label.name},
68
+ )
69
+ )
70
+
71
+ return predictions
72
+
73
+ def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
74
+ """
75
+ Normalize raw inference result to produce LayoutOutput.
76
+
77
+ :param raw_result: Raw inference result from run_inference()
78
+ :return: Inference result with both raw and normalized outputs
79
+ :raises ProviderError: For any normalization failures
80
+ """
81
+ if raw_result.product_type != ProductType.LAYOUT_DETECTION:
82
+ raise ProviderPermanentError(
83
+ f"{self.__class__.__name__} only supports LAYOUT_DETECTION product type, got {raw_result.product_type}"
84
+ )
85
+
86
+ # Parse the response into raw predictions
87
+ response = raw_result.raw_output.get("response", {})
88
+ raw_predictions = self._parse_response(response)
89
+
90
+ output = LayoutOutput(
91
+ task_type="layout_detection",
92
+ example_id=raw_result.request.example_id,
93
+ pipeline_name=raw_result.pipeline_name,
94
+ model=self.model_type,
95
+ image_width=max(int(raw_result.raw_output.get("image_width", 1)), 1),
96
+ image_height=max(int(raw_result.raw_output.get("image_height", 1)), 1),
97
+ predictions=raw_predictions,
98
+ )
99
+
100
+ return InferenceResult(
101
+ request=raw_result.request,
102
+ pipeline_name=raw_result.pipeline_name,
103
+ product_type=raw_result.product_type,
104
+ raw_output=raw_result.raw_output,
105
+ output=output,
106
+ started_at=raw_result.started_at,
107
+ completed_at=raw_result.completed_at,
108
+ latency_in_ms=raw_result.latency_in_ms,
109
+ )
@@ -0,0 +1,64 @@
1
+ """Parse providers — imported lazily to avoid requiring all SDKs."""
2
+
3
+ import importlib
4
+ import logging
5
+
6
+ logger = logging.getLogger(__name__)
7
+
8
+ _PROVIDER_MODULES = [
9
+ "amazon_nova",
10
+ "anthropic",
11
+ "azure_document_intelligence",
12
+ "chandra2",
13
+ "chunkr",
14
+ "databricks_ai_parse",
15
+ "datalab",
16
+ "deepseekocr2",
17
+ "docling",
18
+ "docling_serve",
19
+ "dots_ocr",
20
+ "extend_parse",
21
+ "falconocr",
22
+ "florin_parser_nano",
23
+ "gemma4",
24
+ "glm_zai",
25
+ "google",
26
+ "google_docai",
27
+ "granite_vision",
28
+ "infinity_parser2",
29
+ "kdl_frontier_nano",
30
+ "landingai",
31
+ "liteparse",
32
+ "markitdown",
33
+ "opendataloader",
34
+ "pdf_inspector",
35
+ "pymupdf4llm",
36
+ "rakedoc_nano",
37
+ "llamaparse",
38
+ "llamaparse_v2_normalization",
39
+ "mineru25",
40
+ "mineru2605pro",
41
+ "mineru_diffusion",
42
+ "mistral_ocr",
43
+ "nemotron_omni",
44
+ "openai",
45
+ "paddleocr",
46
+ "pulse",
47
+ "pymupdf",
48
+ "pypdf",
49
+ "qwen",
50
+ "reducto",
51
+ "surya2",
52
+ "tesseract",
53
+ "textract",
54
+ "unlimitedocr",
55
+ "unstructured",
56
+ "warp_ingest",
57
+ "oi_parser",
58
+ ]
59
+
60
+ for _mod in _PROVIDER_MODULES:
61
+ try:
62
+ importlib.import_module(f"parse_bench.inference.providers.parse.{_mod}")
63
+ except ImportError:
64
+ logger.debug("Skipping parse provider %s (missing dependency)", _mod)
@@ -0,0 +1,233 @@
1
+ """Common functionality for docling and docling_serve providers."""
2
+
3
+ from typing import Any
4
+
5
+ from docling_core.types.doc.document import DoclingDocument
6
+
7
+ from parse_bench.layout_label_mapping import (
8
+ UnknownRawLayoutLabelError,
9
+ map_docling_raw_label_to_canonical,
10
+ )
11
+ from parse_bench.schemas.parse_output import (
12
+ LayoutItemIR,
13
+ LayoutSegmentIR,
14
+ ParseLayoutPageIR,
15
+ )
16
+
17
+ _DOCLING_EXCLUDED_LAYOUT_LABELS = frozenset(
18
+ {
19
+ "empty_value",
20
+ "field_heading",
21
+ "field_hint",
22
+ "field_item",
23
+ "field_key",
24
+ "field_region",
25
+ "field_value",
26
+ "marker",
27
+ }
28
+ )
29
+ _DOCLING_TABLE_LABELS = frozenset({"document_index", "table"})
30
+ _DOCLING_IMAGE_LABELS = frozenset({"chart", "picture"})
31
+
32
+
33
+ def _normalize_docling_label(label: object) -> str | None:
34
+ if label is None:
35
+ return None
36
+ value = getattr(label, "value", label)
37
+ if not isinstance(value, str):
38
+ return None
39
+ return value.strip().lower()
40
+
41
+
42
+ def _should_include_docling_label(raw_label: str) -> bool:
43
+ if raw_label in _DOCLING_EXCLUDED_LAYOUT_LABELS:
44
+ return False
45
+ try:
46
+ map_docling_raw_label_to_canonical(raw_label)
47
+ except UnknownRawLayoutLabelError:
48
+ return False
49
+ return True
50
+
51
+
52
+ def _docling_item_type(raw_label: str) -> str:
53
+ if raw_label in _DOCLING_TABLE_LABELS:
54
+ return "table"
55
+ if raw_label in _DOCLING_IMAGE_LABELS:
56
+ return "image"
57
+ return "text"
58
+
59
+
60
+ def _extract_docling_item_value(item: Any, doc: DoclingDocument, raw_label: str) -> str:
61
+ item_type = _docling_item_type(raw_label)
62
+ if item_type == "image":
63
+ return ""
64
+
65
+ if item_type == "table" and hasattr(item, "export_to_html"):
66
+ try:
67
+ html = item.export_to_html(doc=doc, add_caption=True)
68
+ if isinstance(html, str):
69
+ return html
70
+ except Exception:
71
+ pass
72
+
73
+ text = getattr(item, "text", None)
74
+ if isinstance(text, str):
75
+ return text
76
+
77
+ if hasattr(item, "export_to_markdown"):
78
+ try:
79
+ markdown = item.export_to_markdown()
80
+ if isinstance(markdown, str):
81
+ return markdown
82
+ except Exception:
83
+ pass
84
+
85
+ return ""
86
+
87
+
88
+ def _normalize_docling_charspan(
89
+ charspan: object,
90
+ *,
91
+ text_length: int,
92
+ include_span: bool,
93
+ ) -> tuple[int | None, int | None]:
94
+ if not include_span or not isinstance(charspan, (list, tuple)) or len(charspan) != 2:
95
+ return (None, None)
96
+
97
+ start_raw, end_raw = charspan
98
+ if not isinstance(start_raw, int) or not isinstance(end_raw, int):
99
+ return (None, None)
100
+
101
+ start = max(0, min(start_raw, text_length))
102
+ end_exclusive = max(start, min(end_raw, text_length))
103
+ if end_exclusive <= start:
104
+ return (None, None)
105
+
106
+ # Docling charspan behaves like a Python slice [start, end).
107
+ return (start, end_exclusive - 1)
108
+
109
+
110
+ def _build_docling_segment(
111
+ *,
112
+ prov: Any,
113
+ raw_label: str,
114
+ page_width: float,
115
+ page_height: float,
116
+ include_span: bool,
117
+ text_length: int,
118
+ ) -> LayoutSegmentIR | None:
119
+ bbox = getattr(prov, "bbox", None)
120
+ if bbox is None or page_width <= 0 or page_height <= 0:
121
+ return None
122
+
123
+ bbox_top_left = bbox.to_top_left_origin(page_height=page_height)
124
+ width = bbox_top_left.r - bbox_top_left.l
125
+ height = bbox_top_left.b - bbox_top_left.t
126
+ if width <= 0 or height <= 0:
127
+ return None
128
+
129
+ start_index, end_index = _normalize_docling_charspan(
130
+ getattr(prov, "charspan", None),
131
+ text_length=text_length,
132
+ include_span=include_span,
133
+ )
134
+
135
+ return LayoutSegmentIR(
136
+ x=bbox_top_left.l / page_width,
137
+ y=bbox_top_left.t / page_height,
138
+ w=width / page_width,
139
+ h=height / page_height,
140
+ confidence=1.0,
141
+ label=raw_label,
142
+ start_index=start_index,
143
+ end_index=end_index,
144
+ )
145
+
146
+
147
+ def _merge_segments(segments: list[LayoutSegmentIR]) -> LayoutSegmentIR | None:
148
+ if not segments:
149
+ return None
150
+
151
+ x1 = min(segment.x for segment in segments)
152
+ y1 = min(segment.y for segment in segments)
153
+ x2 = max(segment.x + segment.w for segment in segments)
154
+ y2 = max(segment.y + segment.h for segment in segments)
155
+ return LayoutSegmentIR(
156
+ x=x1,
157
+ y=y1,
158
+ w=x2 - x1,
159
+ h=y2 - y1,
160
+ confidence=1.0,
161
+ label=segments[0].label,
162
+ )
163
+
164
+
165
+ def _build_docling_layout_pages(
166
+ *,
167
+ doc: DoclingDocument,
168
+ raw_pages: list[dict[str, Any]],
169
+ ) -> list[ParseLayoutPageIR]:
170
+ page_markdown_by_number: dict[int, str] = {}
171
+ for page_data in raw_pages:
172
+ page_number = page_data.get("page")
173
+ if isinstance(page_number, int) and page_number > 0:
174
+ page_markdown_by_number[page_number] = str(page_data.get("markdown", ""))
175
+
176
+ layout_pages: list[ParseLayoutPageIR] = []
177
+ for page_number in sorted(doc.pages.keys()):
178
+ page = doc.pages[page_number]
179
+ page_width = float(page.size.width)
180
+ page_height = float(page.size.height)
181
+ items: list[LayoutItemIR] = []
182
+
183
+ for item, _level in doc.iterate_items(page_no=page_number):
184
+ raw_label = _normalize_docling_label(getattr(item, "label", None))
185
+ if raw_label is None or not _should_include_docling_label(raw_label):
186
+ continue
187
+
188
+ item_type = _docling_item_type(raw_label)
189
+ item_value = _extract_docling_item_value(item, doc, raw_label)
190
+ include_span = item_type == "text"
191
+
192
+ page_provs = [
193
+ prov for prov in getattr(item, "prov", []) or [] if getattr(prov, "page_no", None) == page_number
194
+ ]
195
+ segments = [
196
+ segment
197
+ for prov in page_provs
198
+ if (
199
+ segment := _build_docling_segment(
200
+ prov=prov,
201
+ raw_label=raw_label,
202
+ page_width=page_width,
203
+ page_height=page_height,
204
+ include_span=include_span,
205
+ text_length=len(item_value),
206
+ )
207
+ )
208
+ is not None
209
+ ]
210
+ if not segments:
211
+ continue
212
+
213
+ merged_bbox = _merge_segments(segments)
214
+ items.append(
215
+ LayoutItemIR(
216
+ type=item_type,
217
+ value=item_value,
218
+ bbox=merged_bbox,
219
+ layout_segments=segments,
220
+ )
221
+ )
222
+
223
+ layout_pages.append(
224
+ ParseLayoutPageIR(
225
+ page_number=page_number,
226
+ width=page_width,
227
+ height=page_height,
228
+ md=page_markdown_by_number.get(page_number, ""),
229
+ items=items,
230
+ )
231
+ )
232
+
233
+ return layout_pages