parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,289 @@
1
+ """Provider for Docling via the official docling-serve HTTP API."""
2
+
3
+ import base64
4
+ import os
5
+ from datetime import datetime
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ import requests
10
+ from docling_core.transforms.serializer.html import HTMLTableSerializer
11
+ from docling_core.transforms.serializer.markdown import MarkdownDocSerializer
12
+ from docling_core.types.doc.base import ImageRefMode
13
+ from docling_core.types.doc.document import DoclingDocument
14
+
15
+ from parse_bench.inference.providers.base import (
16
+ Provider,
17
+ ProviderConfigError,
18
+ ProviderPermanentError,
19
+ ProviderRateLimitError,
20
+ ProviderTransientError,
21
+ )
22
+ from parse_bench.inference.providers.parse._docling_common import _build_docling_layout_pages
23
+ from parse_bench.inference.providers.registry import register_provider
24
+ from parse_bench.schemas.parse_output import (
25
+ PageIR,
26
+ ParseLayoutPageIR,
27
+ ParseOutput,
28
+ )
29
+ from parse_bench.schemas.pipeline import PipelineSpec
30
+ from parse_bench.schemas.pipeline_io import (
31
+ InferenceRequest,
32
+ InferenceResult,
33
+ RawInferenceResult,
34
+ )
35
+ from parse_bench.schemas.product import ProductType
36
+
37
+ _MD_PAGE_BREAK_PLACEHOLDER = "<!-- page-break -->"
38
+
39
+
40
+ @register_provider("docling_serve")
41
+ class DoclingServeProvider(Provider):
42
+ """
43
+ Provider for Docling PDF parsing via the official docling-serve HTTP API.
44
+
45
+ This provider sends PDFs to the docling-serve HTTP API endpoint and returns markdown
46
+ with tables formatted as HTML. It was tested with docling-serve v1.17.0.
47
+ """
48
+
49
+ def __init__(
50
+ self,
51
+ provider_name: str,
52
+ base_config: dict[str, Any] | None = None,
53
+ ):
54
+ """
55
+ Initialize the Docling Serve provider.
56
+
57
+ Args:
58
+ provider_name: Name of the provider
59
+ base_config: Optional configuration with:
60
+ - `api_key`: Optional bearer token for the endpoint
61
+ - `endpoint_url`: Endpoint URL (required)
62
+ - `timeout`: Request timeout in seconds (default: 120)
63
+ """
64
+ super().__init__(provider_name, base_config)
65
+
66
+ self._api_key = self.base_config.get("api_key") or os.getenv("DOCLING_SERVE_API_KEY") or ""
67
+
68
+ # Get endpoint URL (from config or env var)
69
+ self._endpoint_url = self.base_config.get("endpoint_url") or os.getenv("DOCLING_SERVE_ENDPOINT_URL")
70
+ self._endpoint_url = self._endpoint_url.rstrip("/")
71
+ if not self._endpoint_url:
72
+ raise ProviderConfigError(
73
+ "Docling Serve endpoint URL is required. "
74
+ "Set DOCLING_SERVE_ENDPOINT_URL environment variable or "
75
+ "pass endpoint_url in pipeline config."
76
+ )
77
+
78
+ # Get timeout (default 120 seconds - PDF processing can be slow)
79
+ self._timeout = self.base_config.get("timeout", 120)
80
+
81
+ def _call_endpoint(self, pdf_bytes: bytes, filename: str) -> dict[str, Any]:
82
+ """
83
+ Call the Docling endpoint with PDF bytes.
84
+
85
+ Args:
86
+ pdf_bytes: Raw PDF file bytes
87
+ filename: Name of the PDF file
88
+
89
+ Returns:
90
+ Raw JSON response from endpoint
91
+
92
+ Raises:
93
+ ProviderError: For any API errors
94
+ """
95
+ headers = {"Content-Type": "application/json"}
96
+ if self._api_key:
97
+ headers["Authorization"] = f"Bearer {self._api_key}"
98
+
99
+ # Encode PDF as base64
100
+ pdf_base64 = base64.b64encode(pdf_bytes).decode("utf-8")
101
+
102
+ payload = {
103
+ "sources": [
104
+ {
105
+ "base64_string": pdf_base64,
106
+ "filename": filename,
107
+ "kind": "file",
108
+ }
109
+ ],
110
+ "options": {
111
+ "to_formats": ["json"],
112
+ "pipeline": "standard",
113
+ "include_images": False,
114
+ "image_export_mode": "placeholder",
115
+ },
116
+ }
117
+
118
+ try:
119
+ response = requests.post(
120
+ f"{self._endpoint_url}/v1/convert/source",
121
+ headers=headers,
122
+ json=payload,
123
+ timeout=self._timeout,
124
+ )
125
+ response.raise_for_status()
126
+ result_json = response.json()
127
+ if isinstance(result_json, list):
128
+ if not result_json:
129
+ raise ProviderPermanentError("Endpoint returned an empty list response.")
130
+ first_result = result_json[0]
131
+ if not isinstance(first_result, dict):
132
+ raise ProviderPermanentError("Endpoint returned a list response with a non-dict payload.")
133
+ result = first_result
134
+ elif isinstance(result_json, dict):
135
+ result = result_json
136
+ else:
137
+ raise ProviderPermanentError(
138
+ f"Endpoint returned unsupported response type: {type(result_json).__name__}"
139
+ )
140
+ return result
141
+
142
+ except requests.exceptions.Timeout as e:
143
+ raise ProviderTransientError(f"Request timed out: {e}") from e
144
+ except requests.exceptions.ConnectionError as e:
145
+ raise ProviderTransientError(f"Connection error: {e}") from e
146
+ except requests.exceptions.HTTPError as e:
147
+ status_code = e.response.status_code if e.response else None
148
+ if status_code == 422:
149
+ raise ProviderPermanentError(
150
+ "Docling Serve returned 422. Ensure docling-serve >= 1.0 "
151
+ f"(older versions expect 'file_sources' instead of 'sources'): {e}"
152
+ ) from e
153
+ elif status_code == 429:
154
+ raise ProviderRateLimitError(f"Rate limit exceeded: {e}") from e
155
+ elif status_code and 500 <= status_code < 600:
156
+ raise ProviderTransientError(f"Server error ({status_code}): {e}") from e
157
+ elif status_code and 400 <= status_code < 500:
158
+ raise ProviderPermanentError(f"Client error ({status_code}): {e}") from e
159
+ else:
160
+ raise ProviderPermanentError(f"HTTP error: {e}") from e
161
+ except (ProviderPermanentError, ProviderTransientError, ProviderRateLimitError):
162
+ raise
163
+ except Exception as e:
164
+ raise ProviderPermanentError(f"Unexpected error calling endpoint: {e}") from e
165
+
166
+ def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
167
+ """
168
+ Run inference and return raw results.
169
+
170
+ Args:
171
+ pipeline: Pipeline specification
172
+ request: Inference request
173
+
174
+ Returns:
175
+ Raw inference result
176
+
177
+ Raises:
178
+ ProviderError: For any provider-related failures
179
+ """
180
+ if request.product_type != ProductType.PARSE:
181
+ raise ProviderPermanentError(
182
+ f"DoclingServeProvider only supports PARSE product type, got {request.product_type}"
183
+ )
184
+
185
+ started_at = datetime.now()
186
+
187
+ # Check if file exists
188
+ source_path = Path(request.source_file_path)
189
+ if not source_path.exists():
190
+ raise ProviderPermanentError(f"Source file not found: {source_path}")
191
+
192
+ try:
193
+ # Read PDF bytes
194
+ pdf_bytes = source_path.read_bytes()
195
+
196
+ # Call endpoint
197
+ raw_output = self._call_endpoint(pdf_bytes, source_path.name)
198
+
199
+ completed_at = datetime.now()
200
+ latency_ms = int((completed_at - started_at).total_seconds() * 1000)
201
+
202
+ return RawInferenceResult(
203
+ request=request,
204
+ pipeline=pipeline,
205
+ pipeline_name=pipeline.pipeline_name,
206
+ product_type=request.product_type,
207
+ raw_output=raw_output,
208
+ started_at=started_at,
209
+ completed_at=completed_at,
210
+ latency_in_ms=latency_ms,
211
+ )
212
+
213
+ except (ProviderPermanentError, ProviderTransientError, ProviderRateLimitError):
214
+ raise
215
+ except Exception as e:
216
+ raise ProviderPermanentError(f"Unexpected error during inference: {e}") from e
217
+
218
+ def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
219
+ """
220
+ Normalize raw inference result to produce ParseOutput.
221
+
222
+ Args:
223
+ raw_result: Raw inference result from run_inference()
224
+
225
+ Returns:
226
+ Inference result with ParseOutput
227
+
228
+ Raises:
229
+ ProviderError: For any normalization failures
230
+ """
231
+ if raw_result.product_type != ProductType.PARSE:
232
+ raise ProviderPermanentError(
233
+ f"DoclingServeProvider only supports PARSE product type, got {raw_result.product_type}"
234
+ )
235
+
236
+ # Response format:
237
+ # {
238
+ # "document": {
239
+ # "json_content": {...},
240
+ # }
241
+ # }
242
+ full_markdown = ""
243
+ raw_docling_document = raw_result.raw_output.get("document", {}).get("json_content")
244
+ pages: list[PageIR] = []
245
+
246
+ layout_pages: list[ParseLayoutPageIR] = []
247
+ if raw_docling_document is not None:
248
+ try:
249
+ docling_document = DoclingDocument.model_validate(raw_docling_document)
250
+ except Exception as e:
251
+ raise ProviderPermanentError(f"Failed to validate docling_document payload: {e}") from e
252
+
253
+ doc_serializer = MarkdownDocSerializer(doc=docling_document)
254
+ doc_serializer.table_serializer = HTMLTableSerializer()
255
+
256
+ full_markdown = doc_serializer.serialize(
257
+ page_break_placeholder=_MD_PAGE_BREAK_PLACEHOLDER, image_mode=ImageRefMode.PLACEHOLDER
258
+ ).text
259
+ raw_pages_md = full_markdown.split(_MD_PAGE_BREAK_PLACEHOLDER)
260
+ raw_pages_dicts = []
261
+
262
+ for page_index, markdown in enumerate(raw_pages_md):
263
+ pages.append(PageIR(page_index=page_index, markdown=markdown))
264
+ raw_pages_dicts.append({"page": page_index + 1, "markdown": markdown})
265
+
266
+ layout_pages = _build_docling_layout_pages(
267
+ doc=docling_document,
268
+ raw_pages=raw_pages_dicts,
269
+ )
270
+
271
+ output = ParseOutput(
272
+ task_type="parse",
273
+ example_id=raw_result.request.example_id,
274
+ pipeline_name=raw_result.pipeline_name,
275
+ pages=pages,
276
+ layout_pages=layout_pages,
277
+ markdown=full_markdown,
278
+ )
279
+
280
+ return InferenceResult(
281
+ request=raw_result.request,
282
+ pipeline_name=raw_result.pipeline_name,
283
+ product_type=raw_result.product_type,
284
+ raw_output=raw_result.raw_output,
285
+ output=output,
286
+ started_at=raw_result.started_at,
287
+ completed_at=raw_result.completed_at,
288
+ latency_in_ms=raw_result.latency_in_ms,
289
+ )