parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,452 @@
1
+ """Provider for Landing AI PARSE."""
2
+
3
+ import os
4
+ from datetime import datetime
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from landingai_ade import LandingAIADE
9
+
10
+ from parse_bench.inference.providers.base import (
11
+ Provider,
12
+ ProviderConfigError,
13
+ ProviderPermanentError,
14
+ ProviderTransientError,
15
+ )
16
+ from parse_bench.inference.providers.registry import register_provider
17
+ from parse_bench.schemas.parse_output import (
18
+ LayoutItemIR,
19
+ LayoutSegmentIR,
20
+ PageIR,
21
+ ParseLayoutPageIR,
22
+ ParseOutput,
23
+ )
24
+ from parse_bench.schemas.pipeline import PipelineSpec
25
+ from parse_bench.schemas.pipeline_io import (
26
+ InferenceRequest,
27
+ InferenceResult,
28
+ RawInferenceResult,
29
+ )
30
+ from parse_bench.schemas.product import ProductType
31
+
32
+ # LandingAI chunk type -> Canonical17 label string
33
+ LANDINGAI_LABEL_MAP: dict[str, str] = {
34
+ "text": "Text",
35
+ "table": "Table",
36
+ "figure": "Picture",
37
+ "marginalia": "Page-header", # headers/footers/page numbers consolidated
38
+ "logo": "Picture",
39
+ "card": "Key-Value Region",
40
+ # "attestation" and "scan_code" have no canonical equivalent — skipped
41
+ }
42
+
43
+ # Virtual page dimensions for normalized coordinate conversion.
44
+ # LandingAI bbox is already [0,1], so these cancel out during evaluation.
45
+ _VIRTUAL_PAGE_DIM = 1000.0
46
+
47
+
48
+ @register_provider("landingai")
49
+ class LandingAIParseProvider(Provider):
50
+ """
51
+ Provider for Landing AI PARSE.
52
+
53
+ This provider uses the Landing AI ADE API for parsing tasks.
54
+ """
55
+
56
+ CREDIT_RATE_USD = 0.01 # $0.01 per credit (Explore plan)
57
+
58
+ def __init__(
59
+ self,
60
+ provider_name: str,
61
+ base_config: dict[str, Any] | None = None,
62
+ ):
63
+ """
64
+ Initialize the provider.
65
+
66
+ :param provider_name: Name of the provider
67
+ :param base_config: Optional configuration with:
68
+ - `api_key`: Landing AI API key (defaults to LANDING_AI_API_KEY env var)
69
+ - `model`: Model to use (default: "dpt-2-latest")
70
+ - Any other parse parameters from Landing AI API
71
+ """
72
+ super().__init__(provider_name, base_config)
73
+
74
+ # Get API key
75
+ self._api_key = self.base_config.get("api_key") or os.getenv("LANDING_AI_API_KEY")
76
+ if not self._api_key:
77
+ raise ProviderConfigError(
78
+ "Landing AI API key is required. "
79
+ "Set LANDING_AI_API_KEY environment variable or pass api_key in base_config."
80
+ )
81
+
82
+ # Set VISION_AGENT_API_KEY for the SDK (it expects this env var)
83
+ # Only set if not already set to avoid overriding existing values
84
+ if not os.getenv("VISION_AGENT_API_KEY"):
85
+ os.environ["VISION_AGENT_API_KEY"] = self._api_key
86
+
87
+ # Get configuration with defaults
88
+ self._model = self.base_config.get("model", "dpt-2-latest")
89
+
90
+ # Initialize client
91
+ self._client = LandingAIADE()
92
+
93
+ def _parse_document(self, document_path: Path) -> dict[str, Any]:
94
+ """
95
+ Parse a document using Landing AI API.
96
+
97
+ :param document_path: Path to the document file
98
+ :return: Raw API response as dictionary
99
+ :raises ProviderError: For any API errors
100
+ """
101
+ try:
102
+ # Parse the document
103
+ response = self._client.parse(
104
+ document=document_path,
105
+ model=self._model,
106
+ **{k: v for k, v in self.base_config.items() if k not in ["api_key", "model"]},
107
+ )
108
+
109
+ # Convert response to dictionary format
110
+ # The response has markdown, chunks, and grounding attributes
111
+ result: dict[str, Any] = {
112
+ "markdown": response.markdown if hasattr(response, "markdown") else "",
113
+ "chunks": [],
114
+ "splits": [],
115
+ "grounding": {},
116
+ }
117
+
118
+ # Extract chunks if available
119
+ if hasattr(response, "chunks"):
120
+ chunks = response.chunks
121
+ if chunks is not None:
122
+ # Convert chunks to serializable format
123
+ for chunk in chunks:
124
+ chunk_data: dict[str, Any] = {}
125
+ if hasattr(chunk, "id"):
126
+ chunk_data["id"] = chunk.id
127
+ if hasattr(chunk, "type"):
128
+ chunk_data["type"] = chunk.type
129
+ if hasattr(chunk, "markdown"):
130
+ chunk_data["markdown"] = chunk.markdown
131
+ if hasattr(chunk, "grounding") and chunk.grounding is not None:
132
+ # ChunkGrounding is a Pydantic model - convert to dict
133
+ chunk_data["grounding"] = chunk.grounding.model_dump()
134
+ result["chunks"].append(chunk_data)
135
+
136
+ # Extract splits if available (populated when split="page" is used)
137
+ if hasattr(response, "splits") and response.splits is not None:
138
+ for split in response.splits:
139
+ split_data: dict[str, Any] = {}
140
+ if hasattr(split, "markdown"):
141
+ split_data["markdown"] = split.markdown
142
+ if hasattr(split, "pages"):
143
+ split_data["pages"] = split.pages
144
+ if hasattr(split, "chunks"):
145
+ split_data["chunks"] = split.chunks
146
+ if hasattr(split, "class_"):
147
+ split_data["class"] = split.class_
148
+ if hasattr(split, "identifier"):
149
+ split_data["identifier"] = split.identifier
150
+ result["splits"].append(split_data)
151
+
152
+ # Extract grounding if available
153
+ # response.grounding is Dict[str, Grounding] where Grounding is a Pydantic model
154
+ if hasattr(response, "grounding") and response.grounding is not None:
155
+ result["grounding"] = {k: v.model_dump() for k, v in response.grounding.items()}
156
+
157
+ # Extract cost from metadata
158
+ if hasattr(response, "metadata") and response.metadata is not None:
159
+ meta = response.metadata
160
+ credits = getattr(meta, "credit_usage", None)
161
+ num_pages = getattr(meta, "page_count", None)
162
+ if credits is not None and credits > 0:
163
+ cost_usd = credits * self.CREDIT_RATE_USD
164
+ result["credits_used"] = credits
165
+ result["cost_usd"] = cost_usd
166
+ if num_pages and num_pages > 0:
167
+ result["num_pages"] = num_pages
168
+ result["cost_per_page_usd"] = cost_usd / num_pages
169
+
170
+ return result
171
+
172
+ except Exception as e:
173
+ # Check if it's a transient error (network, timeout, etc.)
174
+ error_str = str(e).lower()
175
+ transient_keywords = ["timeout", "network", "connection", "503", "502", "504"]
176
+ if any(keyword in error_str for keyword in transient_keywords):
177
+ raise ProviderTransientError(f"Transient error during parsing: {e}") from e
178
+ else:
179
+ raise ProviderPermanentError(f"Error during parsing: {e}") from e
180
+
181
+ def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
182
+ """
183
+ Run inference and return raw results.
184
+
185
+ :param pipeline: Pipeline specification
186
+ :param request: Inference request
187
+ :return: Raw inference result
188
+ :raises ProviderError: For any provider-related failures
189
+ """
190
+ if request.product_type != ProductType.PARSE:
191
+ raise ProviderPermanentError(
192
+ f"LandingAIParseProvider only supports PARSE product type, got {request.product_type}"
193
+ )
194
+
195
+ started_at = datetime.now()
196
+
197
+ # Check if file exists
198
+ file_path = Path(request.source_file_path)
199
+ if not file_path.exists():
200
+ raise ProviderPermanentError(f"File not found: {file_path}")
201
+
202
+ try:
203
+ # Run parsing
204
+ raw_output = self._parse_document(file_path)
205
+
206
+ completed_at = datetime.now()
207
+ latency_ms = int((completed_at - started_at).total_seconds() * 1000)
208
+
209
+ return RawInferenceResult(
210
+ request=request,
211
+ pipeline=pipeline,
212
+ pipeline_name=pipeline.pipeline_name,
213
+ product_type=request.product_type,
214
+ raw_output=raw_output,
215
+ started_at=started_at,
216
+ completed_at=completed_at,
217
+ latency_in_ms=latency_ms,
218
+ )
219
+
220
+ except ProviderPermanentError:
221
+ # Re-raise provider errors as-is
222
+ raise
223
+ except ProviderTransientError:
224
+ # Re-raise provider errors as-is
225
+ raise
226
+ except Exception as e:
227
+ # Wrap unexpected errors
228
+ raise ProviderPermanentError(f"Unexpected error during inference: {e}") from e
229
+
230
+ def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
231
+ """
232
+ Normalize raw inference result to produce ParseOutput.
233
+
234
+ :param raw_result: Raw inference result from run_inference()
235
+ :return: Inference result with both raw and normalized outputs
236
+ :raises ProviderError: For any normalization failures
237
+ """
238
+ if raw_result.product_type != ProductType.PARSE:
239
+ raise ProviderPermanentError(
240
+ f"LandingAIParseProvider only supports PARSE product type, got {raw_result.product_type}"
241
+ )
242
+
243
+ # Extract markdown from raw output and promote table headers.
244
+ # Landing AI emits all cells as <td>; downstream eval relies on <th>.
245
+ markdown = _promote_first_row_to_header(raw_result.raw_output.get("markdown", ""))
246
+
247
+ pages: list[PageIR] = []
248
+
249
+ # Strategy 1: Use splits data if available (from split="page")
250
+ splits = raw_result.raw_output.get("splits", [])
251
+ if splits:
252
+ for split in splits:
253
+ if isinstance(split, dict) and "markdown" in split and "pages" in split:
254
+ split_pages = split["pages"]
255
+ split_md = _promote_first_row_to_header(split["markdown"])
256
+ # Each split may cover one or more pages; use the first page number
257
+ page_num = split_pages[0] if split_pages else 0
258
+ pages.append(PageIR(page_index=page_num, markdown=split_md))
259
+
260
+ # Strategy 2: Fall back to chunk grounding for page splitting
261
+ if not pages:
262
+ chunks = raw_result.raw_output.get("chunks", [])
263
+ grounding = raw_result.raw_output.get("grounding", {})
264
+
265
+ page_content: dict[int, list[str]] = {}
266
+
267
+ if isinstance(grounding, dict):
268
+ for gid, gdata in grounding.items():
269
+ if isinstance(gdata, dict) and "page" in gdata:
270
+ page_num = gdata["page"]
271
+ if page_num not in page_content:
272
+ page_content[page_num] = []
273
+ for chunk in chunks:
274
+ if isinstance(chunk, dict) and chunk.get("id") == gid:
275
+ if "markdown" in chunk:
276
+ page_content[page_num].append(_promote_first_row_to_header(chunk["markdown"]))
277
+
278
+ if page_content:
279
+ for page_num in sorted(page_content.keys()):
280
+ page_text = "\n".join(page_content[page_num])
281
+ pages.append(PageIR(page_index=page_num, markdown=page_text))
282
+
283
+ # Strategy 3: Fallback — single page with all markdown
284
+ if not pages:
285
+ pages.append(PageIR(page_index=0, markdown=markdown))
286
+
287
+ # Build layout_pages from chunk grounding for layout cross-evaluation
288
+ chunks = raw_result.raw_output.get("chunks", [])
289
+ layout_pages = _build_layout_pages(chunks)
290
+
291
+ output = ParseOutput(
292
+ task_type="parse",
293
+ example_id=raw_result.request.example_id,
294
+ pipeline_name=raw_result.pipeline_name,
295
+ pages=pages,
296
+ layout_pages=layout_pages,
297
+ markdown=markdown,
298
+ job_id=None, # Landing AI parse doesn't return job_id
299
+ )
300
+
301
+ return InferenceResult(
302
+ request=raw_result.request,
303
+ pipeline_name=raw_result.pipeline_name,
304
+ product_type=raw_result.product_type,
305
+ raw_output=raw_result.raw_output,
306
+ output=output,
307
+ started_at=raw_result.started_at,
308
+ completed_at=raw_result.completed_at,
309
+ latency_in_ms=raw_result.latency_in_ms,
310
+ )
311
+
312
+
313
+ def _build_layout_pages(chunks: list[dict[str, Any]]) -> list[ParseLayoutPageIR]:
314
+ """Build layout_pages from LandingAI chunk grounding for layout cross-evaluation.
315
+
316
+ Groups chunks by page number and converts each chunk's normalized [0,1]
317
+ bounding box into a LayoutSegmentIR with canonical label mapping.
318
+ LandingAI grounding pages are 0-indexed; we convert to 1-indexed.
319
+ """
320
+ from collections import defaultdict
321
+
322
+ pages_chunks: dict[int, list[dict[str, Any]]] = defaultdict(list)
323
+ for chunk in chunks:
324
+ grounding = chunk.get("grounding")
325
+ if not isinstance(grounding, dict):
326
+ continue
327
+ # LandingAI pages are 0-indexed
328
+ page_num = grounding.get("page", 0)
329
+ pages_chunks[page_num].append(chunk)
330
+
331
+ layout_pages: list[ParseLayoutPageIR] = []
332
+ for page_num in sorted(pages_chunks.keys()):
333
+ page_chunks = pages_chunks[page_num]
334
+ items: list[LayoutItemIR] = []
335
+
336
+ for chunk in page_chunks:
337
+ chunk_type = chunk.get("type", "")
338
+ canonical_label = LANDINGAI_LABEL_MAP.get(chunk_type)
339
+ if canonical_label is None:
340
+ continue # Skip unmapped types (e.g., attestation, scan_code)
341
+
342
+ grounding = chunk.get("grounding", {})
343
+ box = grounding.get("box", {})
344
+ left = float(box.get("left", 0.0))
345
+ top = float(box.get("top", 0.0))
346
+ right = float(box.get("right", 0.0))
347
+ bottom = float(box.get("bottom", 0.0))
348
+ width = right - left
349
+ height = bottom - top
350
+
351
+ # Parse confidence (DPT-2 provides it)
352
+ conf_raw = grounding.get("confidence")
353
+ try:
354
+ confidence = float(conf_raw) if conf_raw is not None else 1.0
355
+ except (TypeError, ValueError):
356
+ confidence = 1.0
357
+
358
+ seg = LayoutSegmentIR(
359
+ x=left,
360
+ y=top,
361
+ w=width,
362
+ h=height,
363
+ confidence=confidence,
364
+ label=canonical_label,
365
+ )
366
+
367
+ content = chunk.get("markdown", "")
368
+ norm_label = canonical_label.strip().lower()
369
+ if norm_label == "table":
370
+ item_type = "table"
371
+ elif norm_label == "picture":
372
+ item_type = "image"
373
+ else:
374
+ item_type = "text"
375
+
376
+ items.append(
377
+ LayoutItemIR(
378
+ type=item_type,
379
+ value=content,
380
+ bbox=seg,
381
+ layout_segments=[seg],
382
+ )
383
+ )
384
+
385
+ # Convert 0-indexed page to 1-indexed for ParseLayoutPageIR
386
+ layout_pages.append(
387
+ ParseLayoutPageIR(
388
+ page_number=page_num + 1,
389
+ width=_VIRTUAL_PAGE_DIM,
390
+ height=_VIRTUAL_PAGE_DIM,
391
+ items=items,
392
+ )
393
+ )
394
+
395
+ return layout_pages
396
+
397
+
398
+ def _promote_first_row_to_header(html: str) -> str:
399
+ """Rewrite HTML tables so the first row uses ``<th>`` inside ``<thead>``.
400
+
401
+ Landing AI emits all table cells as ``<td>`` with no
402
+ ``<th>``/``<thead>``/``<tbody>``. This promotes the first ``<tr>`` of each
403
+ ``<table>`` to be a header row so that downstream evaluation code (which
404
+ keys on ``<th>``) can identify column headers.
405
+
406
+ Only tables that contain zero ``<th>`` elements are modified — tables that
407
+ already have headers are left untouched.
408
+ """
409
+ from bs4 import BeautifulSoup
410
+
411
+ if "<table" not in html:
412
+ return html
413
+
414
+ soup = BeautifulSoup(html, "lxml")
415
+ modified = False
416
+
417
+ for table in soup.find_all("table"):
418
+ # Skip tables that already have <th> elements
419
+ if table.find("th"):
420
+ continue
421
+
422
+ first_tr = table.find("tr")
423
+ if first_tr is None:
424
+ continue
425
+
426
+ # Promote <td> -> <th> in the first row
427
+ for td in first_tr.find_all("td"):
428
+ td.name = "th"
429
+
430
+ # Wrap first row in <thead>, remaining rows in <tbody>
431
+ thead = soup.new_tag("thead")
432
+ first_tr.extract()
433
+ thead.append(first_tr)
434
+
435
+ tbody = soup.new_tag("tbody")
436
+ for tr in table.find_all("tr"):
437
+ tr.extract()
438
+ tbody.append(tr)
439
+
440
+ table.clear()
441
+ table.append(thead)
442
+ if tbody.find("tr"):
443
+ table.append(tbody)
444
+
445
+ modified = True
446
+
447
+ if not modified:
448
+ return html
449
+
450
+ # Return just the body content to avoid <html><body> wrapper
451
+ body = soup.find("body")
452
+ return body.decode_contents() if body else str(soup)