parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,710 @@
1
+ """Provider for Extend AI PARSE using the official Python SDK.
2
+
3
+ Based on Extend AI documentation: https://docs.extend.ai/product/parsing/parse
4
+ SDK: pip install extend-ai
5
+ """
6
+
7
+ import os
8
+ import threading
9
+ from collections import defaultdict
10
+ from datetime import datetime
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ from extend_ai import Extend, FileFromIdParams, ParseConfigParams
15
+ from extend_ai.core.api_error import ApiError
16
+ from extend_ai.types import ParseConfigChunkingStrategy
17
+ from pypdf import PdfReader
18
+
19
+ from parse_bench.inference.providers.base import (
20
+ Provider,
21
+ ProviderConfigError,
22
+ ProviderPermanentError,
23
+ ProviderRateLimitError,
24
+ ProviderTransientError,
25
+ )
26
+ from parse_bench.inference.providers.registry import register_provider
27
+ from parse_bench.schemas.parse_output import (
28
+ LayoutItemIR,
29
+ LayoutSegmentIR,
30
+ PageIR,
31
+ ParseLayoutPageIR,
32
+ ParseOutput,
33
+ )
34
+ from parse_bench.schemas.pipeline import PipelineSpec
35
+ from parse_bench.schemas.pipeline_io import (
36
+ InferenceRequest,
37
+ InferenceResult,
38
+ RawInferenceResult,
39
+ )
40
+ from parse_bench.schemas.product import ProductType
41
+
42
+ # Extend block type -> Canonical17 label string
43
+ EXTEND_LABEL_MAP: dict[str, str] = {
44
+ "heading": "Section-header",
45
+ "section_heading": "Section-header",
46
+ "text": "Text",
47
+ "table": "Table",
48
+ "figure": "Picture",
49
+ "header": "Page-header",
50
+ "footer": "Page-footer",
51
+ "key_value": "Key-Value Region",
52
+ "page_number": "Page-footer",
53
+ "formula": "Formula",
54
+ }
55
+
56
+ # Virtual page dimensions for normalized coordinate conversion.
57
+ # Extend bboxes are converted to [0,1] using PDF page dims, so these cancel out.
58
+ _VIRTUAL_PAGE_DIM = 1000.0
59
+
60
+
61
+ @register_provider("extend_parse")
62
+ class ExtendParseProvider(Provider):
63
+ """
64
+ Provider for Extend AI document parsing using the official SDK.
65
+
66
+ This provider uses the extend-ai Python SDK for parsing tasks.
67
+ SDK Documentation: https://docs.extend.ai/developers/sd-ks
68
+
69
+ Workflow:
70
+ 1. Upload file via client.file.upload()
71
+ 2. Call client.parse() with configuration options
72
+ 3. Return markdown content from parsed result
73
+ """
74
+
75
+ # Extend meters parsing in credits; one credit is $0.0125.
76
+ CREDIT_RATE_USD = 0.0125
77
+
78
+ # Credits billed per page, keyed by parse engine. The default engine (no
79
+ # explicit `engine` in the pipeline config) bills at the parse_performance rate.
80
+ ENGINE_CREDITS_PER_PAGE: dict[str, float] = {
81
+ "parse_performance": 2.0,
82
+ "parse_light": 0.5,
83
+ }
84
+ DEFAULT_CREDITS_PER_PAGE = 2.0
85
+
86
+ def __init__(
87
+ self,
88
+ provider_name: str,
89
+ base_config: dict[str, Any] | None = None,
90
+ ):
91
+ """
92
+ Initialize the provider.
93
+
94
+ :param provider_name: Name of the provider
95
+ :param base_config: Optional configuration with:
96
+ - `api_key`: Extend AI API key (defaults to EXTEND_API_KEY env var)
97
+ - `base_url`: Optional base URL for different deployments
98
+ (default: https://api.extend.ai, alternatives: https://api.us2.extend.app,
99
+ https://api.eu1.extend.ai)
100
+ - `timeout`: Request timeout in seconds (default: 300)
101
+ - `chunking_strategy`: "page", "section", or "document" (default: "page")
102
+ - `target`: Output format - "markdown" or "spatial" (default: "markdown")
103
+ """
104
+ super().__init__(provider_name, base_config)
105
+
106
+ # Get API key
107
+ api_key = self.base_config.get("api_key") or os.getenv("EXTEND_API_KEY")
108
+ if not api_key:
109
+ raise ProviderConfigError(
110
+ "Extend AI API key is required. Set EXTEND_API_KEY environment variable or pass api_key in base_config."
111
+ )
112
+
113
+ # Configuration
114
+ timeout = self.base_config.get("timeout", 300)
115
+
116
+ # Initialize the Extend client
117
+ client_kwargs: dict[str, Any] = {
118
+ "token": api_key,
119
+ "timeout": float(timeout),
120
+ }
121
+
122
+ # Optional base URL for different deployments (US2, EU1, etc.)
123
+ base_url = self.base_config.get("base_url")
124
+ if base_url:
125
+ client_kwargs["base_url"] = base_url
126
+
127
+ self._client = Extend(**client_kwargs)
128
+
129
+ # Thread lock for file uploads
130
+ self._upload_lock = threading.Lock()
131
+
132
+ def _handle_api_error(self, e: ApiError, context: str) -> None:
133
+ """Convert SDK ApiError to appropriate ProviderError."""
134
+ status_code = getattr(e, "status_code", None)
135
+ error_body = getattr(e, "body", str(e))
136
+
137
+ if status_code == 429:
138
+ raise ProviderRateLimitError(f"Rate limit exceeded during {context}: {error_body}")
139
+ elif status_code in (502, 503, 504):
140
+ raise ProviderTransientError(f"Transient error during {context}: {status_code} - {error_body}")
141
+ elif status_code and status_code >= 400:
142
+ raise ProviderPermanentError(f"Error during {context}: {status_code} - {error_body}")
143
+ else:
144
+ raise ProviderPermanentError(f"API error during {context}: {error_body}")
145
+
146
+ def _is_pdf_file(self, file_path: str) -> bool:
147
+ """
148
+ Check if a file is a PDF by reading its header.
149
+
150
+ :param file_path: Path to the file
151
+ :return: True if the file is a PDF, False otherwise
152
+ """
153
+ try:
154
+ with open(file_path, "rb") as f:
155
+ header = f.read(4)
156
+ return header == b"%PDF"
157
+ except Exception:
158
+ return False
159
+
160
+ def _get_page_count(self, file_path: str) -> int:
161
+ """
162
+ Get the page count for a file. For PDFs, reads the actual page count.
163
+ For images, returns 1.
164
+
165
+ :param file_path: Path to the file
166
+ :return: Number of pages (1 for images, actual count for PDFs)
167
+ """
168
+ if self._is_pdf_file(file_path):
169
+ try:
170
+ reader = PdfReader(file_path)
171
+ return len(reader.pages)
172
+ except Exception:
173
+ return 1
174
+ else:
175
+ return 1
176
+
177
+ def _credits_per_page(self, pipeline_config: dict[str, Any]) -> float | None:
178
+ """
179
+ Resolve how many credits per page the configured engine bills.
180
+
181
+ :param pipeline_config: Pipeline configuration options
182
+ :return: Credits per page, or None if the engine's rate is unknown
183
+ :raises ProviderConfigError: If an explicit override is negative
184
+ """
185
+ override = pipeline_config.get("credits_per_page", self.base_config.get("credits_per_page"))
186
+ if override is not None:
187
+ credits = float(override)
188
+ if credits < 0:
189
+ raise ProviderConfigError("credits_per_page must be non-negative")
190
+ return credits
191
+
192
+ engine = pipeline_config.get("engine")
193
+ if engine is None:
194
+ return self.DEFAULT_CREDITS_PER_PAGE
195
+ # Unknown engine: omit cost rather than report a rate we cannot vouch for.
196
+ return self.ENGINE_CREDITS_PER_PAGE.get(engine)
197
+
198
+ def _upload_file(self, file_path: str) -> str:
199
+ """
200
+ Upload a file to Extend AI.
201
+
202
+ :param file_path: Path to the file to upload
203
+ :return: File ID from Extend AI
204
+ :raises ProviderError: For any upload errors
205
+ """
206
+ try:
207
+ with open(file_path, "rb") as f:
208
+ upload_response = self._client.files.upload(file=f)
209
+
210
+ # Extract file ID from response
211
+ if hasattr(upload_response, "id"):
212
+ return str(upload_response.id)
213
+ elif hasattr(upload_response, "file") and hasattr(upload_response.file, "id"):
214
+ return str(upload_response.file.id)
215
+ elif isinstance(upload_response, dict):
216
+ file_data = upload_response.get("file", upload_response)
217
+ file_id = file_data.get("id") or file_data.get("fileId")
218
+ if file_id:
219
+ return str(file_id)
220
+
221
+ raise ProviderPermanentError(f"No file ID in upload response: {upload_response}")
222
+
223
+ except ApiError as e:
224
+ self._handle_api_error(e, "file upload")
225
+ raise
226
+ except Exception as e:
227
+ error_str = str(e).lower()
228
+ if any(kw in error_str for kw in ["timeout", "timed out", "connection", "network", "readtimeout"]):
229
+ raise ProviderTransientError(f"Transient error during file upload: {e}") from e
230
+ raise ProviderPermanentError(f"Unexpected error during file upload: {e}") from e
231
+
232
+ def _build_parse_config(self, pipeline_config: dict[str, Any]) -> dict[str, Any]:
233
+ """
234
+ Build the parse config from pipeline configuration.
235
+
236
+ :param pipeline_config: Pipeline configuration options
237
+ :return: Parse configuration dict
238
+ """
239
+ config: dict[str, Any] = {}
240
+
241
+ # Target format: "markdown" or "spatial"
242
+ if "target" in pipeline_config:
243
+ config["target"] = pipeline_config["target"]
244
+
245
+ # Chunking strategy: "page", "section", or "document"
246
+ if "chunking_strategy" in pipeline_config:
247
+ config["chunking_strategy"] = ParseConfigChunkingStrategy(type=pipeline_config["chunking_strategy"])
248
+
249
+ # Block options for fine-grained control
250
+ if "block_options" in pipeline_config:
251
+ config["block_options"] = pipeline_config["block_options"]
252
+
253
+ # Advanced options (OCR enhancements, page filtering)
254
+ if "advanced_options" in pipeline_config:
255
+ config["advanced_options"] = pipeline_config["advanced_options"]
256
+
257
+ # Engine selection (e.g. "parse_performance")
258
+ if "engine" in pipeline_config:
259
+ config["engine"] = pipeline_config["engine"]
260
+
261
+ # Engine version (e.g. "2.0.0-beta")
262
+ if "engineVersion" in pipeline_config:
263
+ config["engineVersion"] = pipeline_config["engineVersion"]
264
+
265
+ return config
266
+
267
+ def _parse_document(
268
+ self,
269
+ file_path: str,
270
+ pipeline_config: dict[str, Any],
271
+ ) -> dict[str, Any]:
272
+ """
273
+ Parse a document using Extend AI.
274
+
275
+ :param file_path: Path to the document file
276
+ :param pipeline_config: Pipeline configuration options
277
+ :return: Raw API response with parsed content
278
+ :raises ProviderError: For any parsing errors
279
+ """
280
+ # Get page count and page dimensions (for bbox normalization)
281
+ num_pages = self._get_page_count(file_path)
282
+ page_dims = _get_pdf_page_dims(file_path)
283
+
284
+ # Step 1: Upload file
285
+ file_id = self._upload_file(file_path)
286
+
287
+ # Step 2: Build parse config
288
+ parse_config = self._build_parse_config(pipeline_config)
289
+
290
+ # Step 3: Call parse API
291
+ try:
292
+ # The Extend SDK parse method
293
+ parse_response = self._client.parse(
294
+ file=FileFromIdParams(id=file_id),
295
+ config=ParseConfigParams(**parse_config) if parse_config else None, # type: ignore[typeddict-item]
296
+ )
297
+
298
+ # Convert response to dict
299
+ if hasattr(parse_response, "model_dump"):
300
+ result = parse_response.model_dump()
301
+ elif hasattr(parse_response, "dict"):
302
+ result = parse_response.dict()
303
+ elif isinstance(parse_response, dict):
304
+ result = parse_response
305
+ else:
306
+ # Try to extract attributes manually
307
+ result = {}
308
+ for attr in [
309
+ "id",
310
+ "status",
311
+ "chunks",
312
+ "content",
313
+ "markdown",
314
+ "pages",
315
+ "error",
316
+ "fileId",
317
+ ]:
318
+ if hasattr(parse_response, attr):
319
+ value = getattr(parse_response, attr)
320
+ if not callable(value):
321
+ result[attr] = value
322
+
323
+ # Add metadata
324
+ result["_extend_metadata"] = {
325
+ "file_id": file_id,
326
+ "num_pages": num_pages,
327
+ "page_dims": page_dims,
328
+ "config": parse_config,
329
+ }
330
+
331
+ # Operational stats: page count feeds per-page latency, credits feed cost.
332
+ result["num_pages"] = num_pages
333
+ credits_per_page = self._credits_per_page(pipeline_config)
334
+ if credits_per_page is not None:
335
+ cost_per_page_usd = credits_per_page * self.CREDIT_RATE_USD
336
+ result["credits_used"] = credits_per_page * num_pages
337
+ result["cost_per_page_usd"] = cost_per_page_usd
338
+ result["cost_usd"] = cost_per_page_usd * num_pages
339
+
340
+ return result
341
+
342
+ except ApiError as e:
343
+ self._handle_api_error(e, "document parsing")
344
+ raise
345
+ except Exception as e:
346
+ error_str = str(e).lower()
347
+ if any(kw in error_str for kw in ["timeout", "timed out", "connection", "network", "readtimeout"]):
348
+ raise ProviderTransientError(f"Transient error during parsing: {e}") from e
349
+ raise ProviderPermanentError(f"Unexpected error during parsing: {e}") from e
350
+
351
+ def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
352
+ """
353
+ Run inference and return raw results.
354
+
355
+ :param pipeline: Pipeline specification
356
+ :param request: Inference request
357
+ :return: Raw inference result
358
+ :raises ProviderError: For any provider-related failures
359
+ """
360
+ if request.product_type != ProductType.PARSE:
361
+ raise ProviderPermanentError(
362
+ f"ExtendParseProvider only supports PARSE product type, got {request.product_type}"
363
+ )
364
+
365
+ started_at = datetime.now()
366
+
367
+ # Check if file exists
368
+ file_path = Path(request.source_file_path)
369
+ if not file_path.exists():
370
+ raise ProviderPermanentError(f"File not found: {file_path}")
371
+
372
+ try:
373
+ # Run parsing with pipeline config options
374
+ raw_output = self._parse_document(
375
+ file_path=str(file_path),
376
+ pipeline_config=pipeline.config,
377
+ )
378
+
379
+ completed_at = datetime.now()
380
+ latency_ms = int((completed_at - started_at).total_seconds() * 1000)
381
+
382
+ return RawInferenceResult(
383
+ request=request,
384
+ pipeline=pipeline,
385
+ pipeline_name=pipeline.pipeline_name,
386
+ product_type=request.product_type,
387
+ raw_output=raw_output,
388
+ started_at=started_at,
389
+ completed_at=completed_at,
390
+ latency_in_ms=latency_ms,
391
+ )
392
+
393
+ except (ProviderPermanentError, ProviderTransientError, ProviderRateLimitError):
394
+ raise
395
+ except Exception as e:
396
+ raise ProviderPermanentError(f"Unexpected error during inference: {e}") from e
397
+
398
+ def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
399
+ """
400
+ Normalize raw inference result to produce ParseOutput.
401
+
402
+ :param raw_result: Raw inference result from run_inference()
403
+ :return: Inference result with both raw and normalized outputs
404
+ :raises ProviderError: For any normalization failures
405
+ """
406
+ if raw_result.product_type != ProductType.PARSE:
407
+ raise ProviderPermanentError(
408
+ f"ExtendParseProvider only supports PARSE product type, got {raw_result.product_type}"
409
+ )
410
+
411
+ raw_output = raw_result.raw_output
412
+
413
+ # SDK 1.x wraps content under raw_output["output"]; legacy responses had it at the top level.
414
+ # Source the chunk-bearing payload from whichever shape applies.
415
+ payload = raw_output.get("output") if isinstance(raw_output.get("output"), dict) else raw_output
416
+
417
+ # Extract markdown content from response
418
+ # Extend API can return content in different formats depending on config
419
+ markdown = ""
420
+
421
+ # Try different response formats
422
+ # 1. Direct markdown field
423
+ if "markdown" in payload:
424
+ markdown = payload["markdown"]
425
+ # 2. Content field
426
+ elif "content" in payload:
427
+ content = payload["content"]
428
+ if isinstance(content, str):
429
+ markdown = content
430
+ elif isinstance(content, dict):
431
+ markdown = content.get("markdown", "") or content.get("text", "")
432
+ # 3. Chunks array (similar to Reducto)
433
+ elif "chunks" in payload:
434
+ chunks = payload["chunks"]
435
+ if chunks and isinstance(chunks, list):
436
+ # Concatenate all chunk contents
437
+ chunk_contents = []
438
+ for chunk in chunks:
439
+ if isinstance(chunk, dict):
440
+ chunk_content = chunk.get("content", "") or chunk.get("markdown", "")
441
+ if chunk_content:
442
+ chunk_contents.append(chunk_content)
443
+ elif isinstance(chunk, str):
444
+ chunk_contents.append(chunk)
445
+ markdown = "\n\n".join(chunk_contents)
446
+ # 4. Pages array
447
+ elif "pages" in payload:
448
+ pages = payload["pages"]
449
+ if pages and isinstance(pages, list):
450
+ page_contents = []
451
+ for page in pages:
452
+ if isinstance(page, dict):
453
+ page_content = page.get("markdown", "") or page.get("content", "")
454
+ if page_content:
455
+ page_contents.append(page_content)
456
+ elif isinstance(page, str):
457
+ page_contents.append(page)
458
+ markdown = "\n\n".join(page_contents)
459
+
460
+ # Get job ID if available
461
+ job_id = raw_output.get("id") or raw_output.get("job_id")
462
+
463
+ # Build layout_pages from chunk blocks for layout cross-evaluation
464
+ metadata = raw_output.get("_extend_metadata", {})
465
+ page_dims = metadata.get("page_dims", {})
466
+ chunks = payload.get("chunks", [])
467
+ layout_pages = _build_layout_pages(chunks, page_dims)
468
+
469
+ # Build per-page markdown so that page-scoped rules
470
+ # (``_scope_to_page`` in rules_form.py) match against the correct page
471
+ # instead of falling back to full-document content. Works regardless of
472
+ # ``chunking_strategy`` (page/section/document) because page numbers
473
+ # come from each block's ``metadata.page.number``.
474
+ pages = _build_pages(chunks)
475
+
476
+ output = ParseOutput(
477
+ task_type="parse",
478
+ example_id=raw_result.request.example_id,
479
+ pipeline_name=raw_result.pipeline_name,
480
+ pages=pages,
481
+ layout_pages=layout_pages,
482
+ markdown=markdown,
483
+ job_id=str(job_id) if job_id else None,
484
+ )
485
+
486
+ return InferenceResult(
487
+ request=raw_result.request,
488
+ pipeline_name=raw_result.pipeline_name,
489
+ product_type=raw_result.product_type,
490
+ raw_output=raw_result.raw_output,
491
+ output=output,
492
+ started_at=raw_result.started_at,
493
+ completed_at=raw_result.completed_at,
494
+ latency_in_ms=raw_result.latency_in_ms,
495
+ )
496
+
497
+
498
+ def _get_pdf_page_dims(file_path: str) -> dict[int, tuple[float, float]]:
499
+ """Read per-page dimensions (width, height) in PDF points from a PDF file.
500
+
501
+ Returns a dict mapping 1-indexed page number to (width, height).
502
+ Returns empty dict for non-PDF files or on error.
503
+ """
504
+ try:
505
+ with open(file_path, "rb") as f:
506
+ if f.read(4) != b"%PDF":
507
+ return {}
508
+ reader = PdfReader(file_path)
509
+ dims: dict[int, tuple[float, float]] = {}
510
+ for i, page in enumerate(reader.pages):
511
+ box = page.mediabox
512
+ dims[i + 1] = (float(box.width), float(box.height))
513
+ return dims
514
+ except Exception:
515
+ return {}
516
+
517
+
518
+ def _build_pages(chunks: list[dict[str, Any]]) -> list[PageIR]:
519
+ """Build per-page markdown from Extend chunks/blocks.
520
+
521
+ Extend exposes the page number on each block via ``metadata.page.number``
522
+ (1-indexed). We aggregate block-level ``content`` per page so that
523
+ page-scoped rules can match against the right page's text instead of
524
+ falling back to full-document content.
525
+
526
+ Block content is concatenated with ``\\n\\n``. This works for every
527
+ ``chunking_strategy`` (``page``, ``section``, ``document``) because the
528
+ page assignment is per-block, not per-chunk.
529
+
530
+ Blocks with non-positive or unparseable page numbers are dropped from the
531
+ per-page output (``PageIR.page_index`` requires ``ge=0``). Their content
532
+ still reaches the document-level ``markdown`` via the chunk-level join in
533
+ :py:meth:`ExtendParseProvider.normalize`.
534
+ """
535
+ pages_blocks: dict[int, list[str]] = defaultdict(list)
536
+ for chunk in chunks:
537
+ if not isinstance(chunk, dict):
538
+ continue
539
+ for block in chunk.get("blocks", []):
540
+ if not isinstance(block, dict):
541
+ continue
542
+ block_meta = block.get("metadata", {}) or {}
543
+ block_page_meta = block_meta.get("page", {}) or {}
544
+ # Use ``is not None`` rather than an ``or`` chain so that an
545
+ # explicit ``0`` from the API is treated as a real (invalid) page
546
+ # value and dropped below, not silently rewritten to ``1``.
547
+ raw_page = block_page_meta.get("number")
548
+ if raw_page is None:
549
+ raw_page = block.get("page")
550
+ if raw_page is None:
551
+ raw_page = block.get("pageNumber")
552
+ if raw_page is None:
553
+ raw_page = 1
554
+ try:
555
+ page_num = int(raw_page)
556
+ except (TypeError, ValueError):
557
+ page_num = 1
558
+ if page_num < 1:
559
+ # Defensive: 0-indexed or negative page values would produce
560
+ # PageIR.page_index < 0 and fail Pydantic validation.
561
+ continue
562
+ content = block.get("content", "") or block.get("text", "") or ""
563
+ if content:
564
+ pages_blocks[page_num].append(content)
565
+
566
+ pages: list[PageIR] = []
567
+ for page_num in sorted(pages_blocks.keys()):
568
+ page_md = "\n\n".join(pages_blocks[page_num])
569
+ # Extend pages are 1-indexed; PageIR.page_index is 0-indexed.
570
+ pages.append(PageIR(page_index=page_num - 1, markdown=page_md))
571
+ return pages
572
+
573
+
574
+ def _build_layout_pages(
575
+ chunks: list[dict[str, Any]],
576
+ page_dims: dict[int, tuple[float, float]] | dict[str, Any],
577
+ ) -> list[ParseLayoutPageIR]:
578
+ """Build layout_pages from Extend chunk blocks for layout cross-evaluation.
579
+
580
+ Iterates through chunks and their blocks, normalizes bboxes to [0,1]
581
+ using page dimensions, and groups by page number.
582
+
583
+ The Extend API returns bounding box coordinates in its own pixel coordinate
584
+ system (reported in each block's ``metadata.page.width/height``). We use
585
+ those pixel dimensions for normalization. The ``page_dims`` argument (PDF
586
+ point dimensions) is only used as a fallback when block-level metadata is
587
+ absent.
588
+ """
589
+ # Normalize page_dims keys to int (JSON serialization may stringify them).
590
+ # These are PDF-point dims used only as a last-resort fallback.
591
+ norm_dims: dict[int, tuple[float, float]] = {}
592
+ for k, v in page_dims.items():
593
+ try:
594
+ page_key = int(k)
595
+ if isinstance(v, (list, tuple)) and len(v) == 2:
596
+ norm_dims[page_key] = (float(v[0]), float(v[1]))
597
+ except (TypeError, ValueError):
598
+ continue
599
+
600
+ pages_items: dict[int, list[LayoutItemIR]] = defaultdict(list)
601
+ pages_headers: dict[int, list[str]] = defaultdict(list)
602
+ pages_footers: dict[int, list[str]] = defaultdict(list)
603
+
604
+ for chunk in chunks:
605
+ if not isinstance(chunk, dict):
606
+ continue
607
+
608
+ blocks = chunk.get("blocks", [])
609
+ if not isinstance(blocks, list):
610
+ continue
611
+
612
+ for block in blocks:
613
+ if not isinstance(block, dict):
614
+ continue
615
+
616
+ block_type = block.get("type", "")
617
+ canonical_label = EXTEND_LABEL_MAP.get(block_type)
618
+ if canonical_label is None:
619
+ continue
620
+
621
+ bbox = block.get("boundingBox") or block.get("bounding_box") or {}
622
+ if not isinstance(bbox, dict):
623
+ continue
624
+
625
+ left = float(bbox.get("left", 0.0))
626
+ top = float(bbox.get("top", 0.0))
627
+ right = float(bbox.get("right", 0.0))
628
+ bottom = float(bbox.get("bottom", 0.0))
629
+
630
+ # Extract page number and pixel dimensions from block metadata
631
+ block_meta = block.get("metadata", {}) or {}
632
+ block_page_meta = block_meta.get("page", {}) or {}
633
+ page_num = block_page_meta.get("number") or block.get("page") or block.get("pageNumber") or 1
634
+ if isinstance(page_num, str):
635
+ try:
636
+ page_num = int(page_num)
637
+ except ValueError:
638
+ page_num = 1
639
+
640
+ # Use pixel dimensions from the API's block metadata (the coordinate
641
+ # system the bbox values are expressed in). Fall back to PDF-point
642
+ # dims only when the API does not report per-block page dimensions.
643
+ pixel_w = float(block_page_meta.get("width", 0))
644
+ pixel_h = float(block_page_meta.get("height", 0))
645
+ if pixel_w > 0 and pixel_h > 0:
646
+ pw, ph = pixel_w, pixel_h
647
+ else:
648
+ pw, ph = norm_dims.get(page_num, (0, 0))
649
+
650
+ if pw > 0 and ph > 0:
651
+ x_norm = left / pw
652
+ y_norm = top / ph
653
+ w_norm = (right - left) / pw
654
+ h_norm = (bottom - top) / ph
655
+ else:
656
+ # Fallback: store raw values (adapter will handle as-is)
657
+ x_norm = left
658
+ y_norm = top
659
+ w_norm = right - left
660
+ h_norm = bottom - top
661
+
662
+ confidence = float(block.get("confidence", 1.0))
663
+
664
+ seg = LayoutSegmentIR(
665
+ x=x_norm,
666
+ y=y_norm,
667
+ w=w_norm,
668
+ h=h_norm,
669
+ confidence=confidence,
670
+ label=canonical_label,
671
+ )
672
+
673
+ content = block.get("content", "") or block.get("text", "")
674
+ norm_label = canonical_label.strip().lower()
675
+ if norm_label == "table":
676
+ item_type = "table"
677
+ elif norm_label == "picture":
678
+ item_type = "image"
679
+ else:
680
+ item_type = "text"
681
+
682
+ pages_items[page_num].append(
683
+ LayoutItemIR(
684
+ type=item_type,
685
+ value=content,
686
+ bbox=seg,
687
+ layout_segments=[seg],
688
+ )
689
+ )
690
+
691
+ section_content = f"<page_number>{content}</page_number>" if block_type == "page_number" else content
692
+ if canonical_label == "Page-header" and content:
693
+ pages_headers[page_num].append(section_content)
694
+ elif canonical_label == "Page-footer" and content:
695
+ pages_footers[page_num].append(section_content)
696
+
697
+ layout_pages: list[ParseLayoutPageIR] = []
698
+ for page_num in sorted(pages_items.keys()):
699
+ layout_pages.append(
700
+ ParseLayoutPageIR(
701
+ page_number=page_num,
702
+ width=_VIRTUAL_PAGE_DIM,
703
+ height=_VIRTUAL_PAGE_DIM,
704
+ items=pages_items[page_num],
705
+ page_header_markdown="\n\n".join(pages_headers.get(page_num, [])),
706
+ page_footer_markdown="\n\n".join(pages_footers.get(page_num, [])),
707
+ )
708
+ )
709
+
710
+ return layout_pages