parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
File without changes
@@ -0,0 +1,28 @@
1
+ """Provider implementations for document parsing."""
2
+
3
+ # Import providers to register them
4
+ from parse_bench.inference.providers import (
5
+ extract, # noqa: F401
6
+ layoutdet, # noqa: F401
7
+ parse, # noqa: F401
8
+ )
9
+ from parse_bench.inference.providers.base import (
10
+ Provider,
11
+ ProviderConfigError,
12
+ ProviderError,
13
+ ProviderPermanentError,
14
+ ProviderRateLimitError,
15
+ ProviderTransientError,
16
+ )
17
+ from parse_bench.inference.providers.registry import create_provider, register_provider
18
+
19
+ __all__ = [
20
+ "Provider",
21
+ "ProviderConfigError",
22
+ "ProviderError",
23
+ "ProviderPermanentError",
24
+ "ProviderRateLimitError",
25
+ "ProviderTransientError",
26
+ "create_provider",
27
+ "register_provider",
28
+ ]
@@ -0,0 +1,196 @@
1
+ import asyncio
2
+ import concurrent.futures
3
+ from abc import ABC, abstractmethod
4
+ from collections.abc import Coroutine
5
+ from typing import Any
6
+
7
+ from parse_bench.schemas.pipeline import PipelineSpec
8
+ from parse_bench.schemas.pipeline_io import (
9
+ InferenceRequest,
10
+ InferenceResult,
11
+ RawInferenceResult,
12
+ )
13
+
14
+
15
+ class ProviderError(Exception):
16
+ """Base exception for provider-related failures."""
17
+
18
+ def __init__(self, message: str, *, debug_payload: dict[str, Any] | None = None):
19
+ super().__init__(message)
20
+ self.debug_payload = debug_payload
21
+
22
+
23
+ class ProviderConfigError(ProviderError):
24
+ """Raised when a provider is misconfigured (missing API keys, bad endpoint, etc.)."""
25
+
26
+
27
+ class ProviderRateLimitError(ProviderError):
28
+ """Raised when a provider hits rate limits or quota issues."""
29
+
30
+
31
+ class ProviderTransientError(ProviderError):
32
+ """
33
+ Raised for transient errors that may succeed on retry
34
+ (e.g. network issues, 5xx responses).
35
+ """
36
+
37
+
38
+ class ProviderPermanentError(ProviderError):
39
+ """
40
+ Raised for permanent errors that are not expected to succeed on retry
41
+ (e.g. unsupported file type, invalid request, 4xx errors).
42
+ """
43
+
44
+
45
+ class Provider(ABC):
46
+ """Abstract base class for document parsing providers."""
47
+
48
+ def __init__(
49
+ self,
50
+ provider_name: str,
51
+ base_config: dict[str, Any] | None = None,
52
+ ):
53
+ """
54
+ Initialize a provider.
55
+
56
+ :param provider_name: Name of the provider
57
+ :param base_config: Optional shared configuration dictionary.
58
+ Can include `use_staging` (bool) to use staging environment.
59
+ """
60
+ self._provider_name = provider_name
61
+ self._base_config = base_config or {}
62
+
63
+ @property
64
+ def provider_name(self) -> str:
65
+ """Return the provider name."""
66
+ return self._provider_name
67
+
68
+ @property
69
+ def base_config(self) -> dict[str, Any]:
70
+ """Return the base configuration."""
71
+ return self._base_config
72
+
73
+ @property
74
+ def credit_rate_usd(self) -> float | None:
75
+ """USD cost per credit. Override in subclasses that charge credits."""
76
+ return None
77
+
78
+ @staticmethod
79
+ def run_async_from_sync(coro: Coroutine[Any, Any, Any]) -> Any:
80
+ """
81
+ Run an async coroutine from a synchronous context.
82
+
83
+ This helper handles both cases:
84
+ - If there's no running event loop, uses asyncio.run()
85
+ - If there's a running event loop, runs the coroutine in a new thread with a new event loop
86
+
87
+ :param coro: The coroutine to run
88
+ :return: The result of the coroutine
89
+ """
90
+ try:
91
+ # Try to get the current event loop
92
+ # If we get here, there's a running loop, so we need to run in a thread
93
+ asyncio.get_running_loop()
94
+ with concurrent.futures.ThreadPoolExecutor() as executor:
95
+ future = executor.submit(asyncio.run, coro)
96
+ return future.result()
97
+ except RuntimeError:
98
+ # No running event loop, we can use asyncio.run() directly
99
+ return asyncio.run(coro)
100
+
101
+ @abstractmethod
102
+ def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
103
+ """
104
+ Run inference for a single request and return raw results.
105
+
106
+ This method should only fetch raw data from the provider API.
107
+ Normalization is handled separately by the normalize() method.
108
+
109
+ :param pipeline: Pipeline specification
110
+ :param request: Inference request
111
+ :return: Raw inference result (before normalization)
112
+ :raises ProviderError: For any provider-related failures
113
+ """
114
+ raise NotImplementedError("Subclasses must implement this method")
115
+
116
+ @abstractmethod
117
+ def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
118
+ """
119
+ Normalize raw inference result to produce structured output.
120
+
121
+ This method converts the raw API response into a structured
122
+ format (ParseOutput or ExtractOutput) while preserving the
123
+ raw output for potential re-normalization.
124
+
125
+ Note: Each provider implementation is product-type specific
126
+ and will return either ParseOutput or ExtractOutput, not both.
127
+
128
+ :param raw_result: Raw inference result from run_inference()
129
+ :return: Inference result with both raw and normalized outputs
130
+ :raises ProviderError: For any normalization failures
131
+ """
132
+ raise NotImplementedError("Subclasses must implement this method")
133
+
134
+ def recompute_cost(self, raw_output: dict[str, Any]) -> None:
135
+ """Re-derive the cost fields in ``raw_output`` from the usage it already
136
+ records, using this provider's current pricing.
137
+
138
+ This is the SINGLE generic seam the re-normalization path calls for every
139
+ provider (see ``inference/renormalize.py``), so correcting a pricing table
140
+ and re-running ``bench inference renormalize`` re-prices saved runs with no
141
+ fresh (expensive) inference. It is deliberately pipeline- and
142
+ provider-agnostic at the call site: the runner never names a provider.
143
+
144
+ What is irreducibly provider-specific is the rate card -- only this
145
+ provider knows what it charges for a cached token, a cache write, a tool
146
+ container, or a long-context tier -- so the base implementation is a no-op
147
+ and a provider that can re-price overrides this one method. A provider
148
+ whose cost came from an upstream credits field (not tokens we hold) simply
149
+ does not override it and keeps what it recorded.
150
+
151
+ Contract for overrides: mutate ``raw_output`` in place, derive everything
152
+ from what is already there (so it works on a saved ``.raw.json``), be
153
+ idempotent, and never call the API. Leave cost untouched when the recorded
154
+ usage is missing -- re-pricing a usage-less artifact would overwrite a real
155
+ number with a zero.
156
+ """
157
+ return None
158
+
159
+ def cancel(self, example_id: str) -> bool:
160
+ """
161
+ Cancel any in-flight inference work for the given ``example_id``.
162
+
163
+ Default implementation is a no-op for providers that do not spawn
164
+ external resources (subprocesses, remote jobs) that need explicit
165
+ teardown. Providers that fork subprocesses or hold network handles
166
+ should override this to actually terminate the work — otherwise
167
+ per-file timeouts in the runner will only release the calling
168
+ thread while the underlying work continues to run, racing the
169
+ retry attempt and producing duplicate / zombie processes.
170
+
171
+ :param example_id: The ``InferenceRequest.example_id`` for the
172
+ in-flight request that should be cancelled.
173
+ :return: True if a matching in-flight request was found and a
174
+ cancellation signal was issued; False otherwise.
175
+ """
176
+ return False
177
+
178
+ def consume_active_job_id(self, example_id: str) -> str | None:
179
+ """Consume a provider-owned active remote job id, if one exists."""
180
+
181
+ return None
182
+
183
+ def run_inference_normalized(self, pipeline: PipelineSpec, request: InferenceRequest) -> InferenceResult:
184
+ """
185
+ Run inference and normalize in one step (convenience method).
186
+
187
+ This is a convenience method that combines run_inference() and
188
+ normalize() for backward compatibility and simple use cases.
189
+
190
+ :param pipeline: Pipeline specification
191
+ :param request: Inference request
192
+ :return: Inference result with both raw and normalized outputs
193
+ :raises ProviderError: For any provider-related failures
194
+ """
195
+ raw_result = self.run_inference(pipeline, request)
196
+ return self.normalize(raw_result)
@@ -0,0 +1,137 @@
1
+ """Reusable per-``example_id`` cancellation registry for HTTP/SDK providers.
2
+
3
+ When the runner's per-file timeout fires, ``ThreadPoolExecutor.shutdown(wait=False)``
4
+ only releases the calling thread - any in-flight HTTP request the worker
5
+ thread spawned (httpx polling, requests session, LlamaCloud SDK call) keeps
6
+ running, and the next retry attempt sends a duplicate request to staging.
7
+ Closing the underlying client breaks the provider's polling loop on its
8
+ next iteration so the worker thread unwinds with a transient error and the
9
+ retry loop can submit a fresh request without piling on parallel duplicates.
10
+
11
+ Important caveat - what closing a client does and does not abort:
12
+
13
+ * It DOES break a polling loop that calls ``client.get(...)`` repeatedly
14
+ on a long-running job. The next call after ``close()`` raises immediately
15
+ (httpx: ``RuntimeError: Cannot send a request, as the client has been closed.``;
16
+ requests: ``ConnectionError`` on the next ``session.get``).
17
+
18
+ * It does NOT interrupt a thread already blocked inside a single socket
19
+ read on another thread - Python threads are not OS-cancellable, and
20
+ closing the client object only marks it closed; the kernel ``recv`` call
21
+ finishes only when the server responds or the read timeout expires.
22
+
23
+ For the bench bug - duplicate requests to staging during per-file timeout
24
+ retries - the polling-loop case is the one that matters. Long-running
25
+ parse / extract jobs are polled in tight loops; closing the client makes
26
+ the next poll raise within milliseconds. Per-request read timeouts on the
27
+ underlying client cap the worst-case stalled-read tail.
28
+
29
+ This module provides a tiny helper that:
30
+ * registers a closeable handle (httpx.Client, requests.Session,
31
+ llama_cloud.LlamaCloud, ...) keyed by ``example_id``;
32
+ * exposes a ``cancel(example_id)`` that pops the handle and calls
33
+ ``.close()`` (best-effort - providers swallow secondary errors so the
34
+ cancel path can never break the runner).
35
+
36
+ Each provider holds one ``CancellableClientRegistry`` instance, registers
37
+ its client at the start of a request, and unregisters in a ``finally``.
38
+ The registry is thread-safe so concurrent ``run_inference`` calls (one per
39
+ ``example_id``) do not collide.
40
+ """
41
+
42
+ from __future__ import annotations
43
+
44
+ import logging
45
+ import threading
46
+ from typing import Protocol
47
+
48
+ logger = logging.getLogger(__name__)
49
+
50
+
51
+ class _Closeable(Protocol):
52
+ """Anything with a no-arg ``close()``: ``httpx.Client``, ``requests.Session``,
53
+ ``llama_cloud.LlamaCloud`` (closes its underlying ``httpx.Client``), ..."""
54
+
55
+ def close(self) -> None: ...
56
+
57
+
58
+ class CancellableClientRegistry:
59
+ """Thread-safe per-``example_id`` mapping of in-flight HTTP/SDK clients.
60
+
61
+ Providers should:
62
+
63
+ # In __init__:
64
+ self._inflight = CancellableClientRegistry(provider_name="...")
65
+
66
+ # At the start of run_inference (after the client is built):
67
+ self._inflight.register(request.example_id, client)
68
+ try:
69
+ ...
70
+ finally:
71
+ self._inflight.unregister(request.example_id, client)
72
+
73
+ # In cancel(example_id):
74
+ return self._inflight.cancel(example_id)
75
+
76
+ The registry never raises from ``cancel`` - a broken cancel must not
77
+ break the runner's retry loop.
78
+ """
79
+
80
+ def __init__(self, *, provider_name: str) -> None:
81
+ self._provider_name = provider_name
82
+ self._lock = threading.Lock()
83
+ self._inflight: dict[str, _Closeable] = {}
84
+
85
+ def register(self, example_id: str, client: _Closeable) -> None:
86
+ """Track ``client`` so a later ``cancel(example_id)`` can close it.
87
+
88
+ If the slot is already occupied (e.g. because the previous attempt's
89
+ cleanup raced the next attempt's submit), the new client wins - the
90
+ old one was either already cancelled or about to be unregistered.
91
+ """
92
+ with self._lock:
93
+ self._inflight[example_id] = client
94
+
95
+ def unregister(self, example_id: str, client: _Closeable) -> None:
96
+ """Remove ``client`` from the registry if it is the live entry.
97
+
98
+ Compares by identity so we never clobber a registration from a
99
+ concurrent retry attempt. Idempotent - safe to call from a
100
+ ``finally`` even when ``cancel`` already popped the entry.
101
+ """
102
+ with self._lock:
103
+ current = self._inflight.get(example_id)
104
+ if current is client:
105
+ self._inflight.pop(example_id, None)
106
+
107
+ def cancel(self, example_id: str) -> bool:
108
+ """Pop and close any registered client for ``example_id``.
109
+
110
+ :return: True if a matching client was found and ``close()`` was
111
+ attempted (regardless of whether close itself succeeded), False
112
+ if no client was registered.
113
+ """
114
+ with self._lock:
115
+ client = self._inflight.pop(example_id, None)
116
+ if client is None:
117
+ return False
118
+
119
+ logger.info(
120
+ "%s.cancel: closing in-flight client for example_id=%s",
121
+ self._provider_name,
122
+ example_id,
123
+ )
124
+ try:
125
+ client.close()
126
+ except Exception as exc: # noqa: BLE001 - cancel must never raise
127
+ # Closing a client mid-request can surface various provider-
128
+ # specific exceptions (httpx connection state, broken pipe,
129
+ # SDK wrappers raising their own types). None of them should
130
+ # break the runner's retry loop.
131
+ logger.debug(
132
+ "%s.cancel: client.close() raised for example_id=%s: %s",
133
+ self._provider_name,
134
+ example_id,
135
+ exc,
136
+ )
137
+ return True
@@ -0,0 +1,22 @@
1
+ """Extract providers imported for registry side effects.
2
+
3
+ Only the minimal ParseBench EXTRACT integration providers are registered here.
4
+ Imports are best-effort so base ParseBench imports do not require optional
5
+ provider SDKs such as ``extend-ai``.
6
+ """
7
+
8
+ import importlib
9
+ import logging
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+ _PROVIDER_MODULES = [
14
+ "extend",
15
+ "llamaextract_v2_api",
16
+ ]
17
+
18
+ for _mod in _PROVIDER_MODULES:
19
+ try:
20
+ importlib.import_module(f"parse_bench.inference.providers.extract.{_mod}")
21
+ except ImportError:
22
+ logger.debug("Skipping extract provider %s (missing dependency)", _mod)