parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
File without changes
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Provider implementations for document parsing."""
|
|
2
|
+
|
|
3
|
+
# Import providers to register them
|
|
4
|
+
from parse_bench.inference.providers import (
|
|
5
|
+
extract, # noqa: F401
|
|
6
|
+
layoutdet, # noqa: F401
|
|
7
|
+
parse, # noqa: F401
|
|
8
|
+
)
|
|
9
|
+
from parse_bench.inference.providers.base import (
|
|
10
|
+
Provider,
|
|
11
|
+
ProviderConfigError,
|
|
12
|
+
ProviderError,
|
|
13
|
+
ProviderPermanentError,
|
|
14
|
+
ProviderRateLimitError,
|
|
15
|
+
ProviderTransientError,
|
|
16
|
+
)
|
|
17
|
+
from parse_bench.inference.providers.registry import create_provider, register_provider
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"Provider",
|
|
21
|
+
"ProviderConfigError",
|
|
22
|
+
"ProviderError",
|
|
23
|
+
"ProviderPermanentError",
|
|
24
|
+
"ProviderRateLimitError",
|
|
25
|
+
"ProviderTransientError",
|
|
26
|
+
"create_provider",
|
|
27
|
+
"register_provider",
|
|
28
|
+
]
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import concurrent.futures
|
|
3
|
+
from abc import ABC, abstractmethod
|
|
4
|
+
from collections.abc import Coroutine
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from parse_bench.schemas.pipeline import PipelineSpec
|
|
8
|
+
from parse_bench.schemas.pipeline_io import (
|
|
9
|
+
InferenceRequest,
|
|
10
|
+
InferenceResult,
|
|
11
|
+
RawInferenceResult,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ProviderError(Exception):
|
|
16
|
+
"""Base exception for provider-related failures."""
|
|
17
|
+
|
|
18
|
+
def __init__(self, message: str, *, debug_payload: dict[str, Any] | None = None):
|
|
19
|
+
super().__init__(message)
|
|
20
|
+
self.debug_payload = debug_payload
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class ProviderConfigError(ProviderError):
|
|
24
|
+
"""Raised when a provider is misconfigured (missing API keys, bad endpoint, etc.)."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class ProviderRateLimitError(ProviderError):
|
|
28
|
+
"""Raised when a provider hits rate limits or quota issues."""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ProviderTransientError(ProviderError):
|
|
32
|
+
"""
|
|
33
|
+
Raised for transient errors that may succeed on retry
|
|
34
|
+
(e.g. network issues, 5xx responses).
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ProviderPermanentError(ProviderError):
|
|
39
|
+
"""
|
|
40
|
+
Raised for permanent errors that are not expected to succeed on retry
|
|
41
|
+
(e.g. unsupported file type, invalid request, 4xx errors).
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Provider(ABC):
|
|
46
|
+
"""Abstract base class for document parsing providers."""
|
|
47
|
+
|
|
48
|
+
def __init__(
|
|
49
|
+
self,
|
|
50
|
+
provider_name: str,
|
|
51
|
+
base_config: dict[str, Any] | None = None,
|
|
52
|
+
):
|
|
53
|
+
"""
|
|
54
|
+
Initialize a provider.
|
|
55
|
+
|
|
56
|
+
:param provider_name: Name of the provider
|
|
57
|
+
:param base_config: Optional shared configuration dictionary.
|
|
58
|
+
Can include `use_staging` (bool) to use staging environment.
|
|
59
|
+
"""
|
|
60
|
+
self._provider_name = provider_name
|
|
61
|
+
self._base_config = base_config or {}
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def provider_name(self) -> str:
|
|
65
|
+
"""Return the provider name."""
|
|
66
|
+
return self._provider_name
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def base_config(self) -> dict[str, Any]:
|
|
70
|
+
"""Return the base configuration."""
|
|
71
|
+
return self._base_config
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def credit_rate_usd(self) -> float | None:
|
|
75
|
+
"""USD cost per credit. Override in subclasses that charge credits."""
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
@staticmethod
|
|
79
|
+
def run_async_from_sync(coro: Coroutine[Any, Any, Any]) -> Any:
|
|
80
|
+
"""
|
|
81
|
+
Run an async coroutine from a synchronous context.
|
|
82
|
+
|
|
83
|
+
This helper handles both cases:
|
|
84
|
+
- If there's no running event loop, uses asyncio.run()
|
|
85
|
+
- If there's a running event loop, runs the coroutine in a new thread with a new event loop
|
|
86
|
+
|
|
87
|
+
:param coro: The coroutine to run
|
|
88
|
+
:return: The result of the coroutine
|
|
89
|
+
"""
|
|
90
|
+
try:
|
|
91
|
+
# Try to get the current event loop
|
|
92
|
+
# If we get here, there's a running loop, so we need to run in a thread
|
|
93
|
+
asyncio.get_running_loop()
|
|
94
|
+
with concurrent.futures.ThreadPoolExecutor() as executor:
|
|
95
|
+
future = executor.submit(asyncio.run, coro)
|
|
96
|
+
return future.result()
|
|
97
|
+
except RuntimeError:
|
|
98
|
+
# No running event loop, we can use asyncio.run() directly
|
|
99
|
+
return asyncio.run(coro)
|
|
100
|
+
|
|
101
|
+
@abstractmethod
|
|
102
|
+
def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
|
|
103
|
+
"""
|
|
104
|
+
Run inference for a single request and return raw results.
|
|
105
|
+
|
|
106
|
+
This method should only fetch raw data from the provider API.
|
|
107
|
+
Normalization is handled separately by the normalize() method.
|
|
108
|
+
|
|
109
|
+
:param pipeline: Pipeline specification
|
|
110
|
+
:param request: Inference request
|
|
111
|
+
:return: Raw inference result (before normalization)
|
|
112
|
+
:raises ProviderError: For any provider-related failures
|
|
113
|
+
"""
|
|
114
|
+
raise NotImplementedError("Subclasses must implement this method")
|
|
115
|
+
|
|
116
|
+
@abstractmethod
|
|
117
|
+
def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
|
|
118
|
+
"""
|
|
119
|
+
Normalize raw inference result to produce structured output.
|
|
120
|
+
|
|
121
|
+
This method converts the raw API response into a structured
|
|
122
|
+
format (ParseOutput or ExtractOutput) while preserving the
|
|
123
|
+
raw output for potential re-normalization.
|
|
124
|
+
|
|
125
|
+
Note: Each provider implementation is product-type specific
|
|
126
|
+
and will return either ParseOutput or ExtractOutput, not both.
|
|
127
|
+
|
|
128
|
+
:param raw_result: Raw inference result from run_inference()
|
|
129
|
+
:return: Inference result with both raw and normalized outputs
|
|
130
|
+
:raises ProviderError: For any normalization failures
|
|
131
|
+
"""
|
|
132
|
+
raise NotImplementedError("Subclasses must implement this method")
|
|
133
|
+
|
|
134
|
+
def recompute_cost(self, raw_output: dict[str, Any]) -> None:
|
|
135
|
+
"""Re-derive the cost fields in ``raw_output`` from the usage it already
|
|
136
|
+
records, using this provider's current pricing.
|
|
137
|
+
|
|
138
|
+
This is the SINGLE generic seam the re-normalization path calls for every
|
|
139
|
+
provider (see ``inference/renormalize.py``), so correcting a pricing table
|
|
140
|
+
and re-running ``bench inference renormalize`` re-prices saved runs with no
|
|
141
|
+
fresh (expensive) inference. It is deliberately pipeline- and
|
|
142
|
+
provider-agnostic at the call site: the runner never names a provider.
|
|
143
|
+
|
|
144
|
+
What is irreducibly provider-specific is the rate card -- only this
|
|
145
|
+
provider knows what it charges for a cached token, a cache write, a tool
|
|
146
|
+
container, or a long-context tier -- so the base implementation is a no-op
|
|
147
|
+
and a provider that can re-price overrides this one method. A provider
|
|
148
|
+
whose cost came from an upstream credits field (not tokens we hold) simply
|
|
149
|
+
does not override it and keeps what it recorded.
|
|
150
|
+
|
|
151
|
+
Contract for overrides: mutate ``raw_output`` in place, derive everything
|
|
152
|
+
from what is already there (so it works on a saved ``.raw.json``), be
|
|
153
|
+
idempotent, and never call the API. Leave cost untouched when the recorded
|
|
154
|
+
usage is missing -- re-pricing a usage-less artifact would overwrite a real
|
|
155
|
+
number with a zero.
|
|
156
|
+
"""
|
|
157
|
+
return None
|
|
158
|
+
|
|
159
|
+
def cancel(self, example_id: str) -> bool:
|
|
160
|
+
"""
|
|
161
|
+
Cancel any in-flight inference work for the given ``example_id``.
|
|
162
|
+
|
|
163
|
+
Default implementation is a no-op for providers that do not spawn
|
|
164
|
+
external resources (subprocesses, remote jobs) that need explicit
|
|
165
|
+
teardown. Providers that fork subprocesses or hold network handles
|
|
166
|
+
should override this to actually terminate the work — otherwise
|
|
167
|
+
per-file timeouts in the runner will only release the calling
|
|
168
|
+
thread while the underlying work continues to run, racing the
|
|
169
|
+
retry attempt and producing duplicate / zombie processes.
|
|
170
|
+
|
|
171
|
+
:param example_id: The ``InferenceRequest.example_id`` for the
|
|
172
|
+
in-flight request that should be cancelled.
|
|
173
|
+
:return: True if a matching in-flight request was found and a
|
|
174
|
+
cancellation signal was issued; False otherwise.
|
|
175
|
+
"""
|
|
176
|
+
return False
|
|
177
|
+
|
|
178
|
+
def consume_active_job_id(self, example_id: str) -> str | None:
|
|
179
|
+
"""Consume a provider-owned active remote job id, if one exists."""
|
|
180
|
+
|
|
181
|
+
return None
|
|
182
|
+
|
|
183
|
+
def run_inference_normalized(self, pipeline: PipelineSpec, request: InferenceRequest) -> InferenceResult:
|
|
184
|
+
"""
|
|
185
|
+
Run inference and normalize in one step (convenience method).
|
|
186
|
+
|
|
187
|
+
This is a convenience method that combines run_inference() and
|
|
188
|
+
normalize() for backward compatibility and simple use cases.
|
|
189
|
+
|
|
190
|
+
:param pipeline: Pipeline specification
|
|
191
|
+
:param request: Inference request
|
|
192
|
+
:return: Inference result with both raw and normalized outputs
|
|
193
|
+
:raises ProviderError: For any provider-related failures
|
|
194
|
+
"""
|
|
195
|
+
raw_result = self.run_inference(pipeline, request)
|
|
196
|
+
return self.normalize(raw_result)
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Reusable per-``example_id`` cancellation registry for HTTP/SDK providers.
|
|
2
|
+
|
|
3
|
+
When the runner's per-file timeout fires, ``ThreadPoolExecutor.shutdown(wait=False)``
|
|
4
|
+
only releases the calling thread - any in-flight HTTP request the worker
|
|
5
|
+
thread spawned (httpx polling, requests session, LlamaCloud SDK call) keeps
|
|
6
|
+
running, and the next retry attempt sends a duplicate request to staging.
|
|
7
|
+
Closing the underlying client breaks the provider's polling loop on its
|
|
8
|
+
next iteration so the worker thread unwinds with a transient error and the
|
|
9
|
+
retry loop can submit a fresh request without piling on parallel duplicates.
|
|
10
|
+
|
|
11
|
+
Important caveat - what closing a client does and does not abort:
|
|
12
|
+
|
|
13
|
+
* It DOES break a polling loop that calls ``client.get(...)`` repeatedly
|
|
14
|
+
on a long-running job. The next call after ``close()`` raises immediately
|
|
15
|
+
(httpx: ``RuntimeError: Cannot send a request, as the client has been closed.``;
|
|
16
|
+
requests: ``ConnectionError`` on the next ``session.get``).
|
|
17
|
+
|
|
18
|
+
* It does NOT interrupt a thread already blocked inside a single socket
|
|
19
|
+
read on another thread - Python threads are not OS-cancellable, and
|
|
20
|
+
closing the client object only marks it closed; the kernel ``recv`` call
|
|
21
|
+
finishes only when the server responds or the read timeout expires.
|
|
22
|
+
|
|
23
|
+
For the bench bug - duplicate requests to staging during per-file timeout
|
|
24
|
+
retries - the polling-loop case is the one that matters. Long-running
|
|
25
|
+
parse / extract jobs are polled in tight loops; closing the client makes
|
|
26
|
+
the next poll raise within milliseconds. Per-request read timeouts on the
|
|
27
|
+
underlying client cap the worst-case stalled-read tail.
|
|
28
|
+
|
|
29
|
+
This module provides a tiny helper that:
|
|
30
|
+
* registers a closeable handle (httpx.Client, requests.Session,
|
|
31
|
+
llama_cloud.LlamaCloud, ...) keyed by ``example_id``;
|
|
32
|
+
* exposes a ``cancel(example_id)`` that pops the handle and calls
|
|
33
|
+
``.close()`` (best-effort - providers swallow secondary errors so the
|
|
34
|
+
cancel path can never break the runner).
|
|
35
|
+
|
|
36
|
+
Each provider holds one ``CancellableClientRegistry`` instance, registers
|
|
37
|
+
its client at the start of a request, and unregisters in a ``finally``.
|
|
38
|
+
The registry is thread-safe so concurrent ``run_inference`` calls (one per
|
|
39
|
+
``example_id``) do not collide.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
from __future__ import annotations
|
|
43
|
+
|
|
44
|
+
import logging
|
|
45
|
+
import threading
|
|
46
|
+
from typing import Protocol
|
|
47
|
+
|
|
48
|
+
logger = logging.getLogger(__name__)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class _Closeable(Protocol):
|
|
52
|
+
"""Anything with a no-arg ``close()``: ``httpx.Client``, ``requests.Session``,
|
|
53
|
+
``llama_cloud.LlamaCloud`` (closes its underlying ``httpx.Client``), ..."""
|
|
54
|
+
|
|
55
|
+
def close(self) -> None: ...
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class CancellableClientRegistry:
|
|
59
|
+
"""Thread-safe per-``example_id`` mapping of in-flight HTTP/SDK clients.
|
|
60
|
+
|
|
61
|
+
Providers should:
|
|
62
|
+
|
|
63
|
+
# In __init__:
|
|
64
|
+
self._inflight = CancellableClientRegistry(provider_name="...")
|
|
65
|
+
|
|
66
|
+
# At the start of run_inference (after the client is built):
|
|
67
|
+
self._inflight.register(request.example_id, client)
|
|
68
|
+
try:
|
|
69
|
+
...
|
|
70
|
+
finally:
|
|
71
|
+
self._inflight.unregister(request.example_id, client)
|
|
72
|
+
|
|
73
|
+
# In cancel(example_id):
|
|
74
|
+
return self._inflight.cancel(example_id)
|
|
75
|
+
|
|
76
|
+
The registry never raises from ``cancel`` - a broken cancel must not
|
|
77
|
+
break the runner's retry loop.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
def __init__(self, *, provider_name: str) -> None:
|
|
81
|
+
self._provider_name = provider_name
|
|
82
|
+
self._lock = threading.Lock()
|
|
83
|
+
self._inflight: dict[str, _Closeable] = {}
|
|
84
|
+
|
|
85
|
+
def register(self, example_id: str, client: _Closeable) -> None:
|
|
86
|
+
"""Track ``client`` so a later ``cancel(example_id)`` can close it.
|
|
87
|
+
|
|
88
|
+
If the slot is already occupied (e.g. because the previous attempt's
|
|
89
|
+
cleanup raced the next attempt's submit), the new client wins - the
|
|
90
|
+
old one was either already cancelled or about to be unregistered.
|
|
91
|
+
"""
|
|
92
|
+
with self._lock:
|
|
93
|
+
self._inflight[example_id] = client
|
|
94
|
+
|
|
95
|
+
def unregister(self, example_id: str, client: _Closeable) -> None:
|
|
96
|
+
"""Remove ``client`` from the registry if it is the live entry.
|
|
97
|
+
|
|
98
|
+
Compares by identity so we never clobber a registration from a
|
|
99
|
+
concurrent retry attempt. Idempotent - safe to call from a
|
|
100
|
+
``finally`` even when ``cancel`` already popped the entry.
|
|
101
|
+
"""
|
|
102
|
+
with self._lock:
|
|
103
|
+
current = self._inflight.get(example_id)
|
|
104
|
+
if current is client:
|
|
105
|
+
self._inflight.pop(example_id, None)
|
|
106
|
+
|
|
107
|
+
def cancel(self, example_id: str) -> bool:
|
|
108
|
+
"""Pop and close any registered client for ``example_id``.
|
|
109
|
+
|
|
110
|
+
:return: True if a matching client was found and ``close()`` was
|
|
111
|
+
attempted (regardless of whether close itself succeeded), False
|
|
112
|
+
if no client was registered.
|
|
113
|
+
"""
|
|
114
|
+
with self._lock:
|
|
115
|
+
client = self._inflight.pop(example_id, None)
|
|
116
|
+
if client is None:
|
|
117
|
+
return False
|
|
118
|
+
|
|
119
|
+
logger.info(
|
|
120
|
+
"%s.cancel: closing in-flight client for example_id=%s",
|
|
121
|
+
self._provider_name,
|
|
122
|
+
example_id,
|
|
123
|
+
)
|
|
124
|
+
try:
|
|
125
|
+
client.close()
|
|
126
|
+
except Exception as exc: # noqa: BLE001 - cancel must never raise
|
|
127
|
+
# Closing a client mid-request can surface various provider-
|
|
128
|
+
# specific exceptions (httpx connection state, broken pipe,
|
|
129
|
+
# SDK wrappers raising their own types). None of them should
|
|
130
|
+
# break the runner's retry loop.
|
|
131
|
+
logger.debug(
|
|
132
|
+
"%s.cancel: client.close() raised for example_id=%s: %s",
|
|
133
|
+
self._provider_name,
|
|
134
|
+
example_id,
|
|
135
|
+
exc,
|
|
136
|
+
)
|
|
137
|
+
return True
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Extract providers imported for registry side effects.
|
|
2
|
+
|
|
3
|
+
Only the minimal ParseBench EXTRACT integration providers are registered here.
|
|
4
|
+
Imports are best-effort so base ParseBench imports do not require optional
|
|
5
|
+
provider SDKs such as ``extend-ai``.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import importlib
|
|
9
|
+
import logging
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
_PROVIDER_MODULES = [
|
|
14
|
+
"extend",
|
|
15
|
+
"llamaextract_v2_api",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
for _mod in _PROVIDER_MODULES:
|
|
19
|
+
try:
|
|
20
|
+
importlib.import_module(f"parse_bench.inference.providers.extract.{_mod}")
|
|
21
|
+
except ImportError:
|
|
22
|
+
logger.debug("Skipping extract provider %s (missing dependency)", _mod)
|