parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
"""Provider for Docling via the official docling-serve HTTP API."""
|
|
2
|
+
|
|
3
|
+
import base64
|
|
4
|
+
import os
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import requests
|
|
10
|
+
from docling_core.transforms.serializer.html import HTMLTableSerializer
|
|
11
|
+
from docling_core.transforms.serializer.markdown import MarkdownDocSerializer
|
|
12
|
+
from docling_core.types.doc.base import ImageRefMode
|
|
13
|
+
from docling_core.types.doc.document import DoclingDocument
|
|
14
|
+
|
|
15
|
+
from parse_bench.inference.providers.base import (
|
|
16
|
+
Provider,
|
|
17
|
+
ProviderConfigError,
|
|
18
|
+
ProviderPermanentError,
|
|
19
|
+
ProviderRateLimitError,
|
|
20
|
+
ProviderTransientError,
|
|
21
|
+
)
|
|
22
|
+
from parse_bench.inference.providers.parse._docling_common import _build_docling_layout_pages
|
|
23
|
+
from parse_bench.inference.providers.registry import register_provider
|
|
24
|
+
from parse_bench.schemas.parse_output import (
|
|
25
|
+
PageIR,
|
|
26
|
+
ParseLayoutPageIR,
|
|
27
|
+
ParseOutput,
|
|
28
|
+
)
|
|
29
|
+
from parse_bench.schemas.pipeline import PipelineSpec
|
|
30
|
+
from parse_bench.schemas.pipeline_io import (
|
|
31
|
+
InferenceRequest,
|
|
32
|
+
InferenceResult,
|
|
33
|
+
RawInferenceResult,
|
|
34
|
+
)
|
|
35
|
+
from parse_bench.schemas.product import ProductType
|
|
36
|
+
|
|
37
|
+
_MD_PAGE_BREAK_PLACEHOLDER = "<!-- page-break -->"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@register_provider("docling_serve")
|
|
41
|
+
class DoclingServeProvider(Provider):
|
|
42
|
+
"""
|
|
43
|
+
Provider for Docling PDF parsing via the official docling-serve HTTP API.
|
|
44
|
+
|
|
45
|
+
This provider sends PDFs to the docling-serve HTTP API endpoint and returns markdown
|
|
46
|
+
with tables formatted as HTML. It was tested with docling-serve v1.17.0.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(
|
|
50
|
+
self,
|
|
51
|
+
provider_name: str,
|
|
52
|
+
base_config: dict[str, Any] | None = None,
|
|
53
|
+
):
|
|
54
|
+
"""
|
|
55
|
+
Initialize the Docling Serve provider.
|
|
56
|
+
|
|
57
|
+
Args:
|
|
58
|
+
provider_name: Name of the provider
|
|
59
|
+
base_config: Optional configuration with:
|
|
60
|
+
- `api_key`: Optional bearer token for the endpoint
|
|
61
|
+
- `endpoint_url`: Endpoint URL (required)
|
|
62
|
+
- `timeout`: Request timeout in seconds (default: 120)
|
|
63
|
+
"""
|
|
64
|
+
super().__init__(provider_name, base_config)
|
|
65
|
+
|
|
66
|
+
self._api_key = self.base_config.get("api_key") or os.getenv("DOCLING_SERVE_API_KEY") or ""
|
|
67
|
+
|
|
68
|
+
# Get endpoint URL (from config or env var)
|
|
69
|
+
self._endpoint_url = self.base_config.get("endpoint_url") or os.getenv("DOCLING_SERVE_ENDPOINT_URL")
|
|
70
|
+
self._endpoint_url = self._endpoint_url.rstrip("/")
|
|
71
|
+
if not self._endpoint_url:
|
|
72
|
+
raise ProviderConfigError(
|
|
73
|
+
"Docling Serve endpoint URL is required. "
|
|
74
|
+
"Set DOCLING_SERVE_ENDPOINT_URL environment variable or "
|
|
75
|
+
"pass endpoint_url in pipeline config."
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
# Get timeout (default 120 seconds - PDF processing can be slow)
|
|
79
|
+
self._timeout = self.base_config.get("timeout", 120)
|
|
80
|
+
|
|
81
|
+
def _call_endpoint(self, pdf_bytes: bytes, filename: str) -> dict[str, Any]:
|
|
82
|
+
"""
|
|
83
|
+
Call the Docling endpoint with PDF bytes.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
pdf_bytes: Raw PDF file bytes
|
|
87
|
+
filename: Name of the PDF file
|
|
88
|
+
|
|
89
|
+
Returns:
|
|
90
|
+
Raw JSON response from endpoint
|
|
91
|
+
|
|
92
|
+
Raises:
|
|
93
|
+
ProviderError: For any API errors
|
|
94
|
+
"""
|
|
95
|
+
headers = {"Content-Type": "application/json"}
|
|
96
|
+
if self._api_key:
|
|
97
|
+
headers["Authorization"] = f"Bearer {self._api_key}"
|
|
98
|
+
|
|
99
|
+
# Encode PDF as base64
|
|
100
|
+
pdf_base64 = base64.b64encode(pdf_bytes).decode("utf-8")
|
|
101
|
+
|
|
102
|
+
payload = {
|
|
103
|
+
"sources": [
|
|
104
|
+
{
|
|
105
|
+
"base64_string": pdf_base64,
|
|
106
|
+
"filename": filename,
|
|
107
|
+
"kind": "file",
|
|
108
|
+
}
|
|
109
|
+
],
|
|
110
|
+
"options": {
|
|
111
|
+
"to_formats": ["json"],
|
|
112
|
+
"pipeline": "standard",
|
|
113
|
+
"include_images": False,
|
|
114
|
+
"image_export_mode": "placeholder",
|
|
115
|
+
},
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
try:
|
|
119
|
+
response = requests.post(
|
|
120
|
+
f"{self._endpoint_url}/v1/convert/source",
|
|
121
|
+
headers=headers,
|
|
122
|
+
json=payload,
|
|
123
|
+
timeout=self._timeout,
|
|
124
|
+
)
|
|
125
|
+
response.raise_for_status()
|
|
126
|
+
result_json = response.json()
|
|
127
|
+
if isinstance(result_json, list):
|
|
128
|
+
if not result_json:
|
|
129
|
+
raise ProviderPermanentError("Endpoint returned an empty list response.")
|
|
130
|
+
first_result = result_json[0]
|
|
131
|
+
if not isinstance(first_result, dict):
|
|
132
|
+
raise ProviderPermanentError("Endpoint returned a list response with a non-dict payload.")
|
|
133
|
+
result = first_result
|
|
134
|
+
elif isinstance(result_json, dict):
|
|
135
|
+
result = result_json
|
|
136
|
+
else:
|
|
137
|
+
raise ProviderPermanentError(
|
|
138
|
+
f"Endpoint returned unsupported response type: {type(result_json).__name__}"
|
|
139
|
+
)
|
|
140
|
+
return result
|
|
141
|
+
|
|
142
|
+
except requests.exceptions.Timeout as e:
|
|
143
|
+
raise ProviderTransientError(f"Request timed out: {e}") from e
|
|
144
|
+
except requests.exceptions.ConnectionError as e:
|
|
145
|
+
raise ProviderTransientError(f"Connection error: {e}") from e
|
|
146
|
+
except requests.exceptions.HTTPError as e:
|
|
147
|
+
status_code = e.response.status_code if e.response else None
|
|
148
|
+
if status_code == 422:
|
|
149
|
+
raise ProviderPermanentError(
|
|
150
|
+
"Docling Serve returned 422. Ensure docling-serve >= 1.0 "
|
|
151
|
+
f"(older versions expect 'file_sources' instead of 'sources'): {e}"
|
|
152
|
+
) from e
|
|
153
|
+
elif status_code == 429:
|
|
154
|
+
raise ProviderRateLimitError(f"Rate limit exceeded: {e}") from e
|
|
155
|
+
elif status_code and 500 <= status_code < 600:
|
|
156
|
+
raise ProviderTransientError(f"Server error ({status_code}): {e}") from e
|
|
157
|
+
elif status_code and 400 <= status_code < 500:
|
|
158
|
+
raise ProviderPermanentError(f"Client error ({status_code}): {e}") from e
|
|
159
|
+
else:
|
|
160
|
+
raise ProviderPermanentError(f"HTTP error: {e}") from e
|
|
161
|
+
except (ProviderPermanentError, ProviderTransientError, ProviderRateLimitError):
|
|
162
|
+
raise
|
|
163
|
+
except Exception as e:
|
|
164
|
+
raise ProviderPermanentError(f"Unexpected error calling endpoint: {e}") from e
|
|
165
|
+
|
|
166
|
+
def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
|
|
167
|
+
"""
|
|
168
|
+
Run inference and return raw results.
|
|
169
|
+
|
|
170
|
+
Args:
|
|
171
|
+
pipeline: Pipeline specification
|
|
172
|
+
request: Inference request
|
|
173
|
+
|
|
174
|
+
Returns:
|
|
175
|
+
Raw inference result
|
|
176
|
+
|
|
177
|
+
Raises:
|
|
178
|
+
ProviderError: For any provider-related failures
|
|
179
|
+
"""
|
|
180
|
+
if request.product_type != ProductType.PARSE:
|
|
181
|
+
raise ProviderPermanentError(
|
|
182
|
+
f"DoclingServeProvider only supports PARSE product type, got {request.product_type}"
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
started_at = datetime.now()
|
|
186
|
+
|
|
187
|
+
# Check if file exists
|
|
188
|
+
source_path = Path(request.source_file_path)
|
|
189
|
+
if not source_path.exists():
|
|
190
|
+
raise ProviderPermanentError(f"Source file not found: {source_path}")
|
|
191
|
+
|
|
192
|
+
try:
|
|
193
|
+
# Read PDF bytes
|
|
194
|
+
pdf_bytes = source_path.read_bytes()
|
|
195
|
+
|
|
196
|
+
# Call endpoint
|
|
197
|
+
raw_output = self._call_endpoint(pdf_bytes, source_path.name)
|
|
198
|
+
|
|
199
|
+
completed_at = datetime.now()
|
|
200
|
+
latency_ms = int((completed_at - started_at).total_seconds() * 1000)
|
|
201
|
+
|
|
202
|
+
return RawInferenceResult(
|
|
203
|
+
request=request,
|
|
204
|
+
pipeline=pipeline,
|
|
205
|
+
pipeline_name=pipeline.pipeline_name,
|
|
206
|
+
product_type=request.product_type,
|
|
207
|
+
raw_output=raw_output,
|
|
208
|
+
started_at=started_at,
|
|
209
|
+
completed_at=completed_at,
|
|
210
|
+
latency_in_ms=latency_ms,
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
except (ProviderPermanentError, ProviderTransientError, ProviderRateLimitError):
|
|
214
|
+
raise
|
|
215
|
+
except Exception as e:
|
|
216
|
+
raise ProviderPermanentError(f"Unexpected error during inference: {e}") from e
|
|
217
|
+
|
|
218
|
+
def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
|
|
219
|
+
"""
|
|
220
|
+
Normalize raw inference result to produce ParseOutput.
|
|
221
|
+
|
|
222
|
+
Args:
|
|
223
|
+
raw_result: Raw inference result from run_inference()
|
|
224
|
+
|
|
225
|
+
Returns:
|
|
226
|
+
Inference result with ParseOutput
|
|
227
|
+
|
|
228
|
+
Raises:
|
|
229
|
+
ProviderError: For any normalization failures
|
|
230
|
+
"""
|
|
231
|
+
if raw_result.product_type != ProductType.PARSE:
|
|
232
|
+
raise ProviderPermanentError(
|
|
233
|
+
f"DoclingServeProvider only supports PARSE product type, got {raw_result.product_type}"
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
# Response format:
|
|
237
|
+
# {
|
|
238
|
+
# "document": {
|
|
239
|
+
# "json_content": {...},
|
|
240
|
+
# }
|
|
241
|
+
# }
|
|
242
|
+
full_markdown = ""
|
|
243
|
+
raw_docling_document = raw_result.raw_output.get("document", {}).get("json_content")
|
|
244
|
+
pages: list[PageIR] = []
|
|
245
|
+
|
|
246
|
+
layout_pages: list[ParseLayoutPageIR] = []
|
|
247
|
+
if raw_docling_document is not None:
|
|
248
|
+
try:
|
|
249
|
+
docling_document = DoclingDocument.model_validate(raw_docling_document)
|
|
250
|
+
except Exception as e:
|
|
251
|
+
raise ProviderPermanentError(f"Failed to validate docling_document payload: {e}") from e
|
|
252
|
+
|
|
253
|
+
doc_serializer = MarkdownDocSerializer(doc=docling_document)
|
|
254
|
+
doc_serializer.table_serializer = HTMLTableSerializer()
|
|
255
|
+
|
|
256
|
+
full_markdown = doc_serializer.serialize(
|
|
257
|
+
page_break_placeholder=_MD_PAGE_BREAK_PLACEHOLDER, image_mode=ImageRefMode.PLACEHOLDER
|
|
258
|
+
).text
|
|
259
|
+
raw_pages_md = full_markdown.split(_MD_PAGE_BREAK_PLACEHOLDER)
|
|
260
|
+
raw_pages_dicts = []
|
|
261
|
+
|
|
262
|
+
for page_index, markdown in enumerate(raw_pages_md):
|
|
263
|
+
pages.append(PageIR(page_index=page_index, markdown=markdown))
|
|
264
|
+
raw_pages_dicts.append({"page": page_index + 1, "markdown": markdown})
|
|
265
|
+
|
|
266
|
+
layout_pages = _build_docling_layout_pages(
|
|
267
|
+
doc=docling_document,
|
|
268
|
+
raw_pages=raw_pages_dicts,
|
|
269
|
+
)
|
|
270
|
+
|
|
271
|
+
output = ParseOutput(
|
|
272
|
+
task_type="parse",
|
|
273
|
+
example_id=raw_result.request.example_id,
|
|
274
|
+
pipeline_name=raw_result.pipeline_name,
|
|
275
|
+
pages=pages,
|
|
276
|
+
layout_pages=layout_pages,
|
|
277
|
+
markdown=full_markdown,
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
return InferenceResult(
|
|
281
|
+
request=raw_result.request,
|
|
282
|
+
pipeline_name=raw_result.pipeline_name,
|
|
283
|
+
product_type=raw_result.product_type,
|
|
284
|
+
raw_output=raw_result.raw_output,
|
|
285
|
+
output=output,
|
|
286
|
+
started_at=raw_result.started_at,
|
|
287
|
+
completed_at=raw_result.completed_at,
|
|
288
|
+
latency_in_ms=raw_result.latency_in_ms,
|
|
289
|
+
)
|