parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Provider for YOLO-DocLayNet layout detection."""
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from parse_bench.inference.providers.base import ProviderPermanentError
|
|
6
|
+
from parse_bench.inference.providers.layoutdet.base import HFLayoutDetProvider
|
|
7
|
+
from parse_bench.inference.providers.registry import register_provider
|
|
8
|
+
from parse_bench.schemas.layout_detection_output import (
|
|
9
|
+
LayoutDetectionModel,
|
|
10
|
+
LayoutOutput,
|
|
11
|
+
LayoutPrediction,
|
|
12
|
+
YoloLabel,
|
|
13
|
+
)
|
|
14
|
+
from parse_bench.schemas.pipeline_io import InferenceResult, RawInferenceResult
|
|
15
|
+
from parse_bench.schemas.product import ProductType
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@register_provider("yolo_layout")
|
|
19
|
+
class YoloLayoutProvider(HFLayoutDetProvider):
|
|
20
|
+
"""
|
|
21
|
+
Provider for YOLO-DocLayNet layout detection model.
|
|
22
|
+
|
|
23
|
+
This provider uses the YOLO model trained on DocLayNet served on HuggingFace
|
|
24
|
+
inference endpoints for detecting document layout regions.
|
|
25
|
+
|
|
26
|
+
Response format:
|
|
27
|
+
{
|
|
28
|
+
"pred_boxes": [[x1, y1, x2, y2], ...],
|
|
29
|
+
"pred_classes": [class_id, ...],
|
|
30
|
+
"scores": [score, ...]
|
|
31
|
+
}
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
endpoint_url = "https://exqoktya7l52qu1r.us-east-1.aws.endpoints.huggingface.cloud"
|
|
35
|
+
model_type = LayoutDetectionModel.YOLO_DOCLAYNET
|
|
36
|
+
|
|
37
|
+
def __init__(
|
|
38
|
+
self,
|
|
39
|
+
provider_name: str,
|
|
40
|
+
base_config: dict[str, Any] | None = None,
|
|
41
|
+
):
|
|
42
|
+
"""Initialize the YOLO-DocLayNet layout detection provider."""
|
|
43
|
+
super().__init__(provider_name, base_config)
|
|
44
|
+
|
|
45
|
+
def _parse_response(self, response: dict[str, Any]) -> list[LayoutPrediction]:
|
|
46
|
+
"""
|
|
47
|
+
Parse YOLO-DocLayNet response into layout predictions.
|
|
48
|
+
|
|
49
|
+
:param response: Raw JSON response with pred_boxes, pred_classes, scores
|
|
50
|
+
:return: List of unified LayoutPrediction objects
|
|
51
|
+
"""
|
|
52
|
+
predictions: list[LayoutPrediction] = []
|
|
53
|
+
|
|
54
|
+
boxes = response.get("pred_boxes", [])
|
|
55
|
+
classes = response.get("pred_classes", [])
|
|
56
|
+
scores = response.get("scores", [])
|
|
57
|
+
|
|
58
|
+
for bbox, class_id, score in zip(boxes, classes, scores, strict=False):
|
|
59
|
+
# Convert class_id to YoloLabel enum
|
|
60
|
+
# Model outputs 0-indexed labels (0-10) that match YoloLabel directly
|
|
61
|
+
label = YoloLabel(class_id)
|
|
62
|
+
predictions.append(
|
|
63
|
+
LayoutPrediction(
|
|
64
|
+
bbox=bbox,
|
|
65
|
+
score=score,
|
|
66
|
+
label=str(int(label)),
|
|
67
|
+
provider_metadata={"label_name": label.name},
|
|
68
|
+
)
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
return predictions
|
|
72
|
+
|
|
73
|
+
def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
|
|
74
|
+
"""
|
|
75
|
+
Normalize raw inference result to produce LayoutOutput.
|
|
76
|
+
|
|
77
|
+
:param raw_result: Raw inference result from run_inference()
|
|
78
|
+
:return: Inference result with both raw and normalized outputs
|
|
79
|
+
:raises ProviderError: For any normalization failures
|
|
80
|
+
"""
|
|
81
|
+
if raw_result.product_type != ProductType.LAYOUT_DETECTION:
|
|
82
|
+
raise ProviderPermanentError(
|
|
83
|
+
f"{self.__class__.__name__} only supports LAYOUT_DETECTION product type, got {raw_result.product_type}"
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
# Parse the response into raw predictions
|
|
87
|
+
response = raw_result.raw_output.get("response", {})
|
|
88
|
+
raw_predictions = self._parse_response(response)
|
|
89
|
+
|
|
90
|
+
output = LayoutOutput(
|
|
91
|
+
task_type="layout_detection",
|
|
92
|
+
example_id=raw_result.request.example_id,
|
|
93
|
+
pipeline_name=raw_result.pipeline_name,
|
|
94
|
+
model=self.model_type,
|
|
95
|
+
image_width=max(int(raw_result.raw_output.get("image_width", 1)), 1),
|
|
96
|
+
image_height=max(int(raw_result.raw_output.get("image_height", 1)), 1),
|
|
97
|
+
predictions=raw_predictions,
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
return InferenceResult(
|
|
101
|
+
request=raw_result.request,
|
|
102
|
+
pipeline_name=raw_result.pipeline_name,
|
|
103
|
+
product_type=raw_result.product_type,
|
|
104
|
+
raw_output=raw_result.raw_output,
|
|
105
|
+
output=output,
|
|
106
|
+
started_at=raw_result.started_at,
|
|
107
|
+
completed_at=raw_result.completed_at,
|
|
108
|
+
latency_in_ms=raw_result.latency_in_ms,
|
|
109
|
+
)
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Parse providers — imported lazily to avoid requiring all SDKs."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import logging
|
|
5
|
+
|
|
6
|
+
logger = logging.getLogger(__name__)
|
|
7
|
+
|
|
8
|
+
_PROVIDER_MODULES = [
|
|
9
|
+
"amazon_nova",
|
|
10
|
+
"anthropic",
|
|
11
|
+
"azure_document_intelligence",
|
|
12
|
+
"chandra2",
|
|
13
|
+
"chunkr",
|
|
14
|
+
"databricks_ai_parse",
|
|
15
|
+
"datalab",
|
|
16
|
+
"deepseekocr2",
|
|
17
|
+
"docling",
|
|
18
|
+
"docling_serve",
|
|
19
|
+
"dots_ocr",
|
|
20
|
+
"extend_parse",
|
|
21
|
+
"falconocr",
|
|
22
|
+
"florin_parser_nano",
|
|
23
|
+
"gemma4",
|
|
24
|
+
"glm_zai",
|
|
25
|
+
"google",
|
|
26
|
+
"google_docai",
|
|
27
|
+
"granite_vision",
|
|
28
|
+
"infinity_parser2",
|
|
29
|
+
"kdl_frontier_nano",
|
|
30
|
+
"landingai",
|
|
31
|
+
"liteparse",
|
|
32
|
+
"markitdown",
|
|
33
|
+
"opendataloader",
|
|
34
|
+
"pdf_inspector",
|
|
35
|
+
"pymupdf4llm",
|
|
36
|
+
"rakedoc_nano",
|
|
37
|
+
"llamaparse",
|
|
38
|
+
"llamaparse_v2_normalization",
|
|
39
|
+
"mineru25",
|
|
40
|
+
"mineru2605pro",
|
|
41
|
+
"mineru_diffusion",
|
|
42
|
+
"mistral_ocr",
|
|
43
|
+
"nemotron_omni",
|
|
44
|
+
"openai",
|
|
45
|
+
"paddleocr",
|
|
46
|
+
"pulse",
|
|
47
|
+
"pymupdf",
|
|
48
|
+
"pypdf",
|
|
49
|
+
"qwen",
|
|
50
|
+
"reducto",
|
|
51
|
+
"surya2",
|
|
52
|
+
"tesseract",
|
|
53
|
+
"textract",
|
|
54
|
+
"unlimitedocr",
|
|
55
|
+
"unstructured",
|
|
56
|
+
"warp_ingest",
|
|
57
|
+
"oi_parser",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
for _mod in _PROVIDER_MODULES:
|
|
61
|
+
try:
|
|
62
|
+
importlib.import_module(f"parse_bench.inference.providers.parse.{_mod}")
|
|
63
|
+
except ImportError:
|
|
64
|
+
logger.debug("Skipping parse provider %s (missing dependency)", _mod)
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Common functionality for docling and docling_serve providers."""
|
|
2
|
+
|
|
3
|
+
from typing import Any
|
|
4
|
+
|
|
5
|
+
from docling_core.types.doc.document import DoclingDocument
|
|
6
|
+
|
|
7
|
+
from parse_bench.layout_label_mapping import (
|
|
8
|
+
UnknownRawLayoutLabelError,
|
|
9
|
+
map_docling_raw_label_to_canonical,
|
|
10
|
+
)
|
|
11
|
+
from parse_bench.schemas.parse_output import (
|
|
12
|
+
LayoutItemIR,
|
|
13
|
+
LayoutSegmentIR,
|
|
14
|
+
ParseLayoutPageIR,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
_DOCLING_EXCLUDED_LAYOUT_LABELS = frozenset(
|
|
18
|
+
{
|
|
19
|
+
"empty_value",
|
|
20
|
+
"field_heading",
|
|
21
|
+
"field_hint",
|
|
22
|
+
"field_item",
|
|
23
|
+
"field_key",
|
|
24
|
+
"field_region",
|
|
25
|
+
"field_value",
|
|
26
|
+
"marker",
|
|
27
|
+
}
|
|
28
|
+
)
|
|
29
|
+
_DOCLING_TABLE_LABELS = frozenset({"document_index", "table"})
|
|
30
|
+
_DOCLING_IMAGE_LABELS = frozenset({"chart", "picture"})
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _normalize_docling_label(label: object) -> str | None:
|
|
34
|
+
if label is None:
|
|
35
|
+
return None
|
|
36
|
+
value = getattr(label, "value", label)
|
|
37
|
+
if not isinstance(value, str):
|
|
38
|
+
return None
|
|
39
|
+
return value.strip().lower()
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _should_include_docling_label(raw_label: str) -> bool:
|
|
43
|
+
if raw_label in _DOCLING_EXCLUDED_LAYOUT_LABELS:
|
|
44
|
+
return False
|
|
45
|
+
try:
|
|
46
|
+
map_docling_raw_label_to_canonical(raw_label)
|
|
47
|
+
except UnknownRawLayoutLabelError:
|
|
48
|
+
return False
|
|
49
|
+
return True
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _docling_item_type(raw_label: str) -> str:
|
|
53
|
+
if raw_label in _DOCLING_TABLE_LABELS:
|
|
54
|
+
return "table"
|
|
55
|
+
if raw_label in _DOCLING_IMAGE_LABELS:
|
|
56
|
+
return "image"
|
|
57
|
+
return "text"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _extract_docling_item_value(item: Any, doc: DoclingDocument, raw_label: str) -> str:
|
|
61
|
+
item_type = _docling_item_type(raw_label)
|
|
62
|
+
if item_type == "image":
|
|
63
|
+
return ""
|
|
64
|
+
|
|
65
|
+
if item_type == "table" and hasattr(item, "export_to_html"):
|
|
66
|
+
try:
|
|
67
|
+
html = item.export_to_html(doc=doc, add_caption=True)
|
|
68
|
+
if isinstance(html, str):
|
|
69
|
+
return html
|
|
70
|
+
except Exception:
|
|
71
|
+
pass
|
|
72
|
+
|
|
73
|
+
text = getattr(item, "text", None)
|
|
74
|
+
if isinstance(text, str):
|
|
75
|
+
return text
|
|
76
|
+
|
|
77
|
+
if hasattr(item, "export_to_markdown"):
|
|
78
|
+
try:
|
|
79
|
+
markdown = item.export_to_markdown()
|
|
80
|
+
if isinstance(markdown, str):
|
|
81
|
+
return markdown
|
|
82
|
+
except Exception:
|
|
83
|
+
pass
|
|
84
|
+
|
|
85
|
+
return ""
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _normalize_docling_charspan(
|
|
89
|
+
charspan: object,
|
|
90
|
+
*,
|
|
91
|
+
text_length: int,
|
|
92
|
+
include_span: bool,
|
|
93
|
+
) -> tuple[int | None, int | None]:
|
|
94
|
+
if not include_span or not isinstance(charspan, (list, tuple)) or len(charspan) != 2:
|
|
95
|
+
return (None, None)
|
|
96
|
+
|
|
97
|
+
start_raw, end_raw = charspan
|
|
98
|
+
if not isinstance(start_raw, int) or not isinstance(end_raw, int):
|
|
99
|
+
return (None, None)
|
|
100
|
+
|
|
101
|
+
start = max(0, min(start_raw, text_length))
|
|
102
|
+
end_exclusive = max(start, min(end_raw, text_length))
|
|
103
|
+
if end_exclusive <= start:
|
|
104
|
+
return (None, None)
|
|
105
|
+
|
|
106
|
+
# Docling charspan behaves like a Python slice [start, end).
|
|
107
|
+
return (start, end_exclusive - 1)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _build_docling_segment(
|
|
111
|
+
*,
|
|
112
|
+
prov: Any,
|
|
113
|
+
raw_label: str,
|
|
114
|
+
page_width: float,
|
|
115
|
+
page_height: float,
|
|
116
|
+
include_span: bool,
|
|
117
|
+
text_length: int,
|
|
118
|
+
) -> LayoutSegmentIR | None:
|
|
119
|
+
bbox = getattr(prov, "bbox", None)
|
|
120
|
+
if bbox is None or page_width <= 0 or page_height <= 0:
|
|
121
|
+
return None
|
|
122
|
+
|
|
123
|
+
bbox_top_left = bbox.to_top_left_origin(page_height=page_height)
|
|
124
|
+
width = bbox_top_left.r - bbox_top_left.l
|
|
125
|
+
height = bbox_top_left.b - bbox_top_left.t
|
|
126
|
+
if width <= 0 or height <= 0:
|
|
127
|
+
return None
|
|
128
|
+
|
|
129
|
+
start_index, end_index = _normalize_docling_charspan(
|
|
130
|
+
getattr(prov, "charspan", None),
|
|
131
|
+
text_length=text_length,
|
|
132
|
+
include_span=include_span,
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
return LayoutSegmentIR(
|
|
136
|
+
x=bbox_top_left.l / page_width,
|
|
137
|
+
y=bbox_top_left.t / page_height,
|
|
138
|
+
w=width / page_width,
|
|
139
|
+
h=height / page_height,
|
|
140
|
+
confidence=1.0,
|
|
141
|
+
label=raw_label,
|
|
142
|
+
start_index=start_index,
|
|
143
|
+
end_index=end_index,
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _merge_segments(segments: list[LayoutSegmentIR]) -> LayoutSegmentIR | None:
|
|
148
|
+
if not segments:
|
|
149
|
+
return None
|
|
150
|
+
|
|
151
|
+
x1 = min(segment.x for segment in segments)
|
|
152
|
+
y1 = min(segment.y for segment in segments)
|
|
153
|
+
x2 = max(segment.x + segment.w for segment in segments)
|
|
154
|
+
y2 = max(segment.y + segment.h for segment in segments)
|
|
155
|
+
return LayoutSegmentIR(
|
|
156
|
+
x=x1,
|
|
157
|
+
y=y1,
|
|
158
|
+
w=x2 - x1,
|
|
159
|
+
h=y2 - y1,
|
|
160
|
+
confidence=1.0,
|
|
161
|
+
label=segments[0].label,
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _build_docling_layout_pages(
|
|
166
|
+
*,
|
|
167
|
+
doc: DoclingDocument,
|
|
168
|
+
raw_pages: list[dict[str, Any]],
|
|
169
|
+
) -> list[ParseLayoutPageIR]:
|
|
170
|
+
page_markdown_by_number: dict[int, str] = {}
|
|
171
|
+
for page_data in raw_pages:
|
|
172
|
+
page_number = page_data.get("page")
|
|
173
|
+
if isinstance(page_number, int) and page_number > 0:
|
|
174
|
+
page_markdown_by_number[page_number] = str(page_data.get("markdown", ""))
|
|
175
|
+
|
|
176
|
+
layout_pages: list[ParseLayoutPageIR] = []
|
|
177
|
+
for page_number in sorted(doc.pages.keys()):
|
|
178
|
+
page = doc.pages[page_number]
|
|
179
|
+
page_width = float(page.size.width)
|
|
180
|
+
page_height = float(page.size.height)
|
|
181
|
+
items: list[LayoutItemIR] = []
|
|
182
|
+
|
|
183
|
+
for item, _level in doc.iterate_items(page_no=page_number):
|
|
184
|
+
raw_label = _normalize_docling_label(getattr(item, "label", None))
|
|
185
|
+
if raw_label is None or not _should_include_docling_label(raw_label):
|
|
186
|
+
continue
|
|
187
|
+
|
|
188
|
+
item_type = _docling_item_type(raw_label)
|
|
189
|
+
item_value = _extract_docling_item_value(item, doc, raw_label)
|
|
190
|
+
include_span = item_type == "text"
|
|
191
|
+
|
|
192
|
+
page_provs = [
|
|
193
|
+
prov for prov in getattr(item, "prov", []) or [] if getattr(prov, "page_no", None) == page_number
|
|
194
|
+
]
|
|
195
|
+
segments = [
|
|
196
|
+
segment
|
|
197
|
+
for prov in page_provs
|
|
198
|
+
if (
|
|
199
|
+
segment := _build_docling_segment(
|
|
200
|
+
prov=prov,
|
|
201
|
+
raw_label=raw_label,
|
|
202
|
+
page_width=page_width,
|
|
203
|
+
page_height=page_height,
|
|
204
|
+
include_span=include_span,
|
|
205
|
+
text_length=len(item_value),
|
|
206
|
+
)
|
|
207
|
+
)
|
|
208
|
+
is not None
|
|
209
|
+
]
|
|
210
|
+
if not segments:
|
|
211
|
+
continue
|
|
212
|
+
|
|
213
|
+
merged_bbox = _merge_segments(segments)
|
|
214
|
+
items.append(
|
|
215
|
+
LayoutItemIR(
|
|
216
|
+
type=item_type,
|
|
217
|
+
value=item_value,
|
|
218
|
+
bbox=merged_bbox,
|
|
219
|
+
layout_segments=segments,
|
|
220
|
+
)
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
layout_pages.append(
|
|
224
|
+
ParseLayoutPageIR(
|
|
225
|
+
page_number=page_number,
|
|
226
|
+
width=page_width,
|
|
227
|
+
height=page_height,
|
|
228
|
+
md=page_markdown_by_number.get(page_number, ""),
|
|
229
|
+
items=items,
|
|
230
|
+
)
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
return layout_pages
|