parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,710 @@
|
|
|
1
|
+
"""Provider for Extend AI PARSE using the official Python SDK.
|
|
2
|
+
|
|
3
|
+
Based on Extend AI documentation: https://docs.extend.ai/product/parsing/parse
|
|
4
|
+
SDK: pip install extend-ai
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import os
|
|
8
|
+
import threading
|
|
9
|
+
from collections import defaultdict
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from extend_ai import Extend, FileFromIdParams, ParseConfigParams
|
|
15
|
+
from extend_ai.core.api_error import ApiError
|
|
16
|
+
from extend_ai.types import ParseConfigChunkingStrategy
|
|
17
|
+
from pypdf import PdfReader
|
|
18
|
+
|
|
19
|
+
from parse_bench.inference.providers.base import (
|
|
20
|
+
Provider,
|
|
21
|
+
ProviderConfigError,
|
|
22
|
+
ProviderPermanentError,
|
|
23
|
+
ProviderRateLimitError,
|
|
24
|
+
ProviderTransientError,
|
|
25
|
+
)
|
|
26
|
+
from parse_bench.inference.providers.registry import register_provider
|
|
27
|
+
from parse_bench.schemas.parse_output import (
|
|
28
|
+
LayoutItemIR,
|
|
29
|
+
LayoutSegmentIR,
|
|
30
|
+
PageIR,
|
|
31
|
+
ParseLayoutPageIR,
|
|
32
|
+
ParseOutput,
|
|
33
|
+
)
|
|
34
|
+
from parse_bench.schemas.pipeline import PipelineSpec
|
|
35
|
+
from parse_bench.schemas.pipeline_io import (
|
|
36
|
+
InferenceRequest,
|
|
37
|
+
InferenceResult,
|
|
38
|
+
RawInferenceResult,
|
|
39
|
+
)
|
|
40
|
+
from parse_bench.schemas.product import ProductType
|
|
41
|
+
|
|
42
|
+
# Extend block type -> Canonical17 label string
|
|
43
|
+
EXTEND_LABEL_MAP: dict[str, str] = {
|
|
44
|
+
"heading": "Section-header",
|
|
45
|
+
"section_heading": "Section-header",
|
|
46
|
+
"text": "Text",
|
|
47
|
+
"table": "Table",
|
|
48
|
+
"figure": "Picture",
|
|
49
|
+
"header": "Page-header",
|
|
50
|
+
"footer": "Page-footer",
|
|
51
|
+
"key_value": "Key-Value Region",
|
|
52
|
+
"page_number": "Page-footer",
|
|
53
|
+
"formula": "Formula",
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
# Virtual page dimensions for normalized coordinate conversion.
|
|
57
|
+
# Extend bboxes are converted to [0,1] using PDF page dims, so these cancel out.
|
|
58
|
+
_VIRTUAL_PAGE_DIM = 1000.0
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@register_provider("extend_parse")
|
|
62
|
+
class ExtendParseProvider(Provider):
|
|
63
|
+
"""
|
|
64
|
+
Provider for Extend AI document parsing using the official SDK.
|
|
65
|
+
|
|
66
|
+
This provider uses the extend-ai Python SDK for parsing tasks.
|
|
67
|
+
SDK Documentation: https://docs.extend.ai/developers/sd-ks
|
|
68
|
+
|
|
69
|
+
Workflow:
|
|
70
|
+
1. Upload file via client.file.upload()
|
|
71
|
+
2. Call client.parse() with configuration options
|
|
72
|
+
3. Return markdown content from parsed result
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
# Extend meters parsing in credits; one credit is $0.0125.
|
|
76
|
+
CREDIT_RATE_USD = 0.0125
|
|
77
|
+
|
|
78
|
+
# Credits billed per page, keyed by parse engine. The default engine (no
|
|
79
|
+
# explicit `engine` in the pipeline config) bills at the parse_performance rate.
|
|
80
|
+
ENGINE_CREDITS_PER_PAGE: dict[str, float] = {
|
|
81
|
+
"parse_performance": 2.0,
|
|
82
|
+
"parse_light": 0.5,
|
|
83
|
+
}
|
|
84
|
+
DEFAULT_CREDITS_PER_PAGE = 2.0
|
|
85
|
+
|
|
86
|
+
def __init__(
|
|
87
|
+
self,
|
|
88
|
+
provider_name: str,
|
|
89
|
+
base_config: dict[str, Any] | None = None,
|
|
90
|
+
):
|
|
91
|
+
"""
|
|
92
|
+
Initialize the provider.
|
|
93
|
+
|
|
94
|
+
:param provider_name: Name of the provider
|
|
95
|
+
:param base_config: Optional configuration with:
|
|
96
|
+
- `api_key`: Extend AI API key (defaults to EXTEND_API_KEY env var)
|
|
97
|
+
- `base_url`: Optional base URL for different deployments
|
|
98
|
+
(default: https://api.extend.ai, alternatives: https://api.us2.extend.app,
|
|
99
|
+
https://api.eu1.extend.ai)
|
|
100
|
+
- `timeout`: Request timeout in seconds (default: 300)
|
|
101
|
+
- `chunking_strategy`: "page", "section", or "document" (default: "page")
|
|
102
|
+
- `target`: Output format - "markdown" or "spatial" (default: "markdown")
|
|
103
|
+
"""
|
|
104
|
+
super().__init__(provider_name, base_config)
|
|
105
|
+
|
|
106
|
+
# Get API key
|
|
107
|
+
api_key = self.base_config.get("api_key") or os.getenv("EXTEND_API_KEY")
|
|
108
|
+
if not api_key:
|
|
109
|
+
raise ProviderConfigError(
|
|
110
|
+
"Extend AI API key is required. Set EXTEND_API_KEY environment variable or pass api_key in base_config."
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
# Configuration
|
|
114
|
+
timeout = self.base_config.get("timeout", 300)
|
|
115
|
+
|
|
116
|
+
# Initialize the Extend client
|
|
117
|
+
client_kwargs: dict[str, Any] = {
|
|
118
|
+
"token": api_key,
|
|
119
|
+
"timeout": float(timeout),
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
# Optional base URL for different deployments (US2, EU1, etc.)
|
|
123
|
+
base_url = self.base_config.get("base_url")
|
|
124
|
+
if base_url:
|
|
125
|
+
client_kwargs["base_url"] = base_url
|
|
126
|
+
|
|
127
|
+
self._client = Extend(**client_kwargs)
|
|
128
|
+
|
|
129
|
+
# Thread lock for file uploads
|
|
130
|
+
self._upload_lock = threading.Lock()
|
|
131
|
+
|
|
132
|
+
def _handle_api_error(self, e: ApiError, context: str) -> None:
|
|
133
|
+
"""Convert SDK ApiError to appropriate ProviderError."""
|
|
134
|
+
status_code = getattr(e, "status_code", None)
|
|
135
|
+
error_body = getattr(e, "body", str(e))
|
|
136
|
+
|
|
137
|
+
if status_code == 429:
|
|
138
|
+
raise ProviderRateLimitError(f"Rate limit exceeded during {context}: {error_body}")
|
|
139
|
+
elif status_code in (502, 503, 504):
|
|
140
|
+
raise ProviderTransientError(f"Transient error during {context}: {status_code} - {error_body}")
|
|
141
|
+
elif status_code and status_code >= 400:
|
|
142
|
+
raise ProviderPermanentError(f"Error during {context}: {status_code} - {error_body}")
|
|
143
|
+
else:
|
|
144
|
+
raise ProviderPermanentError(f"API error during {context}: {error_body}")
|
|
145
|
+
|
|
146
|
+
def _is_pdf_file(self, file_path: str) -> bool:
|
|
147
|
+
"""
|
|
148
|
+
Check if a file is a PDF by reading its header.
|
|
149
|
+
|
|
150
|
+
:param file_path: Path to the file
|
|
151
|
+
:return: True if the file is a PDF, False otherwise
|
|
152
|
+
"""
|
|
153
|
+
try:
|
|
154
|
+
with open(file_path, "rb") as f:
|
|
155
|
+
header = f.read(4)
|
|
156
|
+
return header == b"%PDF"
|
|
157
|
+
except Exception:
|
|
158
|
+
return False
|
|
159
|
+
|
|
160
|
+
def _get_page_count(self, file_path: str) -> int:
|
|
161
|
+
"""
|
|
162
|
+
Get the page count for a file. For PDFs, reads the actual page count.
|
|
163
|
+
For images, returns 1.
|
|
164
|
+
|
|
165
|
+
:param file_path: Path to the file
|
|
166
|
+
:return: Number of pages (1 for images, actual count for PDFs)
|
|
167
|
+
"""
|
|
168
|
+
if self._is_pdf_file(file_path):
|
|
169
|
+
try:
|
|
170
|
+
reader = PdfReader(file_path)
|
|
171
|
+
return len(reader.pages)
|
|
172
|
+
except Exception:
|
|
173
|
+
return 1
|
|
174
|
+
else:
|
|
175
|
+
return 1
|
|
176
|
+
|
|
177
|
+
def _credits_per_page(self, pipeline_config: dict[str, Any]) -> float | None:
|
|
178
|
+
"""
|
|
179
|
+
Resolve how many credits per page the configured engine bills.
|
|
180
|
+
|
|
181
|
+
:param pipeline_config: Pipeline configuration options
|
|
182
|
+
:return: Credits per page, or None if the engine's rate is unknown
|
|
183
|
+
:raises ProviderConfigError: If an explicit override is negative
|
|
184
|
+
"""
|
|
185
|
+
override = pipeline_config.get("credits_per_page", self.base_config.get("credits_per_page"))
|
|
186
|
+
if override is not None:
|
|
187
|
+
credits = float(override)
|
|
188
|
+
if credits < 0:
|
|
189
|
+
raise ProviderConfigError("credits_per_page must be non-negative")
|
|
190
|
+
return credits
|
|
191
|
+
|
|
192
|
+
engine = pipeline_config.get("engine")
|
|
193
|
+
if engine is None:
|
|
194
|
+
return self.DEFAULT_CREDITS_PER_PAGE
|
|
195
|
+
# Unknown engine: omit cost rather than report a rate we cannot vouch for.
|
|
196
|
+
return self.ENGINE_CREDITS_PER_PAGE.get(engine)
|
|
197
|
+
|
|
198
|
+
def _upload_file(self, file_path: str) -> str:
|
|
199
|
+
"""
|
|
200
|
+
Upload a file to Extend AI.
|
|
201
|
+
|
|
202
|
+
:param file_path: Path to the file to upload
|
|
203
|
+
:return: File ID from Extend AI
|
|
204
|
+
:raises ProviderError: For any upload errors
|
|
205
|
+
"""
|
|
206
|
+
try:
|
|
207
|
+
with open(file_path, "rb") as f:
|
|
208
|
+
upload_response = self._client.files.upload(file=f)
|
|
209
|
+
|
|
210
|
+
# Extract file ID from response
|
|
211
|
+
if hasattr(upload_response, "id"):
|
|
212
|
+
return str(upload_response.id)
|
|
213
|
+
elif hasattr(upload_response, "file") and hasattr(upload_response.file, "id"):
|
|
214
|
+
return str(upload_response.file.id)
|
|
215
|
+
elif isinstance(upload_response, dict):
|
|
216
|
+
file_data = upload_response.get("file", upload_response)
|
|
217
|
+
file_id = file_data.get("id") or file_data.get("fileId")
|
|
218
|
+
if file_id:
|
|
219
|
+
return str(file_id)
|
|
220
|
+
|
|
221
|
+
raise ProviderPermanentError(f"No file ID in upload response: {upload_response}")
|
|
222
|
+
|
|
223
|
+
except ApiError as e:
|
|
224
|
+
self._handle_api_error(e, "file upload")
|
|
225
|
+
raise
|
|
226
|
+
except Exception as e:
|
|
227
|
+
error_str = str(e).lower()
|
|
228
|
+
if any(kw in error_str for kw in ["timeout", "timed out", "connection", "network", "readtimeout"]):
|
|
229
|
+
raise ProviderTransientError(f"Transient error during file upload: {e}") from e
|
|
230
|
+
raise ProviderPermanentError(f"Unexpected error during file upload: {e}") from e
|
|
231
|
+
|
|
232
|
+
def _build_parse_config(self, pipeline_config: dict[str, Any]) -> dict[str, Any]:
|
|
233
|
+
"""
|
|
234
|
+
Build the parse config from pipeline configuration.
|
|
235
|
+
|
|
236
|
+
:param pipeline_config: Pipeline configuration options
|
|
237
|
+
:return: Parse configuration dict
|
|
238
|
+
"""
|
|
239
|
+
config: dict[str, Any] = {}
|
|
240
|
+
|
|
241
|
+
# Target format: "markdown" or "spatial"
|
|
242
|
+
if "target" in pipeline_config:
|
|
243
|
+
config["target"] = pipeline_config["target"]
|
|
244
|
+
|
|
245
|
+
# Chunking strategy: "page", "section", or "document"
|
|
246
|
+
if "chunking_strategy" in pipeline_config:
|
|
247
|
+
config["chunking_strategy"] = ParseConfigChunkingStrategy(type=pipeline_config["chunking_strategy"])
|
|
248
|
+
|
|
249
|
+
# Block options for fine-grained control
|
|
250
|
+
if "block_options" in pipeline_config:
|
|
251
|
+
config["block_options"] = pipeline_config["block_options"]
|
|
252
|
+
|
|
253
|
+
# Advanced options (OCR enhancements, page filtering)
|
|
254
|
+
if "advanced_options" in pipeline_config:
|
|
255
|
+
config["advanced_options"] = pipeline_config["advanced_options"]
|
|
256
|
+
|
|
257
|
+
# Engine selection (e.g. "parse_performance")
|
|
258
|
+
if "engine" in pipeline_config:
|
|
259
|
+
config["engine"] = pipeline_config["engine"]
|
|
260
|
+
|
|
261
|
+
# Engine version (e.g. "2.0.0-beta")
|
|
262
|
+
if "engineVersion" in pipeline_config:
|
|
263
|
+
config["engineVersion"] = pipeline_config["engineVersion"]
|
|
264
|
+
|
|
265
|
+
return config
|
|
266
|
+
|
|
267
|
+
def _parse_document(
|
|
268
|
+
self,
|
|
269
|
+
file_path: str,
|
|
270
|
+
pipeline_config: dict[str, Any],
|
|
271
|
+
) -> dict[str, Any]:
|
|
272
|
+
"""
|
|
273
|
+
Parse a document using Extend AI.
|
|
274
|
+
|
|
275
|
+
:param file_path: Path to the document file
|
|
276
|
+
:param pipeline_config: Pipeline configuration options
|
|
277
|
+
:return: Raw API response with parsed content
|
|
278
|
+
:raises ProviderError: For any parsing errors
|
|
279
|
+
"""
|
|
280
|
+
# Get page count and page dimensions (for bbox normalization)
|
|
281
|
+
num_pages = self._get_page_count(file_path)
|
|
282
|
+
page_dims = _get_pdf_page_dims(file_path)
|
|
283
|
+
|
|
284
|
+
# Step 1: Upload file
|
|
285
|
+
file_id = self._upload_file(file_path)
|
|
286
|
+
|
|
287
|
+
# Step 2: Build parse config
|
|
288
|
+
parse_config = self._build_parse_config(pipeline_config)
|
|
289
|
+
|
|
290
|
+
# Step 3: Call parse API
|
|
291
|
+
try:
|
|
292
|
+
# The Extend SDK parse method
|
|
293
|
+
parse_response = self._client.parse(
|
|
294
|
+
file=FileFromIdParams(id=file_id),
|
|
295
|
+
config=ParseConfigParams(**parse_config) if parse_config else None, # type: ignore[typeddict-item]
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
# Convert response to dict
|
|
299
|
+
if hasattr(parse_response, "model_dump"):
|
|
300
|
+
result = parse_response.model_dump()
|
|
301
|
+
elif hasattr(parse_response, "dict"):
|
|
302
|
+
result = parse_response.dict()
|
|
303
|
+
elif isinstance(parse_response, dict):
|
|
304
|
+
result = parse_response
|
|
305
|
+
else:
|
|
306
|
+
# Try to extract attributes manually
|
|
307
|
+
result = {}
|
|
308
|
+
for attr in [
|
|
309
|
+
"id",
|
|
310
|
+
"status",
|
|
311
|
+
"chunks",
|
|
312
|
+
"content",
|
|
313
|
+
"markdown",
|
|
314
|
+
"pages",
|
|
315
|
+
"error",
|
|
316
|
+
"fileId",
|
|
317
|
+
]:
|
|
318
|
+
if hasattr(parse_response, attr):
|
|
319
|
+
value = getattr(parse_response, attr)
|
|
320
|
+
if not callable(value):
|
|
321
|
+
result[attr] = value
|
|
322
|
+
|
|
323
|
+
# Add metadata
|
|
324
|
+
result["_extend_metadata"] = {
|
|
325
|
+
"file_id": file_id,
|
|
326
|
+
"num_pages": num_pages,
|
|
327
|
+
"page_dims": page_dims,
|
|
328
|
+
"config": parse_config,
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
# Operational stats: page count feeds per-page latency, credits feed cost.
|
|
332
|
+
result["num_pages"] = num_pages
|
|
333
|
+
credits_per_page = self._credits_per_page(pipeline_config)
|
|
334
|
+
if credits_per_page is not None:
|
|
335
|
+
cost_per_page_usd = credits_per_page * self.CREDIT_RATE_USD
|
|
336
|
+
result["credits_used"] = credits_per_page * num_pages
|
|
337
|
+
result["cost_per_page_usd"] = cost_per_page_usd
|
|
338
|
+
result["cost_usd"] = cost_per_page_usd * num_pages
|
|
339
|
+
|
|
340
|
+
return result
|
|
341
|
+
|
|
342
|
+
except ApiError as e:
|
|
343
|
+
self._handle_api_error(e, "document parsing")
|
|
344
|
+
raise
|
|
345
|
+
except Exception as e:
|
|
346
|
+
error_str = str(e).lower()
|
|
347
|
+
if any(kw in error_str for kw in ["timeout", "timed out", "connection", "network", "readtimeout"]):
|
|
348
|
+
raise ProviderTransientError(f"Transient error during parsing: {e}") from e
|
|
349
|
+
raise ProviderPermanentError(f"Unexpected error during parsing: {e}") from e
|
|
350
|
+
|
|
351
|
+
def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
|
|
352
|
+
"""
|
|
353
|
+
Run inference and return raw results.
|
|
354
|
+
|
|
355
|
+
:param pipeline: Pipeline specification
|
|
356
|
+
:param request: Inference request
|
|
357
|
+
:return: Raw inference result
|
|
358
|
+
:raises ProviderError: For any provider-related failures
|
|
359
|
+
"""
|
|
360
|
+
if request.product_type != ProductType.PARSE:
|
|
361
|
+
raise ProviderPermanentError(
|
|
362
|
+
f"ExtendParseProvider only supports PARSE product type, got {request.product_type}"
|
|
363
|
+
)
|
|
364
|
+
|
|
365
|
+
started_at = datetime.now()
|
|
366
|
+
|
|
367
|
+
# Check if file exists
|
|
368
|
+
file_path = Path(request.source_file_path)
|
|
369
|
+
if not file_path.exists():
|
|
370
|
+
raise ProviderPermanentError(f"File not found: {file_path}")
|
|
371
|
+
|
|
372
|
+
try:
|
|
373
|
+
# Run parsing with pipeline config options
|
|
374
|
+
raw_output = self._parse_document(
|
|
375
|
+
file_path=str(file_path),
|
|
376
|
+
pipeline_config=pipeline.config,
|
|
377
|
+
)
|
|
378
|
+
|
|
379
|
+
completed_at = datetime.now()
|
|
380
|
+
latency_ms = int((completed_at - started_at).total_seconds() * 1000)
|
|
381
|
+
|
|
382
|
+
return RawInferenceResult(
|
|
383
|
+
request=request,
|
|
384
|
+
pipeline=pipeline,
|
|
385
|
+
pipeline_name=pipeline.pipeline_name,
|
|
386
|
+
product_type=request.product_type,
|
|
387
|
+
raw_output=raw_output,
|
|
388
|
+
started_at=started_at,
|
|
389
|
+
completed_at=completed_at,
|
|
390
|
+
latency_in_ms=latency_ms,
|
|
391
|
+
)
|
|
392
|
+
|
|
393
|
+
except (ProviderPermanentError, ProviderTransientError, ProviderRateLimitError):
|
|
394
|
+
raise
|
|
395
|
+
except Exception as e:
|
|
396
|
+
raise ProviderPermanentError(f"Unexpected error during inference: {e}") from e
|
|
397
|
+
|
|
398
|
+
def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
|
|
399
|
+
"""
|
|
400
|
+
Normalize raw inference result to produce ParseOutput.
|
|
401
|
+
|
|
402
|
+
:param raw_result: Raw inference result from run_inference()
|
|
403
|
+
:return: Inference result with both raw and normalized outputs
|
|
404
|
+
:raises ProviderError: For any normalization failures
|
|
405
|
+
"""
|
|
406
|
+
if raw_result.product_type != ProductType.PARSE:
|
|
407
|
+
raise ProviderPermanentError(
|
|
408
|
+
f"ExtendParseProvider only supports PARSE product type, got {raw_result.product_type}"
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
raw_output = raw_result.raw_output
|
|
412
|
+
|
|
413
|
+
# SDK 1.x wraps content under raw_output["output"]; legacy responses had it at the top level.
|
|
414
|
+
# Source the chunk-bearing payload from whichever shape applies.
|
|
415
|
+
payload = raw_output.get("output") if isinstance(raw_output.get("output"), dict) else raw_output
|
|
416
|
+
|
|
417
|
+
# Extract markdown content from response
|
|
418
|
+
# Extend API can return content in different formats depending on config
|
|
419
|
+
markdown = ""
|
|
420
|
+
|
|
421
|
+
# Try different response formats
|
|
422
|
+
# 1. Direct markdown field
|
|
423
|
+
if "markdown" in payload:
|
|
424
|
+
markdown = payload["markdown"]
|
|
425
|
+
# 2. Content field
|
|
426
|
+
elif "content" in payload:
|
|
427
|
+
content = payload["content"]
|
|
428
|
+
if isinstance(content, str):
|
|
429
|
+
markdown = content
|
|
430
|
+
elif isinstance(content, dict):
|
|
431
|
+
markdown = content.get("markdown", "") or content.get("text", "")
|
|
432
|
+
# 3. Chunks array (similar to Reducto)
|
|
433
|
+
elif "chunks" in payload:
|
|
434
|
+
chunks = payload["chunks"]
|
|
435
|
+
if chunks and isinstance(chunks, list):
|
|
436
|
+
# Concatenate all chunk contents
|
|
437
|
+
chunk_contents = []
|
|
438
|
+
for chunk in chunks:
|
|
439
|
+
if isinstance(chunk, dict):
|
|
440
|
+
chunk_content = chunk.get("content", "") or chunk.get("markdown", "")
|
|
441
|
+
if chunk_content:
|
|
442
|
+
chunk_contents.append(chunk_content)
|
|
443
|
+
elif isinstance(chunk, str):
|
|
444
|
+
chunk_contents.append(chunk)
|
|
445
|
+
markdown = "\n\n".join(chunk_contents)
|
|
446
|
+
# 4. Pages array
|
|
447
|
+
elif "pages" in payload:
|
|
448
|
+
pages = payload["pages"]
|
|
449
|
+
if pages and isinstance(pages, list):
|
|
450
|
+
page_contents = []
|
|
451
|
+
for page in pages:
|
|
452
|
+
if isinstance(page, dict):
|
|
453
|
+
page_content = page.get("markdown", "") or page.get("content", "")
|
|
454
|
+
if page_content:
|
|
455
|
+
page_contents.append(page_content)
|
|
456
|
+
elif isinstance(page, str):
|
|
457
|
+
page_contents.append(page)
|
|
458
|
+
markdown = "\n\n".join(page_contents)
|
|
459
|
+
|
|
460
|
+
# Get job ID if available
|
|
461
|
+
job_id = raw_output.get("id") or raw_output.get("job_id")
|
|
462
|
+
|
|
463
|
+
# Build layout_pages from chunk blocks for layout cross-evaluation
|
|
464
|
+
metadata = raw_output.get("_extend_metadata", {})
|
|
465
|
+
page_dims = metadata.get("page_dims", {})
|
|
466
|
+
chunks = payload.get("chunks", [])
|
|
467
|
+
layout_pages = _build_layout_pages(chunks, page_dims)
|
|
468
|
+
|
|
469
|
+
# Build per-page markdown so that page-scoped rules
|
|
470
|
+
# (``_scope_to_page`` in rules_form.py) match against the correct page
|
|
471
|
+
# instead of falling back to full-document content. Works regardless of
|
|
472
|
+
# ``chunking_strategy`` (page/section/document) because page numbers
|
|
473
|
+
# come from each block's ``metadata.page.number``.
|
|
474
|
+
pages = _build_pages(chunks)
|
|
475
|
+
|
|
476
|
+
output = ParseOutput(
|
|
477
|
+
task_type="parse",
|
|
478
|
+
example_id=raw_result.request.example_id,
|
|
479
|
+
pipeline_name=raw_result.pipeline_name,
|
|
480
|
+
pages=pages,
|
|
481
|
+
layout_pages=layout_pages,
|
|
482
|
+
markdown=markdown,
|
|
483
|
+
job_id=str(job_id) if job_id else None,
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
return InferenceResult(
|
|
487
|
+
request=raw_result.request,
|
|
488
|
+
pipeline_name=raw_result.pipeline_name,
|
|
489
|
+
product_type=raw_result.product_type,
|
|
490
|
+
raw_output=raw_result.raw_output,
|
|
491
|
+
output=output,
|
|
492
|
+
started_at=raw_result.started_at,
|
|
493
|
+
completed_at=raw_result.completed_at,
|
|
494
|
+
latency_in_ms=raw_result.latency_in_ms,
|
|
495
|
+
)
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
def _get_pdf_page_dims(file_path: str) -> dict[int, tuple[float, float]]:
|
|
499
|
+
"""Read per-page dimensions (width, height) in PDF points from a PDF file.
|
|
500
|
+
|
|
501
|
+
Returns a dict mapping 1-indexed page number to (width, height).
|
|
502
|
+
Returns empty dict for non-PDF files or on error.
|
|
503
|
+
"""
|
|
504
|
+
try:
|
|
505
|
+
with open(file_path, "rb") as f:
|
|
506
|
+
if f.read(4) != b"%PDF":
|
|
507
|
+
return {}
|
|
508
|
+
reader = PdfReader(file_path)
|
|
509
|
+
dims: dict[int, tuple[float, float]] = {}
|
|
510
|
+
for i, page in enumerate(reader.pages):
|
|
511
|
+
box = page.mediabox
|
|
512
|
+
dims[i + 1] = (float(box.width), float(box.height))
|
|
513
|
+
return dims
|
|
514
|
+
except Exception:
|
|
515
|
+
return {}
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
def _build_pages(chunks: list[dict[str, Any]]) -> list[PageIR]:
|
|
519
|
+
"""Build per-page markdown from Extend chunks/blocks.
|
|
520
|
+
|
|
521
|
+
Extend exposes the page number on each block via ``metadata.page.number``
|
|
522
|
+
(1-indexed). We aggregate block-level ``content`` per page so that
|
|
523
|
+
page-scoped rules can match against the right page's text instead of
|
|
524
|
+
falling back to full-document content.
|
|
525
|
+
|
|
526
|
+
Block content is concatenated with ``\\n\\n``. This works for every
|
|
527
|
+
``chunking_strategy`` (``page``, ``section``, ``document``) because the
|
|
528
|
+
page assignment is per-block, not per-chunk.
|
|
529
|
+
|
|
530
|
+
Blocks with non-positive or unparseable page numbers are dropped from the
|
|
531
|
+
per-page output (``PageIR.page_index`` requires ``ge=0``). Their content
|
|
532
|
+
still reaches the document-level ``markdown`` via the chunk-level join in
|
|
533
|
+
:py:meth:`ExtendParseProvider.normalize`.
|
|
534
|
+
"""
|
|
535
|
+
pages_blocks: dict[int, list[str]] = defaultdict(list)
|
|
536
|
+
for chunk in chunks:
|
|
537
|
+
if not isinstance(chunk, dict):
|
|
538
|
+
continue
|
|
539
|
+
for block in chunk.get("blocks", []):
|
|
540
|
+
if not isinstance(block, dict):
|
|
541
|
+
continue
|
|
542
|
+
block_meta = block.get("metadata", {}) or {}
|
|
543
|
+
block_page_meta = block_meta.get("page", {}) or {}
|
|
544
|
+
# Use ``is not None`` rather than an ``or`` chain so that an
|
|
545
|
+
# explicit ``0`` from the API is treated as a real (invalid) page
|
|
546
|
+
# value and dropped below, not silently rewritten to ``1``.
|
|
547
|
+
raw_page = block_page_meta.get("number")
|
|
548
|
+
if raw_page is None:
|
|
549
|
+
raw_page = block.get("page")
|
|
550
|
+
if raw_page is None:
|
|
551
|
+
raw_page = block.get("pageNumber")
|
|
552
|
+
if raw_page is None:
|
|
553
|
+
raw_page = 1
|
|
554
|
+
try:
|
|
555
|
+
page_num = int(raw_page)
|
|
556
|
+
except (TypeError, ValueError):
|
|
557
|
+
page_num = 1
|
|
558
|
+
if page_num < 1:
|
|
559
|
+
# Defensive: 0-indexed or negative page values would produce
|
|
560
|
+
# PageIR.page_index < 0 and fail Pydantic validation.
|
|
561
|
+
continue
|
|
562
|
+
content = block.get("content", "") or block.get("text", "") or ""
|
|
563
|
+
if content:
|
|
564
|
+
pages_blocks[page_num].append(content)
|
|
565
|
+
|
|
566
|
+
pages: list[PageIR] = []
|
|
567
|
+
for page_num in sorted(pages_blocks.keys()):
|
|
568
|
+
page_md = "\n\n".join(pages_blocks[page_num])
|
|
569
|
+
# Extend pages are 1-indexed; PageIR.page_index is 0-indexed.
|
|
570
|
+
pages.append(PageIR(page_index=page_num - 1, markdown=page_md))
|
|
571
|
+
return pages
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
def _build_layout_pages(
|
|
575
|
+
chunks: list[dict[str, Any]],
|
|
576
|
+
page_dims: dict[int, tuple[float, float]] | dict[str, Any],
|
|
577
|
+
) -> list[ParseLayoutPageIR]:
|
|
578
|
+
"""Build layout_pages from Extend chunk blocks for layout cross-evaluation.
|
|
579
|
+
|
|
580
|
+
Iterates through chunks and their blocks, normalizes bboxes to [0,1]
|
|
581
|
+
using page dimensions, and groups by page number.
|
|
582
|
+
|
|
583
|
+
The Extend API returns bounding box coordinates in its own pixel coordinate
|
|
584
|
+
system (reported in each block's ``metadata.page.width/height``). We use
|
|
585
|
+
those pixel dimensions for normalization. The ``page_dims`` argument (PDF
|
|
586
|
+
point dimensions) is only used as a fallback when block-level metadata is
|
|
587
|
+
absent.
|
|
588
|
+
"""
|
|
589
|
+
# Normalize page_dims keys to int (JSON serialization may stringify them).
|
|
590
|
+
# These are PDF-point dims used only as a last-resort fallback.
|
|
591
|
+
norm_dims: dict[int, tuple[float, float]] = {}
|
|
592
|
+
for k, v in page_dims.items():
|
|
593
|
+
try:
|
|
594
|
+
page_key = int(k)
|
|
595
|
+
if isinstance(v, (list, tuple)) and len(v) == 2:
|
|
596
|
+
norm_dims[page_key] = (float(v[0]), float(v[1]))
|
|
597
|
+
except (TypeError, ValueError):
|
|
598
|
+
continue
|
|
599
|
+
|
|
600
|
+
pages_items: dict[int, list[LayoutItemIR]] = defaultdict(list)
|
|
601
|
+
pages_headers: dict[int, list[str]] = defaultdict(list)
|
|
602
|
+
pages_footers: dict[int, list[str]] = defaultdict(list)
|
|
603
|
+
|
|
604
|
+
for chunk in chunks:
|
|
605
|
+
if not isinstance(chunk, dict):
|
|
606
|
+
continue
|
|
607
|
+
|
|
608
|
+
blocks = chunk.get("blocks", [])
|
|
609
|
+
if not isinstance(blocks, list):
|
|
610
|
+
continue
|
|
611
|
+
|
|
612
|
+
for block in blocks:
|
|
613
|
+
if not isinstance(block, dict):
|
|
614
|
+
continue
|
|
615
|
+
|
|
616
|
+
block_type = block.get("type", "")
|
|
617
|
+
canonical_label = EXTEND_LABEL_MAP.get(block_type)
|
|
618
|
+
if canonical_label is None:
|
|
619
|
+
continue
|
|
620
|
+
|
|
621
|
+
bbox = block.get("boundingBox") or block.get("bounding_box") or {}
|
|
622
|
+
if not isinstance(bbox, dict):
|
|
623
|
+
continue
|
|
624
|
+
|
|
625
|
+
left = float(bbox.get("left", 0.0))
|
|
626
|
+
top = float(bbox.get("top", 0.0))
|
|
627
|
+
right = float(bbox.get("right", 0.0))
|
|
628
|
+
bottom = float(bbox.get("bottom", 0.0))
|
|
629
|
+
|
|
630
|
+
# Extract page number and pixel dimensions from block metadata
|
|
631
|
+
block_meta = block.get("metadata", {}) or {}
|
|
632
|
+
block_page_meta = block_meta.get("page", {}) or {}
|
|
633
|
+
page_num = block_page_meta.get("number") or block.get("page") or block.get("pageNumber") or 1
|
|
634
|
+
if isinstance(page_num, str):
|
|
635
|
+
try:
|
|
636
|
+
page_num = int(page_num)
|
|
637
|
+
except ValueError:
|
|
638
|
+
page_num = 1
|
|
639
|
+
|
|
640
|
+
# Use pixel dimensions from the API's block metadata (the coordinate
|
|
641
|
+
# system the bbox values are expressed in). Fall back to PDF-point
|
|
642
|
+
# dims only when the API does not report per-block page dimensions.
|
|
643
|
+
pixel_w = float(block_page_meta.get("width", 0))
|
|
644
|
+
pixel_h = float(block_page_meta.get("height", 0))
|
|
645
|
+
if pixel_w > 0 and pixel_h > 0:
|
|
646
|
+
pw, ph = pixel_w, pixel_h
|
|
647
|
+
else:
|
|
648
|
+
pw, ph = norm_dims.get(page_num, (0, 0))
|
|
649
|
+
|
|
650
|
+
if pw > 0 and ph > 0:
|
|
651
|
+
x_norm = left / pw
|
|
652
|
+
y_norm = top / ph
|
|
653
|
+
w_norm = (right - left) / pw
|
|
654
|
+
h_norm = (bottom - top) / ph
|
|
655
|
+
else:
|
|
656
|
+
# Fallback: store raw values (adapter will handle as-is)
|
|
657
|
+
x_norm = left
|
|
658
|
+
y_norm = top
|
|
659
|
+
w_norm = right - left
|
|
660
|
+
h_norm = bottom - top
|
|
661
|
+
|
|
662
|
+
confidence = float(block.get("confidence", 1.0))
|
|
663
|
+
|
|
664
|
+
seg = LayoutSegmentIR(
|
|
665
|
+
x=x_norm,
|
|
666
|
+
y=y_norm,
|
|
667
|
+
w=w_norm,
|
|
668
|
+
h=h_norm,
|
|
669
|
+
confidence=confidence,
|
|
670
|
+
label=canonical_label,
|
|
671
|
+
)
|
|
672
|
+
|
|
673
|
+
content = block.get("content", "") or block.get("text", "")
|
|
674
|
+
norm_label = canonical_label.strip().lower()
|
|
675
|
+
if norm_label == "table":
|
|
676
|
+
item_type = "table"
|
|
677
|
+
elif norm_label == "picture":
|
|
678
|
+
item_type = "image"
|
|
679
|
+
else:
|
|
680
|
+
item_type = "text"
|
|
681
|
+
|
|
682
|
+
pages_items[page_num].append(
|
|
683
|
+
LayoutItemIR(
|
|
684
|
+
type=item_type,
|
|
685
|
+
value=content,
|
|
686
|
+
bbox=seg,
|
|
687
|
+
layout_segments=[seg],
|
|
688
|
+
)
|
|
689
|
+
)
|
|
690
|
+
|
|
691
|
+
section_content = f"<page_number>{content}</page_number>" if block_type == "page_number" else content
|
|
692
|
+
if canonical_label == "Page-header" and content:
|
|
693
|
+
pages_headers[page_num].append(section_content)
|
|
694
|
+
elif canonical_label == "Page-footer" and content:
|
|
695
|
+
pages_footers[page_num].append(section_content)
|
|
696
|
+
|
|
697
|
+
layout_pages: list[ParseLayoutPageIR] = []
|
|
698
|
+
for page_num in sorted(pages_items.keys()):
|
|
699
|
+
layout_pages.append(
|
|
700
|
+
ParseLayoutPageIR(
|
|
701
|
+
page_number=page_num,
|
|
702
|
+
width=_VIRTUAL_PAGE_DIM,
|
|
703
|
+
height=_VIRTUAL_PAGE_DIM,
|
|
704
|
+
items=pages_items[page_num],
|
|
705
|
+
page_header_markdown="\n\n".join(pages_headers.get(page_num, [])),
|
|
706
|
+
page_footer_markdown="\n\n".join(pages_footers.get(page_num, [])),
|
|
707
|
+
)
|
|
708
|
+
)
|
|
709
|
+
|
|
710
|
+
return layout_pages
|