parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,452 @@
|
|
|
1
|
+
"""Provider for Landing AI PARSE."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from datetime import datetime
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from landingai_ade import LandingAIADE
|
|
9
|
+
|
|
10
|
+
from parse_bench.inference.providers.base import (
|
|
11
|
+
Provider,
|
|
12
|
+
ProviderConfigError,
|
|
13
|
+
ProviderPermanentError,
|
|
14
|
+
ProviderTransientError,
|
|
15
|
+
)
|
|
16
|
+
from parse_bench.inference.providers.registry import register_provider
|
|
17
|
+
from parse_bench.schemas.parse_output import (
|
|
18
|
+
LayoutItemIR,
|
|
19
|
+
LayoutSegmentIR,
|
|
20
|
+
PageIR,
|
|
21
|
+
ParseLayoutPageIR,
|
|
22
|
+
ParseOutput,
|
|
23
|
+
)
|
|
24
|
+
from parse_bench.schemas.pipeline import PipelineSpec
|
|
25
|
+
from parse_bench.schemas.pipeline_io import (
|
|
26
|
+
InferenceRequest,
|
|
27
|
+
InferenceResult,
|
|
28
|
+
RawInferenceResult,
|
|
29
|
+
)
|
|
30
|
+
from parse_bench.schemas.product import ProductType
|
|
31
|
+
|
|
32
|
+
# LandingAI chunk type -> Canonical17 label string
|
|
33
|
+
LANDINGAI_LABEL_MAP: dict[str, str] = {
|
|
34
|
+
"text": "Text",
|
|
35
|
+
"table": "Table",
|
|
36
|
+
"figure": "Picture",
|
|
37
|
+
"marginalia": "Page-header", # headers/footers/page numbers consolidated
|
|
38
|
+
"logo": "Picture",
|
|
39
|
+
"card": "Key-Value Region",
|
|
40
|
+
# "attestation" and "scan_code" have no canonical equivalent — skipped
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
# Virtual page dimensions for normalized coordinate conversion.
|
|
44
|
+
# LandingAI bbox is already [0,1], so these cancel out during evaluation.
|
|
45
|
+
_VIRTUAL_PAGE_DIM = 1000.0
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@register_provider("landingai")
|
|
49
|
+
class LandingAIParseProvider(Provider):
|
|
50
|
+
"""
|
|
51
|
+
Provider for Landing AI PARSE.
|
|
52
|
+
|
|
53
|
+
This provider uses the Landing AI ADE API for parsing tasks.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
CREDIT_RATE_USD = 0.01 # $0.01 per credit (Explore plan)
|
|
57
|
+
|
|
58
|
+
def __init__(
|
|
59
|
+
self,
|
|
60
|
+
provider_name: str,
|
|
61
|
+
base_config: dict[str, Any] | None = None,
|
|
62
|
+
):
|
|
63
|
+
"""
|
|
64
|
+
Initialize the provider.
|
|
65
|
+
|
|
66
|
+
:param provider_name: Name of the provider
|
|
67
|
+
:param base_config: Optional configuration with:
|
|
68
|
+
- `api_key`: Landing AI API key (defaults to LANDING_AI_API_KEY env var)
|
|
69
|
+
- `model`: Model to use (default: "dpt-2-latest")
|
|
70
|
+
- Any other parse parameters from Landing AI API
|
|
71
|
+
"""
|
|
72
|
+
super().__init__(provider_name, base_config)
|
|
73
|
+
|
|
74
|
+
# Get API key
|
|
75
|
+
self._api_key = self.base_config.get("api_key") or os.getenv("LANDING_AI_API_KEY")
|
|
76
|
+
if not self._api_key:
|
|
77
|
+
raise ProviderConfigError(
|
|
78
|
+
"Landing AI API key is required. "
|
|
79
|
+
"Set LANDING_AI_API_KEY environment variable or pass api_key in base_config."
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
# Set VISION_AGENT_API_KEY for the SDK (it expects this env var)
|
|
83
|
+
# Only set if not already set to avoid overriding existing values
|
|
84
|
+
if not os.getenv("VISION_AGENT_API_KEY"):
|
|
85
|
+
os.environ["VISION_AGENT_API_KEY"] = self._api_key
|
|
86
|
+
|
|
87
|
+
# Get configuration with defaults
|
|
88
|
+
self._model = self.base_config.get("model", "dpt-2-latest")
|
|
89
|
+
|
|
90
|
+
# Initialize client
|
|
91
|
+
self._client = LandingAIADE()
|
|
92
|
+
|
|
93
|
+
def _parse_document(self, document_path: Path) -> dict[str, Any]:
|
|
94
|
+
"""
|
|
95
|
+
Parse a document using Landing AI API.
|
|
96
|
+
|
|
97
|
+
:param document_path: Path to the document file
|
|
98
|
+
:return: Raw API response as dictionary
|
|
99
|
+
:raises ProviderError: For any API errors
|
|
100
|
+
"""
|
|
101
|
+
try:
|
|
102
|
+
# Parse the document
|
|
103
|
+
response = self._client.parse(
|
|
104
|
+
document=document_path,
|
|
105
|
+
model=self._model,
|
|
106
|
+
**{k: v for k, v in self.base_config.items() if k not in ["api_key", "model"]},
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
# Convert response to dictionary format
|
|
110
|
+
# The response has markdown, chunks, and grounding attributes
|
|
111
|
+
result: dict[str, Any] = {
|
|
112
|
+
"markdown": response.markdown if hasattr(response, "markdown") else "",
|
|
113
|
+
"chunks": [],
|
|
114
|
+
"splits": [],
|
|
115
|
+
"grounding": {},
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
# Extract chunks if available
|
|
119
|
+
if hasattr(response, "chunks"):
|
|
120
|
+
chunks = response.chunks
|
|
121
|
+
if chunks is not None:
|
|
122
|
+
# Convert chunks to serializable format
|
|
123
|
+
for chunk in chunks:
|
|
124
|
+
chunk_data: dict[str, Any] = {}
|
|
125
|
+
if hasattr(chunk, "id"):
|
|
126
|
+
chunk_data["id"] = chunk.id
|
|
127
|
+
if hasattr(chunk, "type"):
|
|
128
|
+
chunk_data["type"] = chunk.type
|
|
129
|
+
if hasattr(chunk, "markdown"):
|
|
130
|
+
chunk_data["markdown"] = chunk.markdown
|
|
131
|
+
if hasattr(chunk, "grounding") and chunk.grounding is not None:
|
|
132
|
+
# ChunkGrounding is a Pydantic model - convert to dict
|
|
133
|
+
chunk_data["grounding"] = chunk.grounding.model_dump()
|
|
134
|
+
result["chunks"].append(chunk_data)
|
|
135
|
+
|
|
136
|
+
# Extract splits if available (populated when split="page" is used)
|
|
137
|
+
if hasattr(response, "splits") and response.splits is not None:
|
|
138
|
+
for split in response.splits:
|
|
139
|
+
split_data: dict[str, Any] = {}
|
|
140
|
+
if hasattr(split, "markdown"):
|
|
141
|
+
split_data["markdown"] = split.markdown
|
|
142
|
+
if hasattr(split, "pages"):
|
|
143
|
+
split_data["pages"] = split.pages
|
|
144
|
+
if hasattr(split, "chunks"):
|
|
145
|
+
split_data["chunks"] = split.chunks
|
|
146
|
+
if hasattr(split, "class_"):
|
|
147
|
+
split_data["class"] = split.class_
|
|
148
|
+
if hasattr(split, "identifier"):
|
|
149
|
+
split_data["identifier"] = split.identifier
|
|
150
|
+
result["splits"].append(split_data)
|
|
151
|
+
|
|
152
|
+
# Extract grounding if available
|
|
153
|
+
# response.grounding is Dict[str, Grounding] where Grounding is a Pydantic model
|
|
154
|
+
if hasattr(response, "grounding") and response.grounding is not None:
|
|
155
|
+
result["grounding"] = {k: v.model_dump() for k, v in response.grounding.items()}
|
|
156
|
+
|
|
157
|
+
# Extract cost from metadata
|
|
158
|
+
if hasattr(response, "metadata") and response.metadata is not None:
|
|
159
|
+
meta = response.metadata
|
|
160
|
+
credits = getattr(meta, "credit_usage", None)
|
|
161
|
+
num_pages = getattr(meta, "page_count", None)
|
|
162
|
+
if credits is not None and credits > 0:
|
|
163
|
+
cost_usd = credits * self.CREDIT_RATE_USD
|
|
164
|
+
result["credits_used"] = credits
|
|
165
|
+
result["cost_usd"] = cost_usd
|
|
166
|
+
if num_pages and num_pages > 0:
|
|
167
|
+
result["num_pages"] = num_pages
|
|
168
|
+
result["cost_per_page_usd"] = cost_usd / num_pages
|
|
169
|
+
|
|
170
|
+
return result
|
|
171
|
+
|
|
172
|
+
except Exception as e:
|
|
173
|
+
# Check if it's a transient error (network, timeout, etc.)
|
|
174
|
+
error_str = str(e).lower()
|
|
175
|
+
transient_keywords = ["timeout", "network", "connection", "503", "502", "504"]
|
|
176
|
+
if any(keyword in error_str for keyword in transient_keywords):
|
|
177
|
+
raise ProviderTransientError(f"Transient error during parsing: {e}") from e
|
|
178
|
+
else:
|
|
179
|
+
raise ProviderPermanentError(f"Error during parsing: {e}") from e
|
|
180
|
+
|
|
181
|
+
def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
|
|
182
|
+
"""
|
|
183
|
+
Run inference and return raw results.
|
|
184
|
+
|
|
185
|
+
:param pipeline: Pipeline specification
|
|
186
|
+
:param request: Inference request
|
|
187
|
+
:return: Raw inference result
|
|
188
|
+
:raises ProviderError: For any provider-related failures
|
|
189
|
+
"""
|
|
190
|
+
if request.product_type != ProductType.PARSE:
|
|
191
|
+
raise ProviderPermanentError(
|
|
192
|
+
f"LandingAIParseProvider only supports PARSE product type, got {request.product_type}"
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
started_at = datetime.now()
|
|
196
|
+
|
|
197
|
+
# Check if file exists
|
|
198
|
+
file_path = Path(request.source_file_path)
|
|
199
|
+
if not file_path.exists():
|
|
200
|
+
raise ProviderPermanentError(f"File not found: {file_path}")
|
|
201
|
+
|
|
202
|
+
try:
|
|
203
|
+
# Run parsing
|
|
204
|
+
raw_output = self._parse_document(file_path)
|
|
205
|
+
|
|
206
|
+
completed_at = datetime.now()
|
|
207
|
+
latency_ms = int((completed_at - started_at).total_seconds() * 1000)
|
|
208
|
+
|
|
209
|
+
return RawInferenceResult(
|
|
210
|
+
request=request,
|
|
211
|
+
pipeline=pipeline,
|
|
212
|
+
pipeline_name=pipeline.pipeline_name,
|
|
213
|
+
product_type=request.product_type,
|
|
214
|
+
raw_output=raw_output,
|
|
215
|
+
started_at=started_at,
|
|
216
|
+
completed_at=completed_at,
|
|
217
|
+
latency_in_ms=latency_ms,
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
except ProviderPermanentError:
|
|
221
|
+
# Re-raise provider errors as-is
|
|
222
|
+
raise
|
|
223
|
+
except ProviderTransientError:
|
|
224
|
+
# Re-raise provider errors as-is
|
|
225
|
+
raise
|
|
226
|
+
except Exception as e:
|
|
227
|
+
# Wrap unexpected errors
|
|
228
|
+
raise ProviderPermanentError(f"Unexpected error during inference: {e}") from e
|
|
229
|
+
|
|
230
|
+
def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
|
|
231
|
+
"""
|
|
232
|
+
Normalize raw inference result to produce ParseOutput.
|
|
233
|
+
|
|
234
|
+
:param raw_result: Raw inference result from run_inference()
|
|
235
|
+
:return: Inference result with both raw and normalized outputs
|
|
236
|
+
:raises ProviderError: For any normalization failures
|
|
237
|
+
"""
|
|
238
|
+
if raw_result.product_type != ProductType.PARSE:
|
|
239
|
+
raise ProviderPermanentError(
|
|
240
|
+
f"LandingAIParseProvider only supports PARSE product type, got {raw_result.product_type}"
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
# Extract markdown from raw output and promote table headers.
|
|
244
|
+
# Landing AI emits all cells as <td>; downstream eval relies on <th>.
|
|
245
|
+
markdown = _promote_first_row_to_header(raw_result.raw_output.get("markdown", ""))
|
|
246
|
+
|
|
247
|
+
pages: list[PageIR] = []
|
|
248
|
+
|
|
249
|
+
# Strategy 1: Use splits data if available (from split="page")
|
|
250
|
+
splits = raw_result.raw_output.get("splits", [])
|
|
251
|
+
if splits:
|
|
252
|
+
for split in splits:
|
|
253
|
+
if isinstance(split, dict) and "markdown" in split and "pages" in split:
|
|
254
|
+
split_pages = split["pages"]
|
|
255
|
+
split_md = _promote_first_row_to_header(split["markdown"])
|
|
256
|
+
# Each split may cover one or more pages; use the first page number
|
|
257
|
+
page_num = split_pages[0] if split_pages else 0
|
|
258
|
+
pages.append(PageIR(page_index=page_num, markdown=split_md))
|
|
259
|
+
|
|
260
|
+
# Strategy 2: Fall back to chunk grounding for page splitting
|
|
261
|
+
if not pages:
|
|
262
|
+
chunks = raw_result.raw_output.get("chunks", [])
|
|
263
|
+
grounding = raw_result.raw_output.get("grounding", {})
|
|
264
|
+
|
|
265
|
+
page_content: dict[int, list[str]] = {}
|
|
266
|
+
|
|
267
|
+
if isinstance(grounding, dict):
|
|
268
|
+
for gid, gdata in grounding.items():
|
|
269
|
+
if isinstance(gdata, dict) and "page" in gdata:
|
|
270
|
+
page_num = gdata["page"]
|
|
271
|
+
if page_num not in page_content:
|
|
272
|
+
page_content[page_num] = []
|
|
273
|
+
for chunk in chunks:
|
|
274
|
+
if isinstance(chunk, dict) and chunk.get("id") == gid:
|
|
275
|
+
if "markdown" in chunk:
|
|
276
|
+
page_content[page_num].append(_promote_first_row_to_header(chunk["markdown"]))
|
|
277
|
+
|
|
278
|
+
if page_content:
|
|
279
|
+
for page_num in sorted(page_content.keys()):
|
|
280
|
+
page_text = "\n".join(page_content[page_num])
|
|
281
|
+
pages.append(PageIR(page_index=page_num, markdown=page_text))
|
|
282
|
+
|
|
283
|
+
# Strategy 3: Fallback — single page with all markdown
|
|
284
|
+
if not pages:
|
|
285
|
+
pages.append(PageIR(page_index=0, markdown=markdown))
|
|
286
|
+
|
|
287
|
+
# Build layout_pages from chunk grounding for layout cross-evaluation
|
|
288
|
+
chunks = raw_result.raw_output.get("chunks", [])
|
|
289
|
+
layout_pages = _build_layout_pages(chunks)
|
|
290
|
+
|
|
291
|
+
output = ParseOutput(
|
|
292
|
+
task_type="parse",
|
|
293
|
+
example_id=raw_result.request.example_id,
|
|
294
|
+
pipeline_name=raw_result.pipeline_name,
|
|
295
|
+
pages=pages,
|
|
296
|
+
layout_pages=layout_pages,
|
|
297
|
+
markdown=markdown,
|
|
298
|
+
job_id=None, # Landing AI parse doesn't return job_id
|
|
299
|
+
)
|
|
300
|
+
|
|
301
|
+
return InferenceResult(
|
|
302
|
+
request=raw_result.request,
|
|
303
|
+
pipeline_name=raw_result.pipeline_name,
|
|
304
|
+
product_type=raw_result.product_type,
|
|
305
|
+
raw_output=raw_result.raw_output,
|
|
306
|
+
output=output,
|
|
307
|
+
started_at=raw_result.started_at,
|
|
308
|
+
completed_at=raw_result.completed_at,
|
|
309
|
+
latency_in_ms=raw_result.latency_in_ms,
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _build_layout_pages(chunks: list[dict[str, Any]]) -> list[ParseLayoutPageIR]:
|
|
314
|
+
"""Build layout_pages from LandingAI chunk grounding for layout cross-evaluation.
|
|
315
|
+
|
|
316
|
+
Groups chunks by page number and converts each chunk's normalized [0,1]
|
|
317
|
+
bounding box into a LayoutSegmentIR with canonical label mapping.
|
|
318
|
+
LandingAI grounding pages are 0-indexed; we convert to 1-indexed.
|
|
319
|
+
"""
|
|
320
|
+
from collections import defaultdict
|
|
321
|
+
|
|
322
|
+
pages_chunks: dict[int, list[dict[str, Any]]] = defaultdict(list)
|
|
323
|
+
for chunk in chunks:
|
|
324
|
+
grounding = chunk.get("grounding")
|
|
325
|
+
if not isinstance(grounding, dict):
|
|
326
|
+
continue
|
|
327
|
+
# LandingAI pages are 0-indexed
|
|
328
|
+
page_num = grounding.get("page", 0)
|
|
329
|
+
pages_chunks[page_num].append(chunk)
|
|
330
|
+
|
|
331
|
+
layout_pages: list[ParseLayoutPageIR] = []
|
|
332
|
+
for page_num in sorted(pages_chunks.keys()):
|
|
333
|
+
page_chunks = pages_chunks[page_num]
|
|
334
|
+
items: list[LayoutItemIR] = []
|
|
335
|
+
|
|
336
|
+
for chunk in page_chunks:
|
|
337
|
+
chunk_type = chunk.get("type", "")
|
|
338
|
+
canonical_label = LANDINGAI_LABEL_MAP.get(chunk_type)
|
|
339
|
+
if canonical_label is None:
|
|
340
|
+
continue # Skip unmapped types (e.g., attestation, scan_code)
|
|
341
|
+
|
|
342
|
+
grounding = chunk.get("grounding", {})
|
|
343
|
+
box = grounding.get("box", {})
|
|
344
|
+
left = float(box.get("left", 0.0))
|
|
345
|
+
top = float(box.get("top", 0.0))
|
|
346
|
+
right = float(box.get("right", 0.0))
|
|
347
|
+
bottom = float(box.get("bottom", 0.0))
|
|
348
|
+
width = right - left
|
|
349
|
+
height = bottom - top
|
|
350
|
+
|
|
351
|
+
# Parse confidence (DPT-2 provides it)
|
|
352
|
+
conf_raw = grounding.get("confidence")
|
|
353
|
+
try:
|
|
354
|
+
confidence = float(conf_raw) if conf_raw is not None else 1.0
|
|
355
|
+
except (TypeError, ValueError):
|
|
356
|
+
confidence = 1.0
|
|
357
|
+
|
|
358
|
+
seg = LayoutSegmentIR(
|
|
359
|
+
x=left,
|
|
360
|
+
y=top,
|
|
361
|
+
w=width,
|
|
362
|
+
h=height,
|
|
363
|
+
confidence=confidence,
|
|
364
|
+
label=canonical_label,
|
|
365
|
+
)
|
|
366
|
+
|
|
367
|
+
content = chunk.get("markdown", "")
|
|
368
|
+
norm_label = canonical_label.strip().lower()
|
|
369
|
+
if norm_label == "table":
|
|
370
|
+
item_type = "table"
|
|
371
|
+
elif norm_label == "picture":
|
|
372
|
+
item_type = "image"
|
|
373
|
+
else:
|
|
374
|
+
item_type = "text"
|
|
375
|
+
|
|
376
|
+
items.append(
|
|
377
|
+
LayoutItemIR(
|
|
378
|
+
type=item_type,
|
|
379
|
+
value=content,
|
|
380
|
+
bbox=seg,
|
|
381
|
+
layout_segments=[seg],
|
|
382
|
+
)
|
|
383
|
+
)
|
|
384
|
+
|
|
385
|
+
# Convert 0-indexed page to 1-indexed for ParseLayoutPageIR
|
|
386
|
+
layout_pages.append(
|
|
387
|
+
ParseLayoutPageIR(
|
|
388
|
+
page_number=page_num + 1,
|
|
389
|
+
width=_VIRTUAL_PAGE_DIM,
|
|
390
|
+
height=_VIRTUAL_PAGE_DIM,
|
|
391
|
+
items=items,
|
|
392
|
+
)
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
return layout_pages
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _promote_first_row_to_header(html: str) -> str:
|
|
399
|
+
"""Rewrite HTML tables so the first row uses ``<th>`` inside ``<thead>``.
|
|
400
|
+
|
|
401
|
+
Landing AI emits all table cells as ``<td>`` with no
|
|
402
|
+
``<th>``/``<thead>``/``<tbody>``. This promotes the first ``<tr>`` of each
|
|
403
|
+
``<table>`` to be a header row so that downstream evaluation code (which
|
|
404
|
+
keys on ``<th>``) can identify column headers.
|
|
405
|
+
|
|
406
|
+
Only tables that contain zero ``<th>`` elements are modified — tables that
|
|
407
|
+
already have headers are left untouched.
|
|
408
|
+
"""
|
|
409
|
+
from bs4 import BeautifulSoup
|
|
410
|
+
|
|
411
|
+
if "<table" not in html:
|
|
412
|
+
return html
|
|
413
|
+
|
|
414
|
+
soup = BeautifulSoup(html, "lxml")
|
|
415
|
+
modified = False
|
|
416
|
+
|
|
417
|
+
for table in soup.find_all("table"):
|
|
418
|
+
# Skip tables that already have <th> elements
|
|
419
|
+
if table.find("th"):
|
|
420
|
+
continue
|
|
421
|
+
|
|
422
|
+
first_tr = table.find("tr")
|
|
423
|
+
if first_tr is None:
|
|
424
|
+
continue
|
|
425
|
+
|
|
426
|
+
# Promote <td> -> <th> in the first row
|
|
427
|
+
for td in first_tr.find_all("td"):
|
|
428
|
+
td.name = "th"
|
|
429
|
+
|
|
430
|
+
# Wrap first row in <thead>, remaining rows in <tbody>
|
|
431
|
+
thead = soup.new_tag("thead")
|
|
432
|
+
first_tr.extract()
|
|
433
|
+
thead.append(first_tr)
|
|
434
|
+
|
|
435
|
+
tbody = soup.new_tag("tbody")
|
|
436
|
+
for tr in table.find_all("tr"):
|
|
437
|
+
tr.extract()
|
|
438
|
+
tbody.append(tr)
|
|
439
|
+
|
|
440
|
+
table.clear()
|
|
441
|
+
table.append(thead)
|
|
442
|
+
if tbody.find("tr"):
|
|
443
|
+
table.append(tbody)
|
|
444
|
+
|
|
445
|
+
modified = True
|
|
446
|
+
|
|
447
|
+
if not modified:
|
|
448
|
+
return html
|
|
449
|
+
|
|
450
|
+
# Return just the body content to avoid <html><body> wrapper
|
|
451
|
+
body = soup.find("body")
|
|
452
|
+
return body.decode_contents() if body else str(soup)
|