parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,382 @@
|
|
|
1
|
+
"""Comparison tool for evaluating two different pipeline results."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from parse_bench.evaluation.layout_adapters import create_layout_adapter_for_result
|
|
8
|
+
from parse_bench.evaluation.layout_label_mappers import project_layout_predictions
|
|
9
|
+
from parse_bench.schemas.evaluation import EvaluationResult, EvaluationSummary
|
|
10
|
+
from parse_bench.schemas.pipeline_io import InferenceResult
|
|
11
|
+
from parse_bench.test_cases import load_test_cases
|
|
12
|
+
from parse_bench.test_cases.schema import LayoutDetectionTestCase
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class PipelineComparison:
|
|
16
|
+
"""Compare results from two different pipelines."""
|
|
17
|
+
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
pipeline_a_dir: Path,
|
|
21
|
+
pipeline_b_dir: Path,
|
|
22
|
+
test_cases_dir: Path | None = None,
|
|
23
|
+
):
|
|
24
|
+
"""
|
|
25
|
+
Initialize comparison between two pipelines.
|
|
26
|
+
|
|
27
|
+
:param pipeline_a_dir: Directory containing pipeline A evaluation results
|
|
28
|
+
:param pipeline_b_dir: Directory containing pipeline B evaluation results
|
|
29
|
+
:param test_cases_dir: Optional directory containing test cases
|
|
30
|
+
"""
|
|
31
|
+
self.pipeline_a_dir = Path(pipeline_a_dir)
|
|
32
|
+
self.pipeline_b_dir = Path(pipeline_b_dir)
|
|
33
|
+
self.test_cases_dir = Path(test_cases_dir) if test_cases_dir else None
|
|
34
|
+
|
|
35
|
+
# Ordered metric-name candidates per product type. The first name present
|
|
36
|
+
# in an evaluation result wins. Parse carries a fallback chain because
|
|
37
|
+
# layout-only parse runs (test cases with only ``LayoutTestRule`` entries)
|
|
38
|
+
# emit table-only metrics such as ``grits_trm_composite`` or layout-only
|
|
39
|
+
# metrics such as ``mAP@[.50:.95]`` instead of ``rule_pass_rate``.
|
|
40
|
+
#
|
|
41
|
+
# MUST stay in sync with
|
|
42
|
+
# ``comparison_core.py::COMPARISON_METRIC_CANDIDATES`` — enforced by
|
|
43
|
+
# ``tests/.../test_comparison_consistency.py``. ``rule_pass_rate`` is the
|
|
44
|
+
# canonical parse metric (pass/fail rule semantics from
|
|
45
|
+
# ``ParseEvaluator``); ``grits_trm_composite`` is the primary table-only
|
|
46
|
+
# parse metric. ``normalized_text_score`` is a secondary text-similarity
|
|
47
|
+
# signal and intentionally absent here.
|
|
48
|
+
METRIC_CANDIDATES: dict[str, tuple[str, ...]] = {
|
|
49
|
+
"extract": ("accuracy",),
|
|
50
|
+
"parse": ("rule_pass_rate", "grits_trm_composite", "mAP@[.50:.95]"),
|
|
51
|
+
"layout_detection": ("mAP@[.50:.95]",),
|
|
52
|
+
}
|
|
53
|
+
DEFAULT_METRIC: str = "accuracy"
|
|
54
|
+
|
|
55
|
+
def _detect_product_type(self, summary: EvaluationSummary) -> str:
|
|
56
|
+
"""Detect product type from evaluation results."""
|
|
57
|
+
if summary.per_example_results:
|
|
58
|
+
return summary.per_example_results[0].product_type
|
|
59
|
+
return "parse" # default fallback
|
|
60
|
+
|
|
61
|
+
def _get_directory_suffix(self, pipeline_dir: Path) -> str:
|
|
62
|
+
"""
|
|
63
|
+
Extract a distinguishing suffix from the pipeline directory path.
|
|
64
|
+
|
|
65
|
+
Looks for run IDs, dates, or other identifying info in parent directories.
|
|
66
|
+
Example paths:
|
|
67
|
+
/output/financial_tables_run-21391181794/llamaparse_agentic -> "run-21391181794"
|
|
68
|
+
/output/2025-01-27/llamaparse_agentic -> "2025-01-27"
|
|
69
|
+
/output/experiment_v2/llamaparse_agentic -> "experiment_v2"
|
|
70
|
+
"""
|
|
71
|
+
import re
|
|
72
|
+
|
|
73
|
+
# Get the parent directory name (the run/experiment folder)
|
|
74
|
+
parent_name = pipeline_dir.parent.name
|
|
75
|
+
|
|
76
|
+
# Try to extract a run ID pattern (e.g., run-21391181794). Matrix-leg
|
|
77
|
+
# dirs append a dataset suffix (run-<id>-<dataset-slug>) — keep it, or
|
|
78
|
+
# two legs of the same parent run would get identical labels.
|
|
79
|
+
run_id_match = re.search(r"run-(\d+(?:-[A-Za-z0-9._-]+)?)", parent_name)
|
|
80
|
+
if run_id_match:
|
|
81
|
+
return f"run-{run_id_match.group(1)}"
|
|
82
|
+
|
|
83
|
+
# Try to extract a date pattern (e.g., 2025-01-27)
|
|
84
|
+
date_match = re.search(r"(\d{4}-\d{2}-\d{2})", parent_name)
|
|
85
|
+
if date_match:
|
|
86
|
+
return date_match.group(1)
|
|
87
|
+
|
|
88
|
+
# Fall back to the parent directory name
|
|
89
|
+
if parent_name and parent_name != "output":
|
|
90
|
+
return parent_name
|
|
91
|
+
|
|
92
|
+
# Last resort: use the full parent path's last 2 components
|
|
93
|
+
parts = pipeline_dir.parts
|
|
94
|
+
if len(parts) >= 2:
|
|
95
|
+
return "/".join(parts[-2:])
|
|
96
|
+
|
|
97
|
+
return str(pipeline_dir)
|
|
98
|
+
|
|
99
|
+
def _load_evaluation_summary(self, output_dir: Path) -> EvaluationSummary | None:
|
|
100
|
+
"""Load evaluation summary from a directory."""
|
|
101
|
+
eval_report_path = output_dir / "_evaluation_report.json"
|
|
102
|
+
if not eval_report_path.exists():
|
|
103
|
+
return None
|
|
104
|
+
try:
|
|
105
|
+
with open(eval_report_path) as f:
|
|
106
|
+
data = json.load(f)
|
|
107
|
+
return EvaluationSummary.model_validate(data)
|
|
108
|
+
except Exception:
|
|
109
|
+
return None
|
|
110
|
+
|
|
111
|
+
def _load_inference_result(self, output_dir: Path, test_id: str) -> InferenceResult | None:
|
|
112
|
+
"""Load inference result for a specific test_id."""
|
|
113
|
+
# Try to find the result file
|
|
114
|
+
# Result files are stored as: <test_id>/<test_id>.result.json
|
|
115
|
+
# But test_id might have slashes (group/filename)
|
|
116
|
+
parts = test_id.split("/")
|
|
117
|
+
if len(parts) == 2:
|
|
118
|
+
group, filename = parts
|
|
119
|
+
result_path = output_dir / group / f"{filename}.result.json"
|
|
120
|
+
else:
|
|
121
|
+
# Fallback: search for the file
|
|
122
|
+
result_path = output_dir / f"{test_id}.result.json"
|
|
123
|
+
|
|
124
|
+
if not result_path.exists():
|
|
125
|
+
# Try recursive search
|
|
126
|
+
for result_file in output_dir.rglob(f"*{test_id}*.result.json"):
|
|
127
|
+
result_path = result_file
|
|
128
|
+
break
|
|
129
|
+
else:
|
|
130
|
+
return None
|
|
131
|
+
|
|
132
|
+
try:
|
|
133
|
+
with open(result_path) as f:
|
|
134
|
+
data = json.load(f)
|
|
135
|
+
return InferenceResult.model_validate(data)
|
|
136
|
+
except Exception:
|
|
137
|
+
return None
|
|
138
|
+
|
|
139
|
+
def _get_accuracy(self, eval_result: EvaluationResult) -> float | None:
|
|
140
|
+
"""Extract accuracy metric from evaluation result (backward compatibility)."""
|
|
141
|
+
for metric in eval_result.metrics:
|
|
142
|
+
if metric.metric_name == "accuracy":
|
|
143
|
+
return metric.value
|
|
144
|
+
return None
|
|
145
|
+
|
|
146
|
+
def _candidate_metric_names(self, product_type: str) -> tuple[str, ...]:
|
|
147
|
+
return self.METRIC_CANDIDATES.get(product_type, (self.DEFAULT_METRIC,))
|
|
148
|
+
|
|
149
|
+
def _get_comparison_metric(self, eval_result: EvaluationResult, product_type: str) -> float | None:
|
|
150
|
+
"""Return the first candidate metric emitted by ``eval_result``.
|
|
151
|
+
|
|
152
|
+
Parse emits ``rule_pass_rate`` *or* ``mAP@[.50:.95]`` depending on
|
|
153
|
+
whether the test case carries parse rules or only layout rules.
|
|
154
|
+
"""
|
|
155
|
+
available = {m.metric_name: m.value for m in eval_result.metrics}
|
|
156
|
+
for name in self._candidate_metric_names(product_type):
|
|
157
|
+
if name in available:
|
|
158
|
+
return available[name]
|
|
159
|
+
return None
|
|
160
|
+
|
|
161
|
+
def _resolve_comparison_metric_name(self, summary: EvaluationSummary, product_type: str) -> str:
|
|
162
|
+
"""Pick the metric-name label used in stats/output headers.
|
|
163
|
+
|
|
164
|
+
Scans per-example results and returns the first candidate that at
|
|
165
|
+
least one example emitted; falls back to the first candidate so
|
|
166
|
+
empty summaries still get a sane label.
|
|
167
|
+
"""
|
|
168
|
+
candidates = self._candidate_metric_names(product_type)
|
|
169
|
+
emitted = {m.metric_name for r in summary.per_example_results for m in r.metrics}
|
|
170
|
+
for name in candidates:
|
|
171
|
+
if name in emitted:
|
|
172
|
+
return name
|
|
173
|
+
return candidates[0]
|
|
174
|
+
|
|
175
|
+
def _get_predictions(self, inference: InferenceResult | None) -> list[dict] | None:
|
|
176
|
+
"""Extract predictions as list of dicts for JSON serialization."""
|
|
177
|
+
if not inference or not inference.output:
|
|
178
|
+
return None
|
|
179
|
+
try:
|
|
180
|
+
adapter = create_layout_adapter_for_result(inference)
|
|
181
|
+
layout_output = adapter.to_layout_output(inference)
|
|
182
|
+
projected = project_layout_predictions(
|
|
183
|
+
inference,
|
|
184
|
+
layout_output,
|
|
185
|
+
evaluation_view="core",
|
|
186
|
+
target_ontology="canonical",
|
|
187
|
+
)
|
|
188
|
+
return [
|
|
189
|
+
{
|
|
190
|
+
"bbox": prediction["bbox"],
|
|
191
|
+
"class": prediction["class_name"],
|
|
192
|
+
"score": prediction["score"],
|
|
193
|
+
}
|
|
194
|
+
for prediction in projected
|
|
195
|
+
]
|
|
196
|
+
except Exception:
|
|
197
|
+
return None
|
|
198
|
+
|
|
199
|
+
def _get_gt_annotations(self, test_case: Any) -> list[dict] | None:
|
|
200
|
+
"""Extract GT annotations from test case."""
|
|
201
|
+
if not test_case or not isinstance(test_case, LayoutDetectionTestCase):
|
|
202
|
+
return None
|
|
203
|
+
annotations = test_case.get_layout_annotations()
|
|
204
|
+
if not annotations:
|
|
205
|
+
return None
|
|
206
|
+
return [{"bbox": ann.bbox, "class": ann.canonical_class} for ann in annotations]
|
|
207
|
+
|
|
208
|
+
def compare(self) -> dict[str, Any]:
|
|
209
|
+
"""
|
|
210
|
+
Compare results from both pipelines.
|
|
211
|
+
|
|
212
|
+
Returns a dictionary with comparison data including:
|
|
213
|
+
- matched_results: List of comparisons
|
|
214
|
+
- pipeline_a_only: Results only in pipeline A
|
|
215
|
+
- pipeline_b_only: Results only in pipeline B
|
|
216
|
+
- stats: Summary statistics
|
|
217
|
+
- product_type: The detected product type
|
|
218
|
+
- comparison_metric: The metric used for comparison
|
|
219
|
+
"""
|
|
220
|
+
# Load evaluation summaries
|
|
221
|
+
summary_a = self._load_evaluation_summary(self.pipeline_a_dir)
|
|
222
|
+
summary_b = self._load_evaluation_summary(self.pipeline_b_dir)
|
|
223
|
+
|
|
224
|
+
if not summary_a or not summary_b:
|
|
225
|
+
raise ValueError(
|
|
226
|
+
"Could not load evaluation summaries. "
|
|
227
|
+
"Make sure both directories contain _evaluation_report.json files. "
|
|
228
|
+
"Run evaluation first using: run_evaluation"
|
|
229
|
+
)
|
|
230
|
+
|
|
231
|
+
# Detect product type
|
|
232
|
+
product_type = self._detect_product_type(summary_a)
|
|
233
|
+
comparison_metric = self._resolve_comparison_metric_name(summary_a, product_type)
|
|
234
|
+
|
|
235
|
+
# Create mapping of test_id -> EvaluationResult
|
|
236
|
+
results_a = {r.test_id: r for r in summary_a.per_example_results}
|
|
237
|
+
results_b = {r.test_id: r for r in summary_b.per_example_results}
|
|
238
|
+
|
|
239
|
+
# Load test cases if available
|
|
240
|
+
test_cases: dict[str, Any] = {}
|
|
241
|
+
if self.test_cases_dir and self.test_cases_dir.exists():
|
|
242
|
+
test_cases_list = load_test_cases(self.test_cases_dir)
|
|
243
|
+
test_cases = {tc.test_id: tc for tc in test_cases_list}
|
|
244
|
+
|
|
245
|
+
# Compare matched results
|
|
246
|
+
matched_results = []
|
|
247
|
+
pipeline_a_only = []
|
|
248
|
+
pipeline_b_only = []
|
|
249
|
+
|
|
250
|
+
all_test_ids = set(results_a.keys()) | set(results_b.keys())
|
|
251
|
+
|
|
252
|
+
for test_id in all_test_ids:
|
|
253
|
+
result_a = results_a.get(test_id)
|
|
254
|
+
result_b = results_b.get(test_id)
|
|
255
|
+
|
|
256
|
+
if result_a and result_b:
|
|
257
|
+
# Both have results - compare using product-type-specific metric
|
|
258
|
+
metric_a = self._get_comparison_metric(result_a, product_type)
|
|
259
|
+
metric_b = self._get_comparison_metric(result_b, product_type)
|
|
260
|
+
|
|
261
|
+
# Load inference results for outputs
|
|
262
|
+
inference_a = self._load_inference_result(self.pipeline_a_dir, test_id)
|
|
263
|
+
inference_b = self._load_inference_result(self.pipeline_b_dir, test_id)
|
|
264
|
+
|
|
265
|
+
# Get test case for input file and schema
|
|
266
|
+
test_case = test_cases.get(test_id)
|
|
267
|
+
|
|
268
|
+
comparison: dict[str, Any] = {
|
|
269
|
+
"test_id": test_id,
|
|
270
|
+
"pipeline_a": {
|
|
271
|
+
"pipeline_name": result_a.pipeline_name,
|
|
272
|
+
"metric_value": metric_a,
|
|
273
|
+
"success": result_a.success,
|
|
274
|
+
"error": result_a.error,
|
|
275
|
+
"all_metrics": [m.model_dump() for m in result_a.metrics],
|
|
276
|
+
"all_stats": [s.model_dump() for s in result_a.stats],
|
|
277
|
+
},
|
|
278
|
+
"pipeline_b": {
|
|
279
|
+
"pipeline_name": result_b.pipeline_name,
|
|
280
|
+
"metric_value": metric_b,
|
|
281
|
+
"success": result_b.success,
|
|
282
|
+
"error": result_b.error,
|
|
283
|
+
"all_metrics": [m.model_dump() for m in result_b.metrics],
|
|
284
|
+
"all_stats": [s.model_dump() for s in result_b.stats],
|
|
285
|
+
},
|
|
286
|
+
"input_file": str(test_case.file_path) if test_case else None,
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
# Add product-type-specific data
|
|
290
|
+
if product_type == "layout_detection":
|
|
291
|
+
comparison["pipeline_a"]["predictions"] = self._get_predictions(inference_a)
|
|
292
|
+
comparison["pipeline_b"]["predictions"] = self._get_predictions(inference_b)
|
|
293
|
+
comparison["gt_annotations"] = self._get_gt_annotations(test_case)
|
|
294
|
+
elif product_type == "extract":
|
|
295
|
+
comparison["pipeline_a"]["output"] = (
|
|
296
|
+
inference_a.output.extracted_data
|
|
297
|
+
if inference_a and hasattr(inference_a.output, "extracted_data")
|
|
298
|
+
else None
|
|
299
|
+
)
|
|
300
|
+
comparison["pipeline_b"]["output"] = (
|
|
301
|
+
inference_b.output.extracted_data
|
|
302
|
+
if inference_b and hasattr(inference_b.output, "extracted_data")
|
|
303
|
+
else None
|
|
304
|
+
)
|
|
305
|
+
comparison["schema"] = (
|
|
306
|
+
test_case.data_schema if test_case and hasattr(test_case, "data_schema") else None
|
|
307
|
+
)
|
|
308
|
+
elif product_type == "parse":
|
|
309
|
+
comparison["pipeline_a"]["output"] = (
|
|
310
|
+
inference_a.output.markdown if inference_a and hasattr(inference_a.output, "markdown") else None
|
|
311
|
+
)
|
|
312
|
+
comparison["pipeline_b"]["output"] = (
|
|
313
|
+
inference_b.output.markdown if inference_b and hasattr(inference_b.output, "markdown") else None
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
# Determine comparison category
|
|
317
|
+
if metric_a is not None and metric_b is not None:
|
|
318
|
+
if metric_a > metric_b:
|
|
319
|
+
comparison["category"] = "a_better"
|
|
320
|
+
elif metric_b > metric_a:
|
|
321
|
+
comparison["category"] = "b_better"
|
|
322
|
+
else:
|
|
323
|
+
comparison["category"] = "tie"
|
|
324
|
+
elif metric_a is None and metric_b is None:
|
|
325
|
+
comparison["category"] = "both_bad"
|
|
326
|
+
elif metric_a is None:
|
|
327
|
+
comparison["category"] = "b_better"
|
|
328
|
+
else:
|
|
329
|
+
comparison["category"] = "a_better"
|
|
330
|
+
|
|
331
|
+
matched_results.append(comparison)
|
|
332
|
+
elif result_a:
|
|
333
|
+
pipeline_a_only.append(result_a.test_id)
|
|
334
|
+
elif result_b:
|
|
335
|
+
pipeline_b_only.append(result_b.test_id)
|
|
336
|
+
|
|
337
|
+
# Get pipeline names
|
|
338
|
+
pipeline_a_name = (
|
|
339
|
+
summary_a.per_example_results[0].pipeline_name if summary_a.per_example_results else "Pipeline A"
|
|
340
|
+
)
|
|
341
|
+
pipeline_b_name = (
|
|
342
|
+
summary_b.per_example_results[0].pipeline_name if summary_b.per_example_results else "Pipeline B"
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
# De-duplicate pipeline names if they're the same
|
|
346
|
+
if pipeline_a_name == pipeline_b_name:
|
|
347
|
+
# Extract distinguishing info from directory paths
|
|
348
|
+
suffix_a = self._get_directory_suffix(self.pipeline_a_dir)
|
|
349
|
+
suffix_b = self._get_directory_suffix(self.pipeline_b_dir)
|
|
350
|
+
|
|
351
|
+
if suffix_a != suffix_b:
|
|
352
|
+
pipeline_a_name = f"{pipeline_a_name} ({suffix_a})"
|
|
353
|
+
pipeline_b_name = f"{pipeline_b_name} ({suffix_b})"
|
|
354
|
+
else:
|
|
355
|
+
# Fallback to generic A/B if suffixes are also the same
|
|
356
|
+
pipeline_a_name = f"{pipeline_a_name} (A)"
|
|
357
|
+
pipeline_b_name = f"{pipeline_b_name} (B)"
|
|
358
|
+
|
|
359
|
+
# Calculate statistics
|
|
360
|
+
stats = {
|
|
361
|
+
"total_matched": len(matched_results),
|
|
362
|
+
"a_better": sum(1 for r in matched_results if r["category"] == "a_better"),
|
|
363
|
+
"b_better": sum(1 for r in matched_results if r["category"] == "b_better"),
|
|
364
|
+
"tie": sum(1 for r in matched_results if r["category"] == "tie"),
|
|
365
|
+
"both_bad": sum(1 for r in matched_results if r["category"] == "both_bad"),
|
|
366
|
+
"pipeline_a_only": len(pipeline_a_only),
|
|
367
|
+
"pipeline_b_only": len(pipeline_b_only),
|
|
368
|
+
"pipeline_a_name": pipeline_a_name,
|
|
369
|
+
"pipeline_b_name": pipeline_b_name,
|
|
370
|
+
"product_type": product_type,
|
|
371
|
+
"comparison_metric": comparison_metric,
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
return {
|
|
375
|
+
"matched_results": matched_results,
|
|
376
|
+
"pipeline_a_only": pipeline_a_only,
|
|
377
|
+
"pipeline_b_only": pipeline_b_only,
|
|
378
|
+
"stats": stats,
|
|
379
|
+
"product_type": product_type,
|
|
380
|
+
"comparison_metric": comparison_metric,
|
|
381
|
+
"original_base_path": str(self.test_cases_dir) if self.test_cases_dir else "",
|
|
382
|
+
}
|