parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Lightweight comparison module for evaluating two pipeline results.
|
|
3
|
+
|
|
4
|
+
This module has NO dependencies on Pydantic or other parse_bench modules,
|
|
5
|
+
making it suitable for use in the dashboard deployment where heavy deps aren't installed.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import re
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
# Ordered metric-name candidates per product type. The first name present
|
|
14
|
+
# in an evaluation result wins. Parse carries a fallback chain because
|
|
15
|
+
# layout-only parse runs (test cases with only ``LayoutTestRule`` entries)
|
|
16
|
+
# emit table-only metrics such as ``grits_trm_composite`` or layout-only
|
|
17
|
+
# metrics such as ``mAP@[.50:.95]`` instead of ``rule_pass_rate``.
|
|
18
|
+
#
|
|
19
|
+
# MUST stay in sync with ``comparison.py::PipelineComparison.METRIC_CANDIDATES``
|
|
20
|
+
# — enforced by ``tests/.../test_comparison_consistency.py``. The canonical
|
|
21
|
+
# parse metric is ``rule_pass_rate`` (pass/fail rule semantics from
|
|
22
|
+
# ``ParseEvaluator``). ``grits_trm_composite`` is the primary table-only parse
|
|
23
|
+
# metric. ``normalized_text_score`` is a secondary text-similarity signal and
|
|
24
|
+
# is intentionally NOT in the candidate list — when both are emitted for the
|
|
25
|
+
# same run, we pick the definitive rule-based score.
|
|
26
|
+
COMPARISON_METRIC_CANDIDATES: dict[str, tuple[str, ...]] = {
|
|
27
|
+
"extract": ("accuracy",),
|
|
28
|
+
"parse": ("rule_pass_rate", "grits_trm_composite", "mAP@[.50:.95]"),
|
|
29
|
+
"layout_detection": ("mAP@[.50:.95]",),
|
|
30
|
+
}
|
|
31
|
+
_DEFAULT_COMPARISON_METRIC = "accuracy"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def load_evaluation_report(pipeline_path: Path) -> dict | None:
|
|
35
|
+
"""Load evaluation report JSON from a pipeline directory."""
|
|
36
|
+
report_file = pipeline_path / "_evaluation_report.json"
|
|
37
|
+
if not report_file.exists():
|
|
38
|
+
return None
|
|
39
|
+
try:
|
|
40
|
+
with open(report_file) as f:
|
|
41
|
+
return json.load(f) # type: ignore[no-any-return]
|
|
42
|
+
except Exception:
|
|
43
|
+
return None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def load_inference_result(pipeline_path: Path, test_id: str) -> dict | None:
|
|
47
|
+
"""Load inference result for a specific test_id."""
|
|
48
|
+
# Result files are stored as: <group>/<filename>.result.json
|
|
49
|
+
parts = test_id.split("/")
|
|
50
|
+
if len(parts) == 2:
|
|
51
|
+
group, filename = parts
|
|
52
|
+
result_path = pipeline_path / group / f"{filename}.result.json"
|
|
53
|
+
else:
|
|
54
|
+
result_path = pipeline_path / f"{test_id}.result.json"
|
|
55
|
+
|
|
56
|
+
if not result_path.exists():
|
|
57
|
+
# Fallback: search recursively
|
|
58
|
+
for result_file in pipeline_path.rglob(f"*{test_id}*.result.json"):
|
|
59
|
+
result_path = result_file
|
|
60
|
+
break
|
|
61
|
+
else:
|
|
62
|
+
return None
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
with open(result_path) as f:
|
|
66
|
+
return json.load(f) # type: ignore[no-any-return]
|
|
67
|
+
except Exception:
|
|
68
|
+
return None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def get_metric_value(metrics_list: list, metric_name: str) -> float | None:
|
|
72
|
+
"""Extract a specific metric value from a metrics list."""
|
|
73
|
+
for metric in metrics_list:
|
|
74
|
+
if metric.get("metric_name") == metric_name:
|
|
75
|
+
return metric.get("value") # type: ignore[no-any-return]
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def get_directory_suffix(pipeline_dir: Path) -> str:
|
|
80
|
+
"""
|
|
81
|
+
Extract a distinguishing suffix from the pipeline directory path.
|
|
82
|
+
|
|
83
|
+
Looks for run IDs, dates, or other identifying info in parent directories.
|
|
84
|
+
"""
|
|
85
|
+
parent_name = pipeline_dir.parent.name
|
|
86
|
+
|
|
87
|
+
# Try to extract a run ID pattern (e.g., run-21391181794). Matrix-leg
|
|
88
|
+
# dirs append a dataset suffix (run-<id>-<dataset-slug>) — keep it, or
|
|
89
|
+
# two legs of the same parent run would get identical labels.
|
|
90
|
+
run_id_match = re.search(r"run-(\d+(?:-[A-Za-z0-9._-]+)?)", parent_name)
|
|
91
|
+
if run_id_match:
|
|
92
|
+
return f"run-{run_id_match.group(1)}"
|
|
93
|
+
|
|
94
|
+
# Try to extract a date pattern (e.g., 2025-01-27)
|
|
95
|
+
date_match = re.search(r"(\d{4}-\d{2}-\d{2})", parent_name)
|
|
96
|
+
if date_match:
|
|
97
|
+
return date_match.group(1)
|
|
98
|
+
|
|
99
|
+
# Fall back to the parent directory name
|
|
100
|
+
if parent_name and parent_name != "output":
|
|
101
|
+
return parent_name
|
|
102
|
+
|
|
103
|
+
# Last resort: use the full parent path's last 2 components
|
|
104
|
+
parts = pipeline_dir.parts
|
|
105
|
+
if len(parts) >= 2:
|
|
106
|
+
return "/".join(parts[-2:])
|
|
107
|
+
|
|
108
|
+
return str(pipeline_dir)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def get_predictions_from_inference(inference: dict | None) -> list[dict] | None:
|
|
112
|
+
"""Extract layout predictions from an inference result as list of dicts.
|
|
113
|
+
|
|
114
|
+
Reads ``output.layout_pages[*].items``: each item's ``bbox`` is xywh pixel,
|
|
115
|
+
``label`` (or falling back to ``item.type``) is the class name, and
|
|
116
|
+
``score`` is detector confidence. Returns bboxes in ``[x1, y1, x2, y2]``
|
|
117
|
+
(xyxy) to match ``comparison.py::_get_predictions`` and the dashboard
|
|
118
|
+
overlay renderer, both of which consume xyxy. Returns ``None`` when no
|
|
119
|
+
layout items are present.
|
|
120
|
+
"""
|
|
121
|
+
if not inference:
|
|
122
|
+
return None
|
|
123
|
+
output = inference.get("output")
|
|
124
|
+
if not output:
|
|
125
|
+
return None
|
|
126
|
+
layout_pages = output.get("layout_pages") or []
|
|
127
|
+
predictions: list[dict] = []
|
|
128
|
+
for page in layout_pages:
|
|
129
|
+
for item in page.get("items", []):
|
|
130
|
+
bbox = item.get("bbox")
|
|
131
|
+
if not bbox:
|
|
132
|
+
continue
|
|
133
|
+
x = float(bbox.get("x") or 0)
|
|
134
|
+
y = float(bbox.get("y") or 0)
|
|
135
|
+
w = float(bbox.get("w") or 0)
|
|
136
|
+
h = float(bbox.get("h") or 0)
|
|
137
|
+
predictions.append(
|
|
138
|
+
{
|
|
139
|
+
"bbox": [x, y, x + w, y + h],
|
|
140
|
+
"class": bbox.get("label") or item.get("type"),
|
|
141
|
+
"score": item.get("score"),
|
|
142
|
+
}
|
|
143
|
+
)
|
|
144
|
+
return predictions or None
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def compare_pipelines(
|
|
148
|
+
path_a: Path,
|
|
149
|
+
path_b: Path,
|
|
150
|
+
test_cases_dir: Path | None = None,
|
|
151
|
+
) -> dict[str, Any]:
|
|
152
|
+
"""
|
|
153
|
+
Compare results from two pipeline directories.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
path_a: Directory containing pipeline A evaluation results
|
|
157
|
+
path_b: Directory containing pipeline B evaluation results
|
|
158
|
+
test_cases_dir: Optional directory containing test cases (for input file paths)
|
|
159
|
+
|
|
160
|
+
Returns:
|
|
161
|
+
Dictionary with comparison data including:
|
|
162
|
+
- matched_results: List of per-example comparisons
|
|
163
|
+
- pipeline_a_only: Results only in pipeline A
|
|
164
|
+
- pipeline_b_only: Results only in pipeline B
|
|
165
|
+
- stats: Summary statistics
|
|
166
|
+
- product_type: The detected product type
|
|
167
|
+
- comparison_metric: The metric used for comparison
|
|
168
|
+
"""
|
|
169
|
+
path_a = Path(path_a)
|
|
170
|
+
path_b = Path(path_b)
|
|
171
|
+
|
|
172
|
+
# Load evaluation reports
|
|
173
|
+
report_a = load_evaluation_report(path_a)
|
|
174
|
+
report_b = load_evaluation_report(path_b)
|
|
175
|
+
|
|
176
|
+
if not report_a or not report_b:
|
|
177
|
+
raise ValueError(
|
|
178
|
+
"Could not load evaluation reports. Make sure both directories contain _evaluation_report.json files."
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
# Extract per-example results
|
|
182
|
+
results_a = {r["test_id"]: r for r in report_a.get("per_example_results", [])}
|
|
183
|
+
results_b = {r["test_id"]: r for r in report_b.get("per_example_results", [])}
|
|
184
|
+
|
|
185
|
+
# Detect product type from first result
|
|
186
|
+
product_type = "extract"
|
|
187
|
+
if results_a:
|
|
188
|
+
first_result = next(iter(results_a.values()))
|
|
189
|
+
product_type = first_result.get("product_type", "extract").lower()
|
|
190
|
+
|
|
191
|
+
metric_candidates = COMPARISON_METRIC_CANDIDATES.get(product_type, (_DEFAULT_COMPARISON_METRIC,))
|
|
192
|
+
|
|
193
|
+
def _pick_metric(metrics_list: list) -> float | None:
|
|
194
|
+
"""Return the first candidate metric value present in ``metrics_list``."""
|
|
195
|
+
by_name = {m.get("metric_name"): m.get("value") for m in metrics_list}
|
|
196
|
+
for name in metric_candidates:
|
|
197
|
+
if name in by_name:
|
|
198
|
+
return by_name[name] # type: ignore[no-any-return]
|
|
199
|
+
return None
|
|
200
|
+
|
|
201
|
+
# Resolve the label for the comparison metric to whichever candidate was
|
|
202
|
+
# actually emitted across examples. Mirrors
|
|
203
|
+
# ``comparison.py::_resolve_comparison_metric_name``: scan all emitted
|
|
204
|
+
# metric names, pick the highest-priority candidate present, fall back
|
|
205
|
+
# to the first candidate for empty/no-match runs so an empty summary
|
|
206
|
+
# still gets a sane label. Without this, layout-only parse runs mislabel
|
|
207
|
+
# their ``mAP@[.50:.95]`` values as ``rule_pass_rate``.
|
|
208
|
+
emitted_names = {
|
|
209
|
+
m.get("metric_name") for result in (*results_a.values(), *results_b.values()) for m in result.get("metrics", [])
|
|
210
|
+
}
|
|
211
|
+
comparison_metric = next(
|
|
212
|
+
(name for name in metric_candidates if name in emitted_names),
|
|
213
|
+
metric_candidates[0],
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
# Compare matched results
|
|
217
|
+
matched_results: list[dict[str, Any]] = []
|
|
218
|
+
pipeline_a_only: list[str] = []
|
|
219
|
+
pipeline_b_only: list[str] = []
|
|
220
|
+
|
|
221
|
+
all_test_ids = set(results_a.keys()) | set(results_b.keys())
|
|
222
|
+
|
|
223
|
+
for test_id in all_test_ids:
|
|
224
|
+
result_a = results_a.get(test_id)
|
|
225
|
+
result_b = results_b.get(test_id)
|
|
226
|
+
|
|
227
|
+
if result_a and result_b:
|
|
228
|
+
# Both have results - compare using the first fallback candidate
|
|
229
|
+
# that is emitted by this example (layout-only parse runs fall
|
|
230
|
+
# through to ``mAP@[.50:.95]``).
|
|
231
|
+
metrics_a = result_a.get("metrics", [])
|
|
232
|
+
metrics_b = result_b.get("metrics", [])
|
|
233
|
+
|
|
234
|
+
metric_a = _pick_metric(metrics_a)
|
|
235
|
+
metric_b = _pick_metric(metrics_b)
|
|
236
|
+
|
|
237
|
+
# Load inference results for output data
|
|
238
|
+
inference_a = load_inference_result(path_a, test_id)
|
|
239
|
+
inference_b = load_inference_result(path_b, test_id)
|
|
240
|
+
|
|
241
|
+
# Extract input file path from inference results
|
|
242
|
+
input_file_a = inference_a.get("request", {}).get("source_file_path") if inference_a else None
|
|
243
|
+
input_file_b = inference_b.get("request", {}).get("source_file_path") if inference_b else None
|
|
244
|
+
|
|
245
|
+
comparison: dict[str, Any] = {
|
|
246
|
+
"test_id": test_id,
|
|
247
|
+
"input_file": input_file_a or input_file_b,
|
|
248
|
+
"pipeline_a": {
|
|
249
|
+
"pipeline_name": result_a.get("pipeline_name", "Pipeline A"),
|
|
250
|
+
"metric_value": metric_a,
|
|
251
|
+
"success": result_a.get("success", False),
|
|
252
|
+
"error": result_a.get("error"),
|
|
253
|
+
"all_metrics": metrics_a,
|
|
254
|
+
"all_stats": result_a.get("stats", []),
|
|
255
|
+
},
|
|
256
|
+
"pipeline_b": {
|
|
257
|
+
"pipeline_name": result_b.get("pipeline_name", "Pipeline B"),
|
|
258
|
+
"metric_value": metric_b,
|
|
259
|
+
"success": result_b.get("success", False),
|
|
260
|
+
"error": result_b.get("error"),
|
|
261
|
+
"all_metrics": metrics_b,
|
|
262
|
+
"all_stats": result_b.get("stats", []),
|
|
263
|
+
},
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
# Add product-type-specific output data
|
|
267
|
+
if product_type == "layout_detection":
|
|
268
|
+
comparison["pipeline_a"]["predictions"] = get_predictions_from_inference(inference_a)
|
|
269
|
+
comparison["pipeline_b"]["predictions"] = get_predictions_from_inference(inference_b)
|
|
270
|
+
# GT annotations would need test case loading which we skip for now
|
|
271
|
+
comparison["gt_annotations"] = None
|
|
272
|
+
elif product_type == "extract":
|
|
273
|
+
output_a = inference_a.get("output", {}) if inference_a else {}
|
|
274
|
+
output_b = inference_b.get("output", {}) if inference_b else {}
|
|
275
|
+
comparison["pipeline_a"]["output"] = output_a.get("extracted_data")
|
|
276
|
+
comparison["pipeline_b"]["output"] = output_b.get("extracted_data")
|
|
277
|
+
elif product_type == "parse":
|
|
278
|
+
output_a = inference_a.get("output", {}) if inference_a else {}
|
|
279
|
+
output_b = inference_b.get("output", {}) if inference_b else {}
|
|
280
|
+
comparison["pipeline_a"]["output"] = output_a.get("markdown")
|
|
281
|
+
comparison["pipeline_b"]["output"] = output_b.get("markdown")
|
|
282
|
+
# Surface layout predictions (if any) so the dashboard can
|
|
283
|
+
# render the overlay view for layout-bearing parse runs.
|
|
284
|
+
predictions_a = get_predictions_from_inference(inference_a)
|
|
285
|
+
predictions_b = get_predictions_from_inference(inference_b)
|
|
286
|
+
if predictions_a or predictions_b:
|
|
287
|
+
comparison["pipeline_a"]["predictions"] = predictions_a
|
|
288
|
+
comparison["pipeline_b"]["predictions"] = predictions_b
|
|
289
|
+
comparison["gt_annotations"] = None
|
|
290
|
+
|
|
291
|
+
# Determine comparison category
|
|
292
|
+
if metric_a is not None and metric_b is not None:
|
|
293
|
+
if metric_a > metric_b:
|
|
294
|
+
comparison["category"] = "a_better"
|
|
295
|
+
elif metric_b > metric_a:
|
|
296
|
+
comparison["category"] = "b_better"
|
|
297
|
+
else:
|
|
298
|
+
comparison["category"] = "tie"
|
|
299
|
+
elif metric_a is None and metric_b is None:
|
|
300
|
+
comparison["category"] = "both_bad"
|
|
301
|
+
elif metric_a is None:
|
|
302
|
+
comparison["category"] = "b_better"
|
|
303
|
+
else:
|
|
304
|
+
comparison["category"] = "a_better"
|
|
305
|
+
|
|
306
|
+
matched_results.append(comparison)
|
|
307
|
+
elif result_a:
|
|
308
|
+
pipeline_a_only.append(test_id)
|
|
309
|
+
elif result_b:
|
|
310
|
+
pipeline_b_only.append(test_id)
|
|
311
|
+
|
|
312
|
+
# Get pipeline names from results
|
|
313
|
+
pipeline_a_name = "Pipeline A"
|
|
314
|
+
pipeline_b_name = "Pipeline B"
|
|
315
|
+
if results_a:
|
|
316
|
+
first_a = next(iter(results_a.values()))
|
|
317
|
+
pipeline_a_name = first_a.get("pipeline_name", path_a.name)
|
|
318
|
+
if results_b:
|
|
319
|
+
first_b = next(iter(results_b.values()))
|
|
320
|
+
pipeline_b_name = first_b.get("pipeline_name", path_b.name)
|
|
321
|
+
|
|
322
|
+
# Disambiguate if same name
|
|
323
|
+
if pipeline_a_name == pipeline_b_name:
|
|
324
|
+
suffix_a = get_directory_suffix(path_a)
|
|
325
|
+
suffix_b = get_directory_suffix(path_b)
|
|
326
|
+
|
|
327
|
+
if suffix_a != suffix_b:
|
|
328
|
+
pipeline_a_name = f"{pipeline_a_name} ({suffix_a})"
|
|
329
|
+
pipeline_b_name = f"{pipeline_b_name} ({suffix_b})"
|
|
330
|
+
else:
|
|
331
|
+
pipeline_a_name = f"{pipeline_a_name} (A)"
|
|
332
|
+
pipeline_b_name = f"{pipeline_b_name} (B)"
|
|
333
|
+
|
|
334
|
+
# Calculate statistics
|
|
335
|
+
stats = {
|
|
336
|
+
"total_matched": len(matched_results),
|
|
337
|
+
"a_better": sum(1 for r in matched_results if r["category"] == "a_better"),
|
|
338
|
+
"b_better": sum(1 for r in matched_results if r["category"] == "b_better"),
|
|
339
|
+
"tie": sum(1 for r in matched_results if r["category"] == "tie"),
|
|
340
|
+
"both_bad": sum(1 for r in matched_results if r["category"] == "both_bad"),
|
|
341
|
+
"pipeline_a_only": len(pipeline_a_only),
|
|
342
|
+
"pipeline_b_only": len(pipeline_b_only),
|
|
343
|
+
"pipeline_a_name": pipeline_a_name,
|
|
344
|
+
"pipeline_b_name": pipeline_b_name,
|
|
345
|
+
"product_type": product_type,
|
|
346
|
+
"comparison_metric": comparison_metric,
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
return {
|
|
350
|
+
"matched_results": matched_results,
|
|
351
|
+
"pipeline_a_only": pipeline_a_only,
|
|
352
|
+
"pipeline_b_only": pipeline_b_only,
|
|
353
|
+
"stats": stats,
|
|
354
|
+
"product_type": product_type,
|
|
355
|
+
"comparison_metric": comparison_metric,
|
|
356
|
+
"original_base_path": str(test_cases_dir) if test_cases_dir else "",
|
|
357
|
+
}
|