parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,357 @@
1
+ """
2
+ Lightweight comparison module for evaluating two pipeline results.
3
+
4
+ This module has NO dependencies on Pydantic or other parse_bench modules,
5
+ making it suitable for use in the dashboard deployment where heavy deps aren't installed.
6
+ """
7
+
8
+ import json
9
+ import re
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ # Ordered metric-name candidates per product type. The first name present
14
+ # in an evaluation result wins. Parse carries a fallback chain because
15
+ # layout-only parse runs (test cases with only ``LayoutTestRule`` entries)
16
+ # emit table-only metrics such as ``grits_trm_composite`` or layout-only
17
+ # metrics such as ``mAP@[.50:.95]`` instead of ``rule_pass_rate``.
18
+ #
19
+ # MUST stay in sync with ``comparison.py::PipelineComparison.METRIC_CANDIDATES``
20
+ # — enforced by ``tests/.../test_comparison_consistency.py``. The canonical
21
+ # parse metric is ``rule_pass_rate`` (pass/fail rule semantics from
22
+ # ``ParseEvaluator``). ``grits_trm_composite`` is the primary table-only parse
23
+ # metric. ``normalized_text_score`` is a secondary text-similarity signal and
24
+ # is intentionally NOT in the candidate list — when both are emitted for the
25
+ # same run, we pick the definitive rule-based score.
26
+ COMPARISON_METRIC_CANDIDATES: dict[str, tuple[str, ...]] = {
27
+ "extract": ("accuracy",),
28
+ "parse": ("rule_pass_rate", "grits_trm_composite", "mAP@[.50:.95]"),
29
+ "layout_detection": ("mAP@[.50:.95]",),
30
+ }
31
+ _DEFAULT_COMPARISON_METRIC = "accuracy"
32
+
33
+
34
+ def load_evaluation_report(pipeline_path: Path) -> dict | None:
35
+ """Load evaluation report JSON from a pipeline directory."""
36
+ report_file = pipeline_path / "_evaluation_report.json"
37
+ if not report_file.exists():
38
+ return None
39
+ try:
40
+ with open(report_file) as f:
41
+ return json.load(f) # type: ignore[no-any-return]
42
+ except Exception:
43
+ return None
44
+
45
+
46
+ def load_inference_result(pipeline_path: Path, test_id: str) -> dict | None:
47
+ """Load inference result for a specific test_id."""
48
+ # Result files are stored as: <group>/<filename>.result.json
49
+ parts = test_id.split("/")
50
+ if len(parts) == 2:
51
+ group, filename = parts
52
+ result_path = pipeline_path / group / f"{filename}.result.json"
53
+ else:
54
+ result_path = pipeline_path / f"{test_id}.result.json"
55
+
56
+ if not result_path.exists():
57
+ # Fallback: search recursively
58
+ for result_file in pipeline_path.rglob(f"*{test_id}*.result.json"):
59
+ result_path = result_file
60
+ break
61
+ else:
62
+ return None
63
+
64
+ try:
65
+ with open(result_path) as f:
66
+ return json.load(f) # type: ignore[no-any-return]
67
+ except Exception:
68
+ return None
69
+
70
+
71
+ def get_metric_value(metrics_list: list, metric_name: str) -> float | None:
72
+ """Extract a specific metric value from a metrics list."""
73
+ for metric in metrics_list:
74
+ if metric.get("metric_name") == metric_name:
75
+ return metric.get("value") # type: ignore[no-any-return]
76
+ return None
77
+
78
+
79
+ def get_directory_suffix(pipeline_dir: Path) -> str:
80
+ """
81
+ Extract a distinguishing suffix from the pipeline directory path.
82
+
83
+ Looks for run IDs, dates, or other identifying info in parent directories.
84
+ """
85
+ parent_name = pipeline_dir.parent.name
86
+
87
+ # Try to extract a run ID pattern (e.g., run-21391181794). Matrix-leg
88
+ # dirs append a dataset suffix (run-<id>-<dataset-slug>) — keep it, or
89
+ # two legs of the same parent run would get identical labels.
90
+ run_id_match = re.search(r"run-(\d+(?:-[A-Za-z0-9._-]+)?)", parent_name)
91
+ if run_id_match:
92
+ return f"run-{run_id_match.group(1)}"
93
+
94
+ # Try to extract a date pattern (e.g., 2025-01-27)
95
+ date_match = re.search(r"(\d{4}-\d{2}-\d{2})", parent_name)
96
+ if date_match:
97
+ return date_match.group(1)
98
+
99
+ # Fall back to the parent directory name
100
+ if parent_name and parent_name != "output":
101
+ return parent_name
102
+
103
+ # Last resort: use the full parent path's last 2 components
104
+ parts = pipeline_dir.parts
105
+ if len(parts) >= 2:
106
+ return "/".join(parts[-2:])
107
+
108
+ return str(pipeline_dir)
109
+
110
+
111
+ def get_predictions_from_inference(inference: dict | None) -> list[dict] | None:
112
+ """Extract layout predictions from an inference result as list of dicts.
113
+
114
+ Reads ``output.layout_pages[*].items``: each item's ``bbox`` is xywh pixel,
115
+ ``label`` (or falling back to ``item.type``) is the class name, and
116
+ ``score`` is detector confidence. Returns bboxes in ``[x1, y1, x2, y2]``
117
+ (xyxy) to match ``comparison.py::_get_predictions`` and the dashboard
118
+ overlay renderer, both of which consume xyxy. Returns ``None`` when no
119
+ layout items are present.
120
+ """
121
+ if not inference:
122
+ return None
123
+ output = inference.get("output")
124
+ if not output:
125
+ return None
126
+ layout_pages = output.get("layout_pages") or []
127
+ predictions: list[dict] = []
128
+ for page in layout_pages:
129
+ for item in page.get("items", []):
130
+ bbox = item.get("bbox")
131
+ if not bbox:
132
+ continue
133
+ x = float(bbox.get("x") or 0)
134
+ y = float(bbox.get("y") or 0)
135
+ w = float(bbox.get("w") or 0)
136
+ h = float(bbox.get("h") or 0)
137
+ predictions.append(
138
+ {
139
+ "bbox": [x, y, x + w, y + h],
140
+ "class": bbox.get("label") or item.get("type"),
141
+ "score": item.get("score"),
142
+ }
143
+ )
144
+ return predictions or None
145
+
146
+
147
+ def compare_pipelines(
148
+ path_a: Path,
149
+ path_b: Path,
150
+ test_cases_dir: Path | None = None,
151
+ ) -> dict[str, Any]:
152
+ """
153
+ Compare results from two pipeline directories.
154
+
155
+ Args:
156
+ path_a: Directory containing pipeline A evaluation results
157
+ path_b: Directory containing pipeline B evaluation results
158
+ test_cases_dir: Optional directory containing test cases (for input file paths)
159
+
160
+ Returns:
161
+ Dictionary with comparison data including:
162
+ - matched_results: List of per-example comparisons
163
+ - pipeline_a_only: Results only in pipeline A
164
+ - pipeline_b_only: Results only in pipeline B
165
+ - stats: Summary statistics
166
+ - product_type: The detected product type
167
+ - comparison_metric: The metric used for comparison
168
+ """
169
+ path_a = Path(path_a)
170
+ path_b = Path(path_b)
171
+
172
+ # Load evaluation reports
173
+ report_a = load_evaluation_report(path_a)
174
+ report_b = load_evaluation_report(path_b)
175
+
176
+ if not report_a or not report_b:
177
+ raise ValueError(
178
+ "Could not load evaluation reports. Make sure both directories contain _evaluation_report.json files."
179
+ )
180
+
181
+ # Extract per-example results
182
+ results_a = {r["test_id"]: r for r in report_a.get("per_example_results", [])}
183
+ results_b = {r["test_id"]: r for r in report_b.get("per_example_results", [])}
184
+
185
+ # Detect product type from first result
186
+ product_type = "extract"
187
+ if results_a:
188
+ first_result = next(iter(results_a.values()))
189
+ product_type = first_result.get("product_type", "extract").lower()
190
+
191
+ metric_candidates = COMPARISON_METRIC_CANDIDATES.get(product_type, (_DEFAULT_COMPARISON_METRIC,))
192
+
193
+ def _pick_metric(metrics_list: list) -> float | None:
194
+ """Return the first candidate metric value present in ``metrics_list``."""
195
+ by_name = {m.get("metric_name"): m.get("value") for m in metrics_list}
196
+ for name in metric_candidates:
197
+ if name in by_name:
198
+ return by_name[name] # type: ignore[no-any-return]
199
+ return None
200
+
201
+ # Resolve the label for the comparison metric to whichever candidate was
202
+ # actually emitted across examples. Mirrors
203
+ # ``comparison.py::_resolve_comparison_metric_name``: scan all emitted
204
+ # metric names, pick the highest-priority candidate present, fall back
205
+ # to the first candidate for empty/no-match runs so an empty summary
206
+ # still gets a sane label. Without this, layout-only parse runs mislabel
207
+ # their ``mAP@[.50:.95]`` values as ``rule_pass_rate``.
208
+ emitted_names = {
209
+ m.get("metric_name") for result in (*results_a.values(), *results_b.values()) for m in result.get("metrics", [])
210
+ }
211
+ comparison_metric = next(
212
+ (name for name in metric_candidates if name in emitted_names),
213
+ metric_candidates[0],
214
+ )
215
+
216
+ # Compare matched results
217
+ matched_results: list[dict[str, Any]] = []
218
+ pipeline_a_only: list[str] = []
219
+ pipeline_b_only: list[str] = []
220
+
221
+ all_test_ids = set(results_a.keys()) | set(results_b.keys())
222
+
223
+ for test_id in all_test_ids:
224
+ result_a = results_a.get(test_id)
225
+ result_b = results_b.get(test_id)
226
+
227
+ if result_a and result_b:
228
+ # Both have results - compare using the first fallback candidate
229
+ # that is emitted by this example (layout-only parse runs fall
230
+ # through to ``mAP@[.50:.95]``).
231
+ metrics_a = result_a.get("metrics", [])
232
+ metrics_b = result_b.get("metrics", [])
233
+
234
+ metric_a = _pick_metric(metrics_a)
235
+ metric_b = _pick_metric(metrics_b)
236
+
237
+ # Load inference results for output data
238
+ inference_a = load_inference_result(path_a, test_id)
239
+ inference_b = load_inference_result(path_b, test_id)
240
+
241
+ # Extract input file path from inference results
242
+ input_file_a = inference_a.get("request", {}).get("source_file_path") if inference_a else None
243
+ input_file_b = inference_b.get("request", {}).get("source_file_path") if inference_b else None
244
+
245
+ comparison: dict[str, Any] = {
246
+ "test_id": test_id,
247
+ "input_file": input_file_a or input_file_b,
248
+ "pipeline_a": {
249
+ "pipeline_name": result_a.get("pipeline_name", "Pipeline A"),
250
+ "metric_value": metric_a,
251
+ "success": result_a.get("success", False),
252
+ "error": result_a.get("error"),
253
+ "all_metrics": metrics_a,
254
+ "all_stats": result_a.get("stats", []),
255
+ },
256
+ "pipeline_b": {
257
+ "pipeline_name": result_b.get("pipeline_name", "Pipeline B"),
258
+ "metric_value": metric_b,
259
+ "success": result_b.get("success", False),
260
+ "error": result_b.get("error"),
261
+ "all_metrics": metrics_b,
262
+ "all_stats": result_b.get("stats", []),
263
+ },
264
+ }
265
+
266
+ # Add product-type-specific output data
267
+ if product_type == "layout_detection":
268
+ comparison["pipeline_a"]["predictions"] = get_predictions_from_inference(inference_a)
269
+ comparison["pipeline_b"]["predictions"] = get_predictions_from_inference(inference_b)
270
+ # GT annotations would need test case loading which we skip for now
271
+ comparison["gt_annotations"] = None
272
+ elif product_type == "extract":
273
+ output_a = inference_a.get("output", {}) if inference_a else {}
274
+ output_b = inference_b.get("output", {}) if inference_b else {}
275
+ comparison["pipeline_a"]["output"] = output_a.get("extracted_data")
276
+ comparison["pipeline_b"]["output"] = output_b.get("extracted_data")
277
+ elif product_type == "parse":
278
+ output_a = inference_a.get("output", {}) if inference_a else {}
279
+ output_b = inference_b.get("output", {}) if inference_b else {}
280
+ comparison["pipeline_a"]["output"] = output_a.get("markdown")
281
+ comparison["pipeline_b"]["output"] = output_b.get("markdown")
282
+ # Surface layout predictions (if any) so the dashboard can
283
+ # render the overlay view for layout-bearing parse runs.
284
+ predictions_a = get_predictions_from_inference(inference_a)
285
+ predictions_b = get_predictions_from_inference(inference_b)
286
+ if predictions_a or predictions_b:
287
+ comparison["pipeline_a"]["predictions"] = predictions_a
288
+ comparison["pipeline_b"]["predictions"] = predictions_b
289
+ comparison["gt_annotations"] = None
290
+
291
+ # Determine comparison category
292
+ if metric_a is not None and metric_b is not None:
293
+ if metric_a > metric_b:
294
+ comparison["category"] = "a_better"
295
+ elif metric_b > metric_a:
296
+ comparison["category"] = "b_better"
297
+ else:
298
+ comparison["category"] = "tie"
299
+ elif metric_a is None and metric_b is None:
300
+ comparison["category"] = "both_bad"
301
+ elif metric_a is None:
302
+ comparison["category"] = "b_better"
303
+ else:
304
+ comparison["category"] = "a_better"
305
+
306
+ matched_results.append(comparison)
307
+ elif result_a:
308
+ pipeline_a_only.append(test_id)
309
+ elif result_b:
310
+ pipeline_b_only.append(test_id)
311
+
312
+ # Get pipeline names from results
313
+ pipeline_a_name = "Pipeline A"
314
+ pipeline_b_name = "Pipeline B"
315
+ if results_a:
316
+ first_a = next(iter(results_a.values()))
317
+ pipeline_a_name = first_a.get("pipeline_name", path_a.name)
318
+ if results_b:
319
+ first_b = next(iter(results_b.values()))
320
+ pipeline_b_name = first_b.get("pipeline_name", path_b.name)
321
+
322
+ # Disambiguate if same name
323
+ if pipeline_a_name == pipeline_b_name:
324
+ suffix_a = get_directory_suffix(path_a)
325
+ suffix_b = get_directory_suffix(path_b)
326
+
327
+ if suffix_a != suffix_b:
328
+ pipeline_a_name = f"{pipeline_a_name} ({suffix_a})"
329
+ pipeline_b_name = f"{pipeline_b_name} ({suffix_b})"
330
+ else:
331
+ pipeline_a_name = f"{pipeline_a_name} (A)"
332
+ pipeline_b_name = f"{pipeline_b_name} (B)"
333
+
334
+ # Calculate statistics
335
+ stats = {
336
+ "total_matched": len(matched_results),
337
+ "a_better": sum(1 for r in matched_results if r["category"] == "a_better"),
338
+ "b_better": sum(1 for r in matched_results if r["category"] == "b_better"),
339
+ "tie": sum(1 for r in matched_results if r["category"] == "tie"),
340
+ "both_bad": sum(1 for r in matched_results if r["category"] == "both_bad"),
341
+ "pipeline_a_only": len(pipeline_a_only),
342
+ "pipeline_b_only": len(pipeline_b_only),
343
+ "pipeline_a_name": pipeline_a_name,
344
+ "pipeline_b_name": pipeline_b_name,
345
+ "product_type": product_type,
346
+ "comparison_metric": comparison_metric,
347
+ }
348
+
349
+ return {
350
+ "matched_results": matched_results,
351
+ "pipeline_a_only": pipeline_a_only,
352
+ "pipeline_b_only": pipeline_b_only,
353
+ "stats": stats,
354
+ "product_type": product_type,
355
+ "comparison_metric": comparison_metric,
356
+ "original_base_path": str(test_cases_dir) if test_cases_dir else "",
357
+ }