parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,382 @@
1
+ """Comparison tool for evaluating two different pipeline results."""
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Any
6
+
7
+ from parse_bench.evaluation.layout_adapters import create_layout_adapter_for_result
8
+ from parse_bench.evaluation.layout_label_mappers import project_layout_predictions
9
+ from parse_bench.schemas.evaluation import EvaluationResult, EvaluationSummary
10
+ from parse_bench.schemas.pipeline_io import InferenceResult
11
+ from parse_bench.test_cases import load_test_cases
12
+ from parse_bench.test_cases.schema import LayoutDetectionTestCase
13
+
14
+
15
+ class PipelineComparison:
16
+ """Compare results from two different pipelines."""
17
+
18
+ def __init__(
19
+ self,
20
+ pipeline_a_dir: Path,
21
+ pipeline_b_dir: Path,
22
+ test_cases_dir: Path | None = None,
23
+ ):
24
+ """
25
+ Initialize comparison between two pipelines.
26
+
27
+ :param pipeline_a_dir: Directory containing pipeline A evaluation results
28
+ :param pipeline_b_dir: Directory containing pipeline B evaluation results
29
+ :param test_cases_dir: Optional directory containing test cases
30
+ """
31
+ self.pipeline_a_dir = Path(pipeline_a_dir)
32
+ self.pipeline_b_dir = Path(pipeline_b_dir)
33
+ self.test_cases_dir = Path(test_cases_dir) if test_cases_dir else None
34
+
35
+ # Ordered metric-name candidates per product type. The first name present
36
+ # in an evaluation result wins. Parse carries a fallback chain because
37
+ # layout-only parse runs (test cases with only ``LayoutTestRule`` entries)
38
+ # emit table-only metrics such as ``grits_trm_composite`` or layout-only
39
+ # metrics such as ``mAP@[.50:.95]`` instead of ``rule_pass_rate``.
40
+ #
41
+ # MUST stay in sync with
42
+ # ``comparison_core.py::COMPARISON_METRIC_CANDIDATES`` — enforced by
43
+ # ``tests/.../test_comparison_consistency.py``. ``rule_pass_rate`` is the
44
+ # canonical parse metric (pass/fail rule semantics from
45
+ # ``ParseEvaluator``); ``grits_trm_composite`` is the primary table-only
46
+ # parse metric. ``normalized_text_score`` is a secondary text-similarity
47
+ # signal and intentionally absent here.
48
+ METRIC_CANDIDATES: dict[str, tuple[str, ...]] = {
49
+ "extract": ("accuracy",),
50
+ "parse": ("rule_pass_rate", "grits_trm_composite", "mAP@[.50:.95]"),
51
+ "layout_detection": ("mAP@[.50:.95]",),
52
+ }
53
+ DEFAULT_METRIC: str = "accuracy"
54
+
55
+ def _detect_product_type(self, summary: EvaluationSummary) -> str:
56
+ """Detect product type from evaluation results."""
57
+ if summary.per_example_results:
58
+ return summary.per_example_results[0].product_type
59
+ return "parse" # default fallback
60
+
61
+ def _get_directory_suffix(self, pipeline_dir: Path) -> str:
62
+ """
63
+ Extract a distinguishing suffix from the pipeline directory path.
64
+
65
+ Looks for run IDs, dates, or other identifying info in parent directories.
66
+ Example paths:
67
+ /output/financial_tables_run-21391181794/llamaparse_agentic -> "run-21391181794"
68
+ /output/2025-01-27/llamaparse_agentic -> "2025-01-27"
69
+ /output/experiment_v2/llamaparse_agentic -> "experiment_v2"
70
+ """
71
+ import re
72
+
73
+ # Get the parent directory name (the run/experiment folder)
74
+ parent_name = pipeline_dir.parent.name
75
+
76
+ # Try to extract a run ID pattern (e.g., run-21391181794). Matrix-leg
77
+ # dirs append a dataset suffix (run-<id>-<dataset-slug>) — keep it, or
78
+ # two legs of the same parent run would get identical labels.
79
+ run_id_match = re.search(r"run-(\d+(?:-[A-Za-z0-9._-]+)?)", parent_name)
80
+ if run_id_match:
81
+ return f"run-{run_id_match.group(1)}"
82
+
83
+ # Try to extract a date pattern (e.g., 2025-01-27)
84
+ date_match = re.search(r"(\d{4}-\d{2}-\d{2})", parent_name)
85
+ if date_match:
86
+ return date_match.group(1)
87
+
88
+ # Fall back to the parent directory name
89
+ if parent_name and parent_name != "output":
90
+ return parent_name
91
+
92
+ # Last resort: use the full parent path's last 2 components
93
+ parts = pipeline_dir.parts
94
+ if len(parts) >= 2:
95
+ return "/".join(parts[-2:])
96
+
97
+ return str(pipeline_dir)
98
+
99
+ def _load_evaluation_summary(self, output_dir: Path) -> EvaluationSummary | None:
100
+ """Load evaluation summary from a directory."""
101
+ eval_report_path = output_dir / "_evaluation_report.json"
102
+ if not eval_report_path.exists():
103
+ return None
104
+ try:
105
+ with open(eval_report_path) as f:
106
+ data = json.load(f)
107
+ return EvaluationSummary.model_validate(data)
108
+ except Exception:
109
+ return None
110
+
111
+ def _load_inference_result(self, output_dir: Path, test_id: str) -> InferenceResult | None:
112
+ """Load inference result for a specific test_id."""
113
+ # Try to find the result file
114
+ # Result files are stored as: <test_id>/<test_id>.result.json
115
+ # But test_id might have slashes (group/filename)
116
+ parts = test_id.split("/")
117
+ if len(parts) == 2:
118
+ group, filename = parts
119
+ result_path = output_dir / group / f"{filename}.result.json"
120
+ else:
121
+ # Fallback: search for the file
122
+ result_path = output_dir / f"{test_id}.result.json"
123
+
124
+ if not result_path.exists():
125
+ # Try recursive search
126
+ for result_file in output_dir.rglob(f"*{test_id}*.result.json"):
127
+ result_path = result_file
128
+ break
129
+ else:
130
+ return None
131
+
132
+ try:
133
+ with open(result_path) as f:
134
+ data = json.load(f)
135
+ return InferenceResult.model_validate(data)
136
+ except Exception:
137
+ return None
138
+
139
+ def _get_accuracy(self, eval_result: EvaluationResult) -> float | None:
140
+ """Extract accuracy metric from evaluation result (backward compatibility)."""
141
+ for metric in eval_result.metrics:
142
+ if metric.metric_name == "accuracy":
143
+ return metric.value
144
+ return None
145
+
146
+ def _candidate_metric_names(self, product_type: str) -> tuple[str, ...]:
147
+ return self.METRIC_CANDIDATES.get(product_type, (self.DEFAULT_METRIC,))
148
+
149
+ def _get_comparison_metric(self, eval_result: EvaluationResult, product_type: str) -> float | None:
150
+ """Return the first candidate metric emitted by ``eval_result``.
151
+
152
+ Parse emits ``rule_pass_rate`` *or* ``mAP@[.50:.95]`` depending on
153
+ whether the test case carries parse rules or only layout rules.
154
+ """
155
+ available = {m.metric_name: m.value for m in eval_result.metrics}
156
+ for name in self._candidate_metric_names(product_type):
157
+ if name in available:
158
+ return available[name]
159
+ return None
160
+
161
+ def _resolve_comparison_metric_name(self, summary: EvaluationSummary, product_type: str) -> str:
162
+ """Pick the metric-name label used in stats/output headers.
163
+
164
+ Scans per-example results and returns the first candidate that at
165
+ least one example emitted; falls back to the first candidate so
166
+ empty summaries still get a sane label.
167
+ """
168
+ candidates = self._candidate_metric_names(product_type)
169
+ emitted = {m.metric_name for r in summary.per_example_results for m in r.metrics}
170
+ for name in candidates:
171
+ if name in emitted:
172
+ return name
173
+ return candidates[0]
174
+
175
+ def _get_predictions(self, inference: InferenceResult | None) -> list[dict] | None:
176
+ """Extract predictions as list of dicts for JSON serialization."""
177
+ if not inference or not inference.output:
178
+ return None
179
+ try:
180
+ adapter = create_layout_adapter_for_result(inference)
181
+ layout_output = adapter.to_layout_output(inference)
182
+ projected = project_layout_predictions(
183
+ inference,
184
+ layout_output,
185
+ evaluation_view="core",
186
+ target_ontology="canonical",
187
+ )
188
+ return [
189
+ {
190
+ "bbox": prediction["bbox"],
191
+ "class": prediction["class_name"],
192
+ "score": prediction["score"],
193
+ }
194
+ for prediction in projected
195
+ ]
196
+ except Exception:
197
+ return None
198
+
199
+ def _get_gt_annotations(self, test_case: Any) -> list[dict] | None:
200
+ """Extract GT annotations from test case."""
201
+ if not test_case or not isinstance(test_case, LayoutDetectionTestCase):
202
+ return None
203
+ annotations = test_case.get_layout_annotations()
204
+ if not annotations:
205
+ return None
206
+ return [{"bbox": ann.bbox, "class": ann.canonical_class} for ann in annotations]
207
+
208
+ def compare(self) -> dict[str, Any]:
209
+ """
210
+ Compare results from both pipelines.
211
+
212
+ Returns a dictionary with comparison data including:
213
+ - matched_results: List of comparisons
214
+ - pipeline_a_only: Results only in pipeline A
215
+ - pipeline_b_only: Results only in pipeline B
216
+ - stats: Summary statistics
217
+ - product_type: The detected product type
218
+ - comparison_metric: The metric used for comparison
219
+ """
220
+ # Load evaluation summaries
221
+ summary_a = self._load_evaluation_summary(self.pipeline_a_dir)
222
+ summary_b = self._load_evaluation_summary(self.pipeline_b_dir)
223
+
224
+ if not summary_a or not summary_b:
225
+ raise ValueError(
226
+ "Could not load evaluation summaries. "
227
+ "Make sure both directories contain _evaluation_report.json files. "
228
+ "Run evaluation first using: run_evaluation"
229
+ )
230
+
231
+ # Detect product type
232
+ product_type = self._detect_product_type(summary_a)
233
+ comparison_metric = self._resolve_comparison_metric_name(summary_a, product_type)
234
+
235
+ # Create mapping of test_id -> EvaluationResult
236
+ results_a = {r.test_id: r for r in summary_a.per_example_results}
237
+ results_b = {r.test_id: r for r in summary_b.per_example_results}
238
+
239
+ # Load test cases if available
240
+ test_cases: dict[str, Any] = {}
241
+ if self.test_cases_dir and self.test_cases_dir.exists():
242
+ test_cases_list = load_test_cases(self.test_cases_dir)
243
+ test_cases = {tc.test_id: tc for tc in test_cases_list}
244
+
245
+ # Compare matched results
246
+ matched_results = []
247
+ pipeline_a_only = []
248
+ pipeline_b_only = []
249
+
250
+ all_test_ids = set(results_a.keys()) | set(results_b.keys())
251
+
252
+ for test_id in all_test_ids:
253
+ result_a = results_a.get(test_id)
254
+ result_b = results_b.get(test_id)
255
+
256
+ if result_a and result_b:
257
+ # Both have results - compare using product-type-specific metric
258
+ metric_a = self._get_comparison_metric(result_a, product_type)
259
+ metric_b = self._get_comparison_metric(result_b, product_type)
260
+
261
+ # Load inference results for outputs
262
+ inference_a = self._load_inference_result(self.pipeline_a_dir, test_id)
263
+ inference_b = self._load_inference_result(self.pipeline_b_dir, test_id)
264
+
265
+ # Get test case for input file and schema
266
+ test_case = test_cases.get(test_id)
267
+
268
+ comparison: dict[str, Any] = {
269
+ "test_id": test_id,
270
+ "pipeline_a": {
271
+ "pipeline_name": result_a.pipeline_name,
272
+ "metric_value": metric_a,
273
+ "success": result_a.success,
274
+ "error": result_a.error,
275
+ "all_metrics": [m.model_dump() for m in result_a.metrics],
276
+ "all_stats": [s.model_dump() for s in result_a.stats],
277
+ },
278
+ "pipeline_b": {
279
+ "pipeline_name": result_b.pipeline_name,
280
+ "metric_value": metric_b,
281
+ "success": result_b.success,
282
+ "error": result_b.error,
283
+ "all_metrics": [m.model_dump() for m in result_b.metrics],
284
+ "all_stats": [s.model_dump() for s in result_b.stats],
285
+ },
286
+ "input_file": str(test_case.file_path) if test_case else None,
287
+ }
288
+
289
+ # Add product-type-specific data
290
+ if product_type == "layout_detection":
291
+ comparison["pipeline_a"]["predictions"] = self._get_predictions(inference_a)
292
+ comparison["pipeline_b"]["predictions"] = self._get_predictions(inference_b)
293
+ comparison["gt_annotations"] = self._get_gt_annotations(test_case)
294
+ elif product_type == "extract":
295
+ comparison["pipeline_a"]["output"] = (
296
+ inference_a.output.extracted_data
297
+ if inference_a and hasattr(inference_a.output, "extracted_data")
298
+ else None
299
+ )
300
+ comparison["pipeline_b"]["output"] = (
301
+ inference_b.output.extracted_data
302
+ if inference_b and hasattr(inference_b.output, "extracted_data")
303
+ else None
304
+ )
305
+ comparison["schema"] = (
306
+ test_case.data_schema if test_case and hasattr(test_case, "data_schema") else None
307
+ )
308
+ elif product_type == "parse":
309
+ comparison["pipeline_a"]["output"] = (
310
+ inference_a.output.markdown if inference_a and hasattr(inference_a.output, "markdown") else None
311
+ )
312
+ comparison["pipeline_b"]["output"] = (
313
+ inference_b.output.markdown if inference_b and hasattr(inference_b.output, "markdown") else None
314
+ )
315
+
316
+ # Determine comparison category
317
+ if metric_a is not None and metric_b is not None:
318
+ if metric_a > metric_b:
319
+ comparison["category"] = "a_better"
320
+ elif metric_b > metric_a:
321
+ comparison["category"] = "b_better"
322
+ else:
323
+ comparison["category"] = "tie"
324
+ elif metric_a is None and metric_b is None:
325
+ comparison["category"] = "both_bad"
326
+ elif metric_a is None:
327
+ comparison["category"] = "b_better"
328
+ else:
329
+ comparison["category"] = "a_better"
330
+
331
+ matched_results.append(comparison)
332
+ elif result_a:
333
+ pipeline_a_only.append(result_a.test_id)
334
+ elif result_b:
335
+ pipeline_b_only.append(result_b.test_id)
336
+
337
+ # Get pipeline names
338
+ pipeline_a_name = (
339
+ summary_a.per_example_results[0].pipeline_name if summary_a.per_example_results else "Pipeline A"
340
+ )
341
+ pipeline_b_name = (
342
+ summary_b.per_example_results[0].pipeline_name if summary_b.per_example_results else "Pipeline B"
343
+ )
344
+
345
+ # De-duplicate pipeline names if they're the same
346
+ if pipeline_a_name == pipeline_b_name:
347
+ # Extract distinguishing info from directory paths
348
+ suffix_a = self._get_directory_suffix(self.pipeline_a_dir)
349
+ suffix_b = self._get_directory_suffix(self.pipeline_b_dir)
350
+
351
+ if suffix_a != suffix_b:
352
+ pipeline_a_name = f"{pipeline_a_name} ({suffix_a})"
353
+ pipeline_b_name = f"{pipeline_b_name} ({suffix_b})"
354
+ else:
355
+ # Fallback to generic A/B if suffixes are also the same
356
+ pipeline_a_name = f"{pipeline_a_name} (A)"
357
+ pipeline_b_name = f"{pipeline_b_name} (B)"
358
+
359
+ # Calculate statistics
360
+ stats = {
361
+ "total_matched": len(matched_results),
362
+ "a_better": sum(1 for r in matched_results if r["category"] == "a_better"),
363
+ "b_better": sum(1 for r in matched_results if r["category"] == "b_better"),
364
+ "tie": sum(1 for r in matched_results if r["category"] == "tie"),
365
+ "both_bad": sum(1 for r in matched_results if r["category"] == "both_bad"),
366
+ "pipeline_a_only": len(pipeline_a_only),
367
+ "pipeline_b_only": len(pipeline_b_only),
368
+ "pipeline_a_name": pipeline_a_name,
369
+ "pipeline_b_name": pipeline_b_name,
370
+ "product_type": product_type,
371
+ "comparison_metric": comparison_metric,
372
+ }
373
+
374
+ return {
375
+ "matched_results": matched_results,
376
+ "pipeline_a_only": pipeline_a_only,
377
+ "pipeline_b_only": pipeline_b_only,
378
+ "stats": stats,
379
+ "product_type": product_type,
380
+ "comparison_metric": comparison_metric,
381
+ "original_base_path": str(self.test_cases_dir) if self.test_cases_dir else "",
382
+ }