parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,472 @@
1
+ """Command-line interface for analysis tools."""
2
+
3
+ import json
4
+ import sys
5
+ import webbrowser
6
+ from pathlib import Path
7
+
8
+ import fire
9
+
10
+ from parse_bench.analysis.aggregation_report import generate_aggregation_report
11
+ from parse_bench.analysis.comparison import PipelineComparison
12
+ from parse_bench.analysis.comparison_report import generate_comparison_html
13
+ from parse_bench.analysis.detailed_report import generate_detailed_html_report
14
+ from parse_bench.analysis.leaderboard_report import generate_leaderboard_report
15
+ from parse_bench.schemas.evaluation import EvaluationSummary
16
+
17
+
18
+ class AnalysisCLI:
19
+ """Command-line interface for analyzing and comparing pipeline results."""
20
+
21
+ def compare_pipelines(
22
+ self,
23
+ pipeline_a_dir: str | Path,
24
+ pipeline_b_dir: str | Path,
25
+ test_cases_dir: str | Path | None = None,
26
+ output_file: str | Path | None = None,
27
+ ) -> int:
28
+ """
29
+ Compare results from two different pipelines.
30
+
31
+ Args:
32
+ pipeline_a_dir: Directory containing pipeline A evaluation results
33
+ pipeline_b_dir: Directory containing pipeline B evaluation results
34
+ test_cases_dir: Optional directory containing test cases (for input files and schemas)
35
+ output_file: Path to save the comparison HTML report
36
+ (default: pipeline_a_dir/comparison.html)
37
+
38
+ Returns:
39
+ Exit code (0 for success, non-zero for failure)
40
+ """
41
+ try:
42
+ pipeline_a_path = Path(pipeline_a_dir)
43
+ pipeline_b_path = Path(pipeline_b_dir)
44
+
45
+ if not pipeline_a_path.exists():
46
+ print(
47
+ f"Error: Pipeline A directory does not exist: {pipeline_a_path}",
48
+ file=sys.stderr,
49
+ )
50
+ return 1
51
+
52
+ if not pipeline_b_path.exists():
53
+ print(
54
+ f"Error: Pipeline B directory does not exist: {pipeline_b_path}",
55
+ file=sys.stderr,
56
+ )
57
+ return 1
58
+
59
+ # Auto-detect test_cases_dir if not provided
60
+ if test_cases_dir is None:
61
+ # Try to get from pipeline A metadata
62
+ metadata_path = pipeline_a_path / "_metadata.json"
63
+ if metadata_path.exists():
64
+ try:
65
+ import json
66
+
67
+ with open(metadata_path) as f:
68
+ metadata = json.load(f)
69
+ if "test_cases_dir" in metadata:
70
+ candidate = Path(metadata["test_cases_dir"])
71
+ if candidate.exists() and candidate.is_dir():
72
+ test_cases_dir = candidate
73
+ except Exception:
74
+ pass
75
+
76
+ test_cases_path = Path(test_cases_dir) if test_cases_dir else None
77
+
78
+ print("Comparing pipelines:")
79
+ print(f" Pipeline A: {pipeline_a_path}")
80
+ print(f" Pipeline B: {pipeline_b_path}")
81
+ if test_cases_path:
82
+ print(f" Test Cases: {test_cases_path}")
83
+
84
+ # Run comparison
85
+ comparison = PipelineComparison(
86
+ pipeline_a_dir=pipeline_a_path,
87
+ pipeline_b_dir=pipeline_b_path,
88
+ test_cases_dir=test_cases_path,
89
+ )
90
+
91
+ print("\nLoading and comparing results...")
92
+ comparison_data = comparison.compare()
93
+
94
+ stats = comparison_data["stats"]
95
+ print("\nComparison Results:")
96
+ print(f" Total Matched: {stats['total_matched']}")
97
+ print(f" {stats['pipeline_a_name']} Better: {stats['a_better']}")
98
+ print(f" {stats['pipeline_b_name']} Better: {stats['b_better']}")
99
+ print(f" Both Bad: {stats['both_bad']}")
100
+ print(f" Tie: {stats['tie']}")
101
+
102
+ # Generate HTML report
103
+ if output_file is None:
104
+ output_file = pipeline_a_path / "comparison.html"
105
+ else:
106
+ output_file = Path(output_file)
107
+
108
+ print("\nGenerating comparison report...")
109
+ report_path = generate_comparison_html(comparison_data, output_file)
110
+
111
+ print(f"\n✓ Comparison report saved to: {report_path.absolute()}") # type: ignore[union-attr]
112
+ print(" Open in browser to view interactive comparison")
113
+
114
+ return 0
115
+ except Exception as e:
116
+ import traceback
117
+
118
+ print(f"Error: {e}", file=sys.stderr)
119
+ traceback.print_exc()
120
+ return 1
121
+
122
+ def generate_report(
123
+ self,
124
+ evaluation_dir: str | Path,
125
+ test_cases_dir: str | Path | None = None,
126
+ output_dir: str | Path | None = None,
127
+ output_file: str | Path | None = None,
128
+ pdf_base_url: str | None = None,
129
+ pipeline_name: str | None = None,
130
+ group: str | None = None,
131
+ ) -> int:
132
+ """
133
+ Generate a detailed interactive HTML report from evaluation results.
134
+
135
+ This loads the evaluation summary JSON and generates an interactive HTML report
136
+ with drill-down capabilities for each test case, showing input files, outputs,
137
+ and metrics.
138
+
139
+ Args:
140
+ evaluation_dir: Directory containing evaluation results
141
+ (should have _evaluation_report.json)
142
+ test_cases_dir: Optional directory containing test cases
143
+ (for input files and schemas)
144
+ output_dir: Directory containing inference results
145
+ (*.result.json files). If not provided, defaults to
146
+ evaluation_dir. Use this when evaluation results are
147
+ stored separately.
148
+ output_file: Path to save the HTML report
149
+ (default: evaluation_dir/_evaluation_report_detailed.html)
150
+ pdf_base_url: Base URL for PDF files (e.g., http://localhost:8080/data).
151
+ If provided, this URL is pre-populated in the report for viewing PDFs.
152
+
153
+ Returns:
154
+ Exit code (0 for success, non-zero for failure)
155
+ """
156
+ try:
157
+ evaluation_path = Path(evaluation_dir)
158
+
159
+ if not evaluation_path.exists():
160
+ print(
161
+ f"Error: Evaluation directory does not exist: {evaluation_path}",
162
+ file=sys.stderr,
163
+ )
164
+ return 1
165
+
166
+ # Check for _evaluation_report.json at top level (single-category)
167
+ summary_json_path = evaluation_path / "_evaluation_report.json"
168
+
169
+ if not summary_json_path.exists():
170
+ # Auto-detect multi-category: look for subdirectories with reports
171
+ category_dirs = sorted(
172
+ d
173
+ for d in evaluation_path.iterdir()
174
+ if d.is_dir() and not d.name.startswith("_") and (d / "_evaluation_report.json").exists()
175
+ )
176
+ if category_dirs:
177
+ print(
178
+ f"Multi-category output detected. Generating reports for: "
179
+ f"{', '.join(d.name for d in category_dirs)}"
180
+ )
181
+ generated = []
182
+ for cat_dir in category_dirs:
183
+ print(f"\n--- {cat_dir.name} ---")
184
+ ret = self.generate_report(
185
+ evaluation_dir=str(cat_dir),
186
+ test_cases_dir=test_cases_dir,
187
+ output_dir=str(cat_dir) if output_dir is None else output_dir,
188
+ output_file=None,
189
+ pdf_base_url=pdf_base_url,
190
+ )
191
+ if ret == 0:
192
+ generated.append(cat_dir.name)
193
+ print(f"\n✓ Generated reports for: {', '.join(generated)}")
194
+ return 0
195
+ else:
196
+ print(
197
+ f"Error: Evaluation report not found: {summary_json_path}",
198
+ file=sys.stderr,
199
+ )
200
+ print(
201
+ " No per-category reports found either. Run evaluation first.",
202
+ file=sys.stderr,
203
+ )
204
+ return 1
205
+
206
+ print(f"Loading evaluation summary from: {summary_json_path}")
207
+ with open(summary_json_path) as f:
208
+ summary_data = json.load(f)
209
+ summary = EvaluationSummary.model_validate(summary_data)
210
+
211
+ # Auto-detect test_cases_dir if not provided
212
+ if test_cases_dir is None:
213
+ metadata_path = evaluation_path / "_metadata.json"
214
+ if not metadata_path.exists():
215
+ # Check parent for multi-category layout
216
+ metadata_path = evaluation_path.parent / "_metadata.json"
217
+ if metadata_path.exists():
218
+ try:
219
+ with open(metadata_path) as f:
220
+ metadata = json.load(f)
221
+ if "test_cases_dir" in metadata:
222
+ candidate = Path(metadata["test_cases_dir"])
223
+ if candidate.exists() and candidate.is_dir():
224
+ test_cases_dir = candidate
225
+ except Exception:
226
+ pass
227
+
228
+ test_cases_path = Path(test_cases_dir) if test_cases_dir else None
229
+
230
+ # Determine output_dir (where inference *.result.json files are)
231
+ if output_dir is None:
232
+ metadata_path = evaluation_path / "_metadata.json"
233
+ if not metadata_path.exists():
234
+ metadata_path = evaluation_path.parent / "_metadata.json"
235
+ if metadata_path.exists():
236
+ try:
237
+ with open(metadata_path) as f:
238
+ metadata = json.load(f)
239
+ if "output_dir" in metadata:
240
+ candidate = Path(metadata["output_dir"])
241
+ if candidate.exists() and candidate.is_dir():
242
+ output_dir = candidate
243
+ except Exception:
244
+ pass
245
+ if output_dir is None:
246
+ output_dir = evaluation_path
247
+ output_path = Path(output_dir)
248
+
249
+ # Determine output file
250
+ if output_file is None:
251
+ output_file = evaluation_path / "_evaluation_report_detailed.html"
252
+ else:
253
+ output_file = Path(output_file)
254
+
255
+ print("Generating detailed HTML report...")
256
+ print(f" Evaluation dir: {evaluation_path}")
257
+ print(f" Output dir (inference): {output_path}")
258
+ if test_cases_path:
259
+ print(f" Test cases dir: {test_cases_path}")
260
+ print(f" Output file: {output_file}")
261
+
262
+ # Generate report
263
+ report_path = generate_detailed_html_report(
264
+ summary=summary,
265
+ report_dir=evaluation_path,
266
+ output_dir=output_path,
267
+ test_cases_dir=test_cases_path,
268
+ pdf_base_url=pdf_base_url,
269
+ pipeline_name=pipeline_name,
270
+ group=group,
271
+ )
272
+
273
+ print(f"\n✓ Detailed report saved to: {report_path.absolute()}")
274
+ print(" Open in browser to view interactive report")
275
+
276
+ return 0
277
+ except Exception as e:
278
+ import traceback
279
+
280
+ print(f"Error: {e}", file=sys.stderr)
281
+ traceback.print_exc()
282
+ return 1
283
+
284
+ def generate_leaderboard(
285
+ self,
286
+ output_dir: str | Path = "./output",
287
+ pipelines: list[str] | None = None,
288
+ output_file: str | Path | None = None,
289
+ ) -> int:
290
+ """Generate a leaderboard comparing all pipelines side-by-side.
291
+
292
+ Args:
293
+ output_dir: Parent directory containing pipeline subdirectories (default: ./output)
294
+ pipelines: Optional list of pipeline directory names to include.
295
+ If not provided, auto-discovers all pipelines in output_dir.
296
+ output_file: Path to save the leaderboard HTML
297
+ (default: output_dir/_leaderboard.html)
298
+
299
+ Returns:
300
+ Exit code (0 for success, non-zero for failure)
301
+ """
302
+ try:
303
+ output_path = Path(output_dir)
304
+ if not output_path.exists():
305
+ print(f"Error: Output directory does not exist: {output_path}", file=sys.stderr)
306
+ return 1
307
+
308
+ pipeline_names = list(pipelines) if pipelines else None
309
+ out_file = Path(output_file) if output_file else None
310
+
311
+ print(f"Scanning for pipelines in: {output_path}")
312
+ report_path = generate_leaderboard_report(
313
+ output_dir=output_path,
314
+ pipeline_names=pipeline_names,
315
+ output_file=out_file,
316
+ )
317
+
318
+ print(f"\n✓ Leaderboard saved to: {report_path.absolute()}")
319
+ webbrowser.open(f"file://{report_path.absolute()}")
320
+ return 0
321
+ except Exception as e:
322
+ import traceback
323
+
324
+ print(f"Error: {e}", file=sys.stderr)
325
+ traceback.print_exc()
326
+ return 1
327
+
328
+ def serve(
329
+ self,
330
+ pipeline_dir: str | Path | None = None,
331
+ port: int = 8080,
332
+ root: str | Path = ".",
333
+ ) -> int:
334
+ """Start a local HTTP server to view reports with PDF rendering support.
335
+
336
+ Browsers block file:// access to PDFs for security reasons. This serves
337
+ the project root over HTTP so both reports and PDFs are accessible.
338
+
339
+ Args:
340
+ pipeline_dir: Pipeline output directory to open in browser
341
+ (e.g., ./output/llamaparse_agentic). If provided, opens the
342
+ dashboard or detailed report automatically.
343
+ port: Port number (default: 8080)
344
+ root: Root directory to serve (default: current directory).
345
+ Must contain both data/ and output/ subdirectories.
346
+
347
+ Returns:
348
+ Exit code (0 for success, non-zero for failure)
349
+ """
350
+ import http.server
351
+ import os
352
+ import socketserver
353
+ import webbrowser
354
+
355
+ serve_path = Path(root).resolve()
356
+ if not serve_path.exists():
357
+ print(f"Error: Directory does not exist: {serve_path}", file=sys.stderr)
358
+ return 1
359
+
360
+ os.chdir(serve_path)
361
+ handler = http.server.SimpleHTTPRequestHandler
362
+
363
+ # Find an available port, starting from the requested one
364
+ actual_port = port
365
+ httpd = None
366
+ for attempt_port in range(port, port + 100):
367
+ try:
368
+ httpd = socketserver.TCPServer(("", attempt_port), handler)
369
+ actual_port = attempt_port
370
+ break
371
+ except OSError:
372
+ continue
373
+
374
+ if httpd is None:
375
+ print(f"Error: Could not find an available port in range {port}-{port + 99}", file=sys.stderr)
376
+ return 1
377
+
378
+ url = f"http://localhost:{actual_port}"
379
+
380
+ # Determine what to open in browser
381
+ open_url = url
382
+ if pipeline_dir is not None:
383
+ rel_path = Path(pipeline_dir)
384
+ dashboard = rel_path / "_evaluation_report_dashboard.html"
385
+ detailed = rel_path / "_evaluation_report_detailed.html"
386
+ if dashboard.exists():
387
+ open_url = f"{url}/{dashboard}"
388
+ elif detailed.exists():
389
+ open_url = f"{url}/{detailed}"
390
+ else:
391
+ open_url = f"{url}/{rel_path}"
392
+
393
+ print(f"Serving from: {serve_path}")
394
+ print(f"URL: {url}")
395
+ if actual_port != port:
396
+ print(f" (port {port} was in use, using {actual_port})")
397
+ print(f"\nOpening: {open_url}")
398
+ print("Press Ctrl+C to stop\n")
399
+
400
+ webbrowser.open(open_url)
401
+
402
+ try:
403
+ httpd.serve_forever()
404
+ except KeyboardInterrupt:
405
+ print("\nServer stopped.")
406
+ finally:
407
+ httpd.server_close()
408
+ return 0
409
+
410
+ def generate_dashboard(
411
+ self,
412
+ evaluation_dir: str | Path,
413
+ groups: list[str] | None = None,
414
+ pipeline_name: str = "",
415
+ ) -> int:
416
+ """Generate an aggregation dashboard from per-category evaluation results.
417
+
418
+ Args:
419
+ evaluation_dir: Directory containing per-category subdirectories,
420
+ each with _evaluation_report.json.
421
+ groups: List of category names. If not provided, auto-discovers
422
+ subdirectories containing _evaluation_report.json.
423
+ pipeline_name: Pipeline name for display in the report header.
424
+
425
+ Returns:
426
+ Exit code (0 for success, non-zero for failure)
427
+ """
428
+ try:
429
+ eval_path = Path(evaluation_dir)
430
+ if not eval_path.exists():
431
+ print(f"Error: Directory does not exist: {eval_path}", file=sys.stderr)
432
+ return 1
433
+
434
+ # Auto-discover groups if not provided
435
+ if groups is None:
436
+ groups = sorted(
437
+ d.name
438
+ for d in eval_path.iterdir()
439
+ if d.is_dir() and not d.name.startswith("_") and (d / "_evaluation_report.json").exists()
440
+ )
441
+
442
+ if not groups:
443
+ print("Error: No category evaluation reports found", file=sys.stderr)
444
+ return 1
445
+
446
+ print(f"Generating dashboard for categories: {', '.join(groups)}")
447
+ report_path = generate_aggregation_report(
448
+ pipeline_output_dir=eval_path,
449
+ groups=groups,
450
+ pipeline_name=pipeline_name,
451
+ )
452
+ print(f"\n✓ Dashboard saved to: {report_path.absolute()}")
453
+ return 0
454
+ except Exception as e:
455
+ import traceback
456
+
457
+ print(f"Error: {e}", file=sys.stderr)
458
+ traceback.print_exc()
459
+ return 1
460
+
461
+
462
+ def main() -> int:
463
+ """Main entry point."""
464
+ cli = AnalysisCLI()
465
+ result = fire.Fire(cli)
466
+ if isinstance(result, int):
467
+ return result
468
+ return 0
469
+
470
+
471
+ if __name__ == "__main__":
472
+ sys.exit(main())