parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,472 @@
|
|
|
1
|
+
"""Command-line interface for analysis tools."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import sys
|
|
5
|
+
import webbrowser
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
import fire
|
|
9
|
+
|
|
10
|
+
from parse_bench.analysis.aggregation_report import generate_aggregation_report
|
|
11
|
+
from parse_bench.analysis.comparison import PipelineComparison
|
|
12
|
+
from parse_bench.analysis.comparison_report import generate_comparison_html
|
|
13
|
+
from parse_bench.analysis.detailed_report import generate_detailed_html_report
|
|
14
|
+
from parse_bench.analysis.leaderboard_report import generate_leaderboard_report
|
|
15
|
+
from parse_bench.schemas.evaluation import EvaluationSummary
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class AnalysisCLI:
|
|
19
|
+
"""Command-line interface for analyzing and comparing pipeline results."""
|
|
20
|
+
|
|
21
|
+
def compare_pipelines(
|
|
22
|
+
self,
|
|
23
|
+
pipeline_a_dir: str | Path,
|
|
24
|
+
pipeline_b_dir: str | Path,
|
|
25
|
+
test_cases_dir: str | Path | None = None,
|
|
26
|
+
output_file: str | Path | None = None,
|
|
27
|
+
) -> int:
|
|
28
|
+
"""
|
|
29
|
+
Compare results from two different pipelines.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
pipeline_a_dir: Directory containing pipeline A evaluation results
|
|
33
|
+
pipeline_b_dir: Directory containing pipeline B evaluation results
|
|
34
|
+
test_cases_dir: Optional directory containing test cases (for input files and schemas)
|
|
35
|
+
output_file: Path to save the comparison HTML report
|
|
36
|
+
(default: pipeline_a_dir/comparison.html)
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
Exit code (0 for success, non-zero for failure)
|
|
40
|
+
"""
|
|
41
|
+
try:
|
|
42
|
+
pipeline_a_path = Path(pipeline_a_dir)
|
|
43
|
+
pipeline_b_path = Path(pipeline_b_dir)
|
|
44
|
+
|
|
45
|
+
if not pipeline_a_path.exists():
|
|
46
|
+
print(
|
|
47
|
+
f"Error: Pipeline A directory does not exist: {pipeline_a_path}",
|
|
48
|
+
file=sys.stderr,
|
|
49
|
+
)
|
|
50
|
+
return 1
|
|
51
|
+
|
|
52
|
+
if not pipeline_b_path.exists():
|
|
53
|
+
print(
|
|
54
|
+
f"Error: Pipeline B directory does not exist: {pipeline_b_path}",
|
|
55
|
+
file=sys.stderr,
|
|
56
|
+
)
|
|
57
|
+
return 1
|
|
58
|
+
|
|
59
|
+
# Auto-detect test_cases_dir if not provided
|
|
60
|
+
if test_cases_dir is None:
|
|
61
|
+
# Try to get from pipeline A metadata
|
|
62
|
+
metadata_path = pipeline_a_path / "_metadata.json"
|
|
63
|
+
if metadata_path.exists():
|
|
64
|
+
try:
|
|
65
|
+
import json
|
|
66
|
+
|
|
67
|
+
with open(metadata_path) as f:
|
|
68
|
+
metadata = json.load(f)
|
|
69
|
+
if "test_cases_dir" in metadata:
|
|
70
|
+
candidate = Path(metadata["test_cases_dir"])
|
|
71
|
+
if candidate.exists() and candidate.is_dir():
|
|
72
|
+
test_cases_dir = candidate
|
|
73
|
+
except Exception:
|
|
74
|
+
pass
|
|
75
|
+
|
|
76
|
+
test_cases_path = Path(test_cases_dir) if test_cases_dir else None
|
|
77
|
+
|
|
78
|
+
print("Comparing pipelines:")
|
|
79
|
+
print(f" Pipeline A: {pipeline_a_path}")
|
|
80
|
+
print(f" Pipeline B: {pipeline_b_path}")
|
|
81
|
+
if test_cases_path:
|
|
82
|
+
print(f" Test Cases: {test_cases_path}")
|
|
83
|
+
|
|
84
|
+
# Run comparison
|
|
85
|
+
comparison = PipelineComparison(
|
|
86
|
+
pipeline_a_dir=pipeline_a_path,
|
|
87
|
+
pipeline_b_dir=pipeline_b_path,
|
|
88
|
+
test_cases_dir=test_cases_path,
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
print("\nLoading and comparing results...")
|
|
92
|
+
comparison_data = comparison.compare()
|
|
93
|
+
|
|
94
|
+
stats = comparison_data["stats"]
|
|
95
|
+
print("\nComparison Results:")
|
|
96
|
+
print(f" Total Matched: {stats['total_matched']}")
|
|
97
|
+
print(f" {stats['pipeline_a_name']} Better: {stats['a_better']}")
|
|
98
|
+
print(f" {stats['pipeline_b_name']} Better: {stats['b_better']}")
|
|
99
|
+
print(f" Both Bad: {stats['both_bad']}")
|
|
100
|
+
print(f" Tie: {stats['tie']}")
|
|
101
|
+
|
|
102
|
+
# Generate HTML report
|
|
103
|
+
if output_file is None:
|
|
104
|
+
output_file = pipeline_a_path / "comparison.html"
|
|
105
|
+
else:
|
|
106
|
+
output_file = Path(output_file)
|
|
107
|
+
|
|
108
|
+
print("\nGenerating comparison report...")
|
|
109
|
+
report_path = generate_comparison_html(comparison_data, output_file)
|
|
110
|
+
|
|
111
|
+
print(f"\n✓ Comparison report saved to: {report_path.absolute()}") # type: ignore[union-attr]
|
|
112
|
+
print(" Open in browser to view interactive comparison")
|
|
113
|
+
|
|
114
|
+
return 0
|
|
115
|
+
except Exception as e:
|
|
116
|
+
import traceback
|
|
117
|
+
|
|
118
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
119
|
+
traceback.print_exc()
|
|
120
|
+
return 1
|
|
121
|
+
|
|
122
|
+
def generate_report(
|
|
123
|
+
self,
|
|
124
|
+
evaluation_dir: str | Path,
|
|
125
|
+
test_cases_dir: str | Path | None = None,
|
|
126
|
+
output_dir: str | Path | None = None,
|
|
127
|
+
output_file: str | Path | None = None,
|
|
128
|
+
pdf_base_url: str | None = None,
|
|
129
|
+
pipeline_name: str | None = None,
|
|
130
|
+
group: str | None = None,
|
|
131
|
+
) -> int:
|
|
132
|
+
"""
|
|
133
|
+
Generate a detailed interactive HTML report from evaluation results.
|
|
134
|
+
|
|
135
|
+
This loads the evaluation summary JSON and generates an interactive HTML report
|
|
136
|
+
with drill-down capabilities for each test case, showing input files, outputs,
|
|
137
|
+
and metrics.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
evaluation_dir: Directory containing evaluation results
|
|
141
|
+
(should have _evaluation_report.json)
|
|
142
|
+
test_cases_dir: Optional directory containing test cases
|
|
143
|
+
(for input files and schemas)
|
|
144
|
+
output_dir: Directory containing inference results
|
|
145
|
+
(*.result.json files). If not provided, defaults to
|
|
146
|
+
evaluation_dir. Use this when evaluation results are
|
|
147
|
+
stored separately.
|
|
148
|
+
output_file: Path to save the HTML report
|
|
149
|
+
(default: evaluation_dir/_evaluation_report_detailed.html)
|
|
150
|
+
pdf_base_url: Base URL for PDF files (e.g., http://localhost:8080/data).
|
|
151
|
+
If provided, this URL is pre-populated in the report for viewing PDFs.
|
|
152
|
+
|
|
153
|
+
Returns:
|
|
154
|
+
Exit code (0 for success, non-zero for failure)
|
|
155
|
+
"""
|
|
156
|
+
try:
|
|
157
|
+
evaluation_path = Path(evaluation_dir)
|
|
158
|
+
|
|
159
|
+
if not evaluation_path.exists():
|
|
160
|
+
print(
|
|
161
|
+
f"Error: Evaluation directory does not exist: {evaluation_path}",
|
|
162
|
+
file=sys.stderr,
|
|
163
|
+
)
|
|
164
|
+
return 1
|
|
165
|
+
|
|
166
|
+
# Check for _evaluation_report.json at top level (single-category)
|
|
167
|
+
summary_json_path = evaluation_path / "_evaluation_report.json"
|
|
168
|
+
|
|
169
|
+
if not summary_json_path.exists():
|
|
170
|
+
# Auto-detect multi-category: look for subdirectories with reports
|
|
171
|
+
category_dirs = sorted(
|
|
172
|
+
d
|
|
173
|
+
for d in evaluation_path.iterdir()
|
|
174
|
+
if d.is_dir() and not d.name.startswith("_") and (d / "_evaluation_report.json").exists()
|
|
175
|
+
)
|
|
176
|
+
if category_dirs:
|
|
177
|
+
print(
|
|
178
|
+
f"Multi-category output detected. Generating reports for: "
|
|
179
|
+
f"{', '.join(d.name for d in category_dirs)}"
|
|
180
|
+
)
|
|
181
|
+
generated = []
|
|
182
|
+
for cat_dir in category_dirs:
|
|
183
|
+
print(f"\n--- {cat_dir.name} ---")
|
|
184
|
+
ret = self.generate_report(
|
|
185
|
+
evaluation_dir=str(cat_dir),
|
|
186
|
+
test_cases_dir=test_cases_dir,
|
|
187
|
+
output_dir=str(cat_dir) if output_dir is None else output_dir,
|
|
188
|
+
output_file=None,
|
|
189
|
+
pdf_base_url=pdf_base_url,
|
|
190
|
+
)
|
|
191
|
+
if ret == 0:
|
|
192
|
+
generated.append(cat_dir.name)
|
|
193
|
+
print(f"\n✓ Generated reports for: {', '.join(generated)}")
|
|
194
|
+
return 0
|
|
195
|
+
else:
|
|
196
|
+
print(
|
|
197
|
+
f"Error: Evaluation report not found: {summary_json_path}",
|
|
198
|
+
file=sys.stderr,
|
|
199
|
+
)
|
|
200
|
+
print(
|
|
201
|
+
" No per-category reports found either. Run evaluation first.",
|
|
202
|
+
file=sys.stderr,
|
|
203
|
+
)
|
|
204
|
+
return 1
|
|
205
|
+
|
|
206
|
+
print(f"Loading evaluation summary from: {summary_json_path}")
|
|
207
|
+
with open(summary_json_path) as f:
|
|
208
|
+
summary_data = json.load(f)
|
|
209
|
+
summary = EvaluationSummary.model_validate(summary_data)
|
|
210
|
+
|
|
211
|
+
# Auto-detect test_cases_dir if not provided
|
|
212
|
+
if test_cases_dir is None:
|
|
213
|
+
metadata_path = evaluation_path / "_metadata.json"
|
|
214
|
+
if not metadata_path.exists():
|
|
215
|
+
# Check parent for multi-category layout
|
|
216
|
+
metadata_path = evaluation_path.parent / "_metadata.json"
|
|
217
|
+
if metadata_path.exists():
|
|
218
|
+
try:
|
|
219
|
+
with open(metadata_path) as f:
|
|
220
|
+
metadata = json.load(f)
|
|
221
|
+
if "test_cases_dir" in metadata:
|
|
222
|
+
candidate = Path(metadata["test_cases_dir"])
|
|
223
|
+
if candidate.exists() and candidate.is_dir():
|
|
224
|
+
test_cases_dir = candidate
|
|
225
|
+
except Exception:
|
|
226
|
+
pass
|
|
227
|
+
|
|
228
|
+
test_cases_path = Path(test_cases_dir) if test_cases_dir else None
|
|
229
|
+
|
|
230
|
+
# Determine output_dir (where inference *.result.json files are)
|
|
231
|
+
if output_dir is None:
|
|
232
|
+
metadata_path = evaluation_path / "_metadata.json"
|
|
233
|
+
if not metadata_path.exists():
|
|
234
|
+
metadata_path = evaluation_path.parent / "_metadata.json"
|
|
235
|
+
if metadata_path.exists():
|
|
236
|
+
try:
|
|
237
|
+
with open(metadata_path) as f:
|
|
238
|
+
metadata = json.load(f)
|
|
239
|
+
if "output_dir" in metadata:
|
|
240
|
+
candidate = Path(metadata["output_dir"])
|
|
241
|
+
if candidate.exists() and candidate.is_dir():
|
|
242
|
+
output_dir = candidate
|
|
243
|
+
except Exception:
|
|
244
|
+
pass
|
|
245
|
+
if output_dir is None:
|
|
246
|
+
output_dir = evaluation_path
|
|
247
|
+
output_path = Path(output_dir)
|
|
248
|
+
|
|
249
|
+
# Determine output file
|
|
250
|
+
if output_file is None:
|
|
251
|
+
output_file = evaluation_path / "_evaluation_report_detailed.html"
|
|
252
|
+
else:
|
|
253
|
+
output_file = Path(output_file)
|
|
254
|
+
|
|
255
|
+
print("Generating detailed HTML report...")
|
|
256
|
+
print(f" Evaluation dir: {evaluation_path}")
|
|
257
|
+
print(f" Output dir (inference): {output_path}")
|
|
258
|
+
if test_cases_path:
|
|
259
|
+
print(f" Test cases dir: {test_cases_path}")
|
|
260
|
+
print(f" Output file: {output_file}")
|
|
261
|
+
|
|
262
|
+
# Generate report
|
|
263
|
+
report_path = generate_detailed_html_report(
|
|
264
|
+
summary=summary,
|
|
265
|
+
report_dir=evaluation_path,
|
|
266
|
+
output_dir=output_path,
|
|
267
|
+
test_cases_dir=test_cases_path,
|
|
268
|
+
pdf_base_url=pdf_base_url,
|
|
269
|
+
pipeline_name=pipeline_name,
|
|
270
|
+
group=group,
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
print(f"\n✓ Detailed report saved to: {report_path.absolute()}")
|
|
274
|
+
print(" Open in browser to view interactive report")
|
|
275
|
+
|
|
276
|
+
return 0
|
|
277
|
+
except Exception as e:
|
|
278
|
+
import traceback
|
|
279
|
+
|
|
280
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
281
|
+
traceback.print_exc()
|
|
282
|
+
return 1
|
|
283
|
+
|
|
284
|
+
def generate_leaderboard(
|
|
285
|
+
self,
|
|
286
|
+
output_dir: str | Path = "./output",
|
|
287
|
+
pipelines: list[str] | None = None,
|
|
288
|
+
output_file: str | Path | None = None,
|
|
289
|
+
) -> int:
|
|
290
|
+
"""Generate a leaderboard comparing all pipelines side-by-side.
|
|
291
|
+
|
|
292
|
+
Args:
|
|
293
|
+
output_dir: Parent directory containing pipeline subdirectories (default: ./output)
|
|
294
|
+
pipelines: Optional list of pipeline directory names to include.
|
|
295
|
+
If not provided, auto-discovers all pipelines in output_dir.
|
|
296
|
+
output_file: Path to save the leaderboard HTML
|
|
297
|
+
(default: output_dir/_leaderboard.html)
|
|
298
|
+
|
|
299
|
+
Returns:
|
|
300
|
+
Exit code (0 for success, non-zero for failure)
|
|
301
|
+
"""
|
|
302
|
+
try:
|
|
303
|
+
output_path = Path(output_dir)
|
|
304
|
+
if not output_path.exists():
|
|
305
|
+
print(f"Error: Output directory does not exist: {output_path}", file=sys.stderr)
|
|
306
|
+
return 1
|
|
307
|
+
|
|
308
|
+
pipeline_names = list(pipelines) if pipelines else None
|
|
309
|
+
out_file = Path(output_file) if output_file else None
|
|
310
|
+
|
|
311
|
+
print(f"Scanning for pipelines in: {output_path}")
|
|
312
|
+
report_path = generate_leaderboard_report(
|
|
313
|
+
output_dir=output_path,
|
|
314
|
+
pipeline_names=pipeline_names,
|
|
315
|
+
output_file=out_file,
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
print(f"\n✓ Leaderboard saved to: {report_path.absolute()}")
|
|
319
|
+
webbrowser.open(f"file://{report_path.absolute()}")
|
|
320
|
+
return 0
|
|
321
|
+
except Exception as e:
|
|
322
|
+
import traceback
|
|
323
|
+
|
|
324
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
325
|
+
traceback.print_exc()
|
|
326
|
+
return 1
|
|
327
|
+
|
|
328
|
+
def serve(
|
|
329
|
+
self,
|
|
330
|
+
pipeline_dir: str | Path | None = None,
|
|
331
|
+
port: int = 8080,
|
|
332
|
+
root: str | Path = ".",
|
|
333
|
+
) -> int:
|
|
334
|
+
"""Start a local HTTP server to view reports with PDF rendering support.
|
|
335
|
+
|
|
336
|
+
Browsers block file:// access to PDFs for security reasons. This serves
|
|
337
|
+
the project root over HTTP so both reports and PDFs are accessible.
|
|
338
|
+
|
|
339
|
+
Args:
|
|
340
|
+
pipeline_dir: Pipeline output directory to open in browser
|
|
341
|
+
(e.g., ./output/llamaparse_agentic). If provided, opens the
|
|
342
|
+
dashboard or detailed report automatically.
|
|
343
|
+
port: Port number (default: 8080)
|
|
344
|
+
root: Root directory to serve (default: current directory).
|
|
345
|
+
Must contain both data/ and output/ subdirectories.
|
|
346
|
+
|
|
347
|
+
Returns:
|
|
348
|
+
Exit code (0 for success, non-zero for failure)
|
|
349
|
+
"""
|
|
350
|
+
import http.server
|
|
351
|
+
import os
|
|
352
|
+
import socketserver
|
|
353
|
+
import webbrowser
|
|
354
|
+
|
|
355
|
+
serve_path = Path(root).resolve()
|
|
356
|
+
if not serve_path.exists():
|
|
357
|
+
print(f"Error: Directory does not exist: {serve_path}", file=sys.stderr)
|
|
358
|
+
return 1
|
|
359
|
+
|
|
360
|
+
os.chdir(serve_path)
|
|
361
|
+
handler = http.server.SimpleHTTPRequestHandler
|
|
362
|
+
|
|
363
|
+
# Find an available port, starting from the requested one
|
|
364
|
+
actual_port = port
|
|
365
|
+
httpd = None
|
|
366
|
+
for attempt_port in range(port, port + 100):
|
|
367
|
+
try:
|
|
368
|
+
httpd = socketserver.TCPServer(("", attempt_port), handler)
|
|
369
|
+
actual_port = attempt_port
|
|
370
|
+
break
|
|
371
|
+
except OSError:
|
|
372
|
+
continue
|
|
373
|
+
|
|
374
|
+
if httpd is None:
|
|
375
|
+
print(f"Error: Could not find an available port in range {port}-{port + 99}", file=sys.stderr)
|
|
376
|
+
return 1
|
|
377
|
+
|
|
378
|
+
url = f"http://localhost:{actual_port}"
|
|
379
|
+
|
|
380
|
+
# Determine what to open in browser
|
|
381
|
+
open_url = url
|
|
382
|
+
if pipeline_dir is not None:
|
|
383
|
+
rel_path = Path(pipeline_dir)
|
|
384
|
+
dashboard = rel_path / "_evaluation_report_dashboard.html"
|
|
385
|
+
detailed = rel_path / "_evaluation_report_detailed.html"
|
|
386
|
+
if dashboard.exists():
|
|
387
|
+
open_url = f"{url}/{dashboard}"
|
|
388
|
+
elif detailed.exists():
|
|
389
|
+
open_url = f"{url}/{detailed}"
|
|
390
|
+
else:
|
|
391
|
+
open_url = f"{url}/{rel_path}"
|
|
392
|
+
|
|
393
|
+
print(f"Serving from: {serve_path}")
|
|
394
|
+
print(f"URL: {url}")
|
|
395
|
+
if actual_port != port:
|
|
396
|
+
print(f" (port {port} was in use, using {actual_port})")
|
|
397
|
+
print(f"\nOpening: {open_url}")
|
|
398
|
+
print("Press Ctrl+C to stop\n")
|
|
399
|
+
|
|
400
|
+
webbrowser.open(open_url)
|
|
401
|
+
|
|
402
|
+
try:
|
|
403
|
+
httpd.serve_forever()
|
|
404
|
+
except KeyboardInterrupt:
|
|
405
|
+
print("\nServer stopped.")
|
|
406
|
+
finally:
|
|
407
|
+
httpd.server_close()
|
|
408
|
+
return 0
|
|
409
|
+
|
|
410
|
+
def generate_dashboard(
|
|
411
|
+
self,
|
|
412
|
+
evaluation_dir: str | Path,
|
|
413
|
+
groups: list[str] | None = None,
|
|
414
|
+
pipeline_name: str = "",
|
|
415
|
+
) -> int:
|
|
416
|
+
"""Generate an aggregation dashboard from per-category evaluation results.
|
|
417
|
+
|
|
418
|
+
Args:
|
|
419
|
+
evaluation_dir: Directory containing per-category subdirectories,
|
|
420
|
+
each with _evaluation_report.json.
|
|
421
|
+
groups: List of category names. If not provided, auto-discovers
|
|
422
|
+
subdirectories containing _evaluation_report.json.
|
|
423
|
+
pipeline_name: Pipeline name for display in the report header.
|
|
424
|
+
|
|
425
|
+
Returns:
|
|
426
|
+
Exit code (0 for success, non-zero for failure)
|
|
427
|
+
"""
|
|
428
|
+
try:
|
|
429
|
+
eval_path = Path(evaluation_dir)
|
|
430
|
+
if not eval_path.exists():
|
|
431
|
+
print(f"Error: Directory does not exist: {eval_path}", file=sys.stderr)
|
|
432
|
+
return 1
|
|
433
|
+
|
|
434
|
+
# Auto-discover groups if not provided
|
|
435
|
+
if groups is None:
|
|
436
|
+
groups = sorted(
|
|
437
|
+
d.name
|
|
438
|
+
for d in eval_path.iterdir()
|
|
439
|
+
if d.is_dir() and not d.name.startswith("_") and (d / "_evaluation_report.json").exists()
|
|
440
|
+
)
|
|
441
|
+
|
|
442
|
+
if not groups:
|
|
443
|
+
print("Error: No category evaluation reports found", file=sys.stderr)
|
|
444
|
+
return 1
|
|
445
|
+
|
|
446
|
+
print(f"Generating dashboard for categories: {', '.join(groups)}")
|
|
447
|
+
report_path = generate_aggregation_report(
|
|
448
|
+
pipeline_output_dir=eval_path,
|
|
449
|
+
groups=groups,
|
|
450
|
+
pipeline_name=pipeline_name,
|
|
451
|
+
)
|
|
452
|
+
print(f"\n✓ Dashboard saved to: {report_path.absolute()}")
|
|
453
|
+
return 0
|
|
454
|
+
except Exception as e:
|
|
455
|
+
import traceback
|
|
456
|
+
|
|
457
|
+
print(f"Error: {e}", file=sys.stderr)
|
|
458
|
+
traceback.print_exc()
|
|
459
|
+
return 1
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def main() -> int:
|
|
463
|
+
"""Main entry point."""
|
|
464
|
+
cli = AnalysisCLI()
|
|
465
|
+
result = fire.Fire(cli)
|
|
466
|
+
if isinstance(result, int):
|
|
467
|
+
return result
|
|
468
|
+
return 0
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
if __name__ == "__main__":
|
|
472
|
+
sys.exit(main())
|