parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,852 @@
|
|
|
1
|
+
"""Multi-pipeline leaderboard report.
|
|
2
|
+
|
|
3
|
+
Generates a self-contained HTML leaderboard comparing all pipelines in the
|
|
4
|
+
output directory side-by-side, with per-category metric selectors, best-score
|
|
5
|
+
highlighting, and links to individual pipeline dashboards.
|
|
6
|
+
|
|
7
|
+
Uses the same design system (Newsreader / Plus Jakarta Sans / JetBrains Mono,
|
|
8
|
+
warm editorial palette) as the other reports.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from datetime import UTC, datetime
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from parse_bench.analysis.aggregation_report import _DEFAULT_METRICS
|
|
19
|
+
from parse_bench.analysis.metric_definitions import display_name as _display_name
|
|
20
|
+
from parse_bench.schemas.evaluation import EvaluationSummary
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _load_pipeline_data(pipeline_dir: Path) -> dict[str, Any] | None:
|
|
24
|
+
"""Load pipeline metadata and per-category avg metrics from a pipeline output dir."""
|
|
25
|
+
metadata_path = pipeline_dir / "_metadata.json"
|
|
26
|
+
if not metadata_path.exists():
|
|
27
|
+
return None
|
|
28
|
+
|
|
29
|
+
try:
|
|
30
|
+
metadata = json.loads(metadata_path.read_text(encoding="utf-8"))
|
|
31
|
+
except Exception:
|
|
32
|
+
return None
|
|
33
|
+
|
|
34
|
+
pm = metadata.get("pipeline", {})
|
|
35
|
+
pipeline_name = pm.get("pipeline_name", pipeline_dir.name)
|
|
36
|
+
|
|
37
|
+
# Discover categories (subdirs with _evaluation_report.json)
|
|
38
|
+
categories: list[dict[str, Any]] = []
|
|
39
|
+
for subdir in sorted(pipeline_dir.iterdir()):
|
|
40
|
+
if not subdir.is_dir():
|
|
41
|
+
continue
|
|
42
|
+
report_path = subdir / "_evaluation_report.json"
|
|
43
|
+
if not report_path.exists():
|
|
44
|
+
continue
|
|
45
|
+
try:
|
|
46
|
+
summary = EvaluationSummary.model_validate(json.loads(report_path.read_text(encoding="utf-8")))
|
|
47
|
+
except Exception:
|
|
48
|
+
continue
|
|
49
|
+
|
|
50
|
+
# Extract avg metrics, same filtering as aggregation_report
|
|
51
|
+
metrics_dict: dict[str, float] = {}
|
|
52
|
+
for key in sorted(summary.aggregate_metrics.keys()):
|
|
53
|
+
if not key.startswith("avg_"):
|
|
54
|
+
continue
|
|
55
|
+
metric_name = key[len("avg_") :]
|
|
56
|
+
if "_predicted" in metric_name or "_judge" in metric_name:
|
|
57
|
+
continue
|
|
58
|
+
metrics_dict[metric_name] = summary.aggregate_metrics[key]
|
|
59
|
+
|
|
60
|
+
categories.append(
|
|
61
|
+
{
|
|
62
|
+
"name": subdir.name,
|
|
63
|
+
"files": summary.total_examples,
|
|
64
|
+
"metrics": metrics_dict,
|
|
65
|
+
}
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
if not categories:
|
|
69
|
+
return None
|
|
70
|
+
|
|
71
|
+
return {
|
|
72
|
+
"name": pipeline_name,
|
|
73
|
+
"dirName": pipeline_dir.name,
|
|
74
|
+
"displayName": pipeline_name.replace("_", " ").title(),
|
|
75
|
+
"provider": pm.get("provider_name", ""),
|
|
76
|
+
"productType": pm.get("product_type", ""),
|
|
77
|
+
"config": pm.get("config", {}),
|
|
78
|
+
"categories": categories,
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def generate_leaderboard_report(
|
|
83
|
+
output_dir: Path,
|
|
84
|
+
pipeline_names: list[str] | None = None,
|
|
85
|
+
output_file: Path | None = None,
|
|
86
|
+
) -> Path:
|
|
87
|
+
"""Generate a leaderboard HTML comparing multiple pipelines.
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
output_dir: Parent directory containing pipeline subdirectories.
|
|
91
|
+
pipeline_names: Optional list of pipeline dir names to include.
|
|
92
|
+
If None, auto-discovers all subdirs with _metadata.json.
|
|
93
|
+
output_file: Path for the output HTML. Defaults to output_dir/_leaderboard.html.
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
Path to the generated HTML file.
|
|
97
|
+
"""
|
|
98
|
+
output_dir = Path(output_dir)
|
|
99
|
+
|
|
100
|
+
# Discover or filter pipelines
|
|
101
|
+
if pipeline_names:
|
|
102
|
+
dirs = [output_dir / name for name in pipeline_names]
|
|
103
|
+
else:
|
|
104
|
+
dirs = sorted(d for d in output_dir.iterdir() if d.is_dir() and (d / "_metadata.json").exists())
|
|
105
|
+
|
|
106
|
+
pipelines: list[dict[str, Any]] = []
|
|
107
|
+
for d in dirs:
|
|
108
|
+
data = _load_pipeline_data(d)
|
|
109
|
+
if data is not None:
|
|
110
|
+
pipelines.append(data)
|
|
111
|
+
|
|
112
|
+
if not pipelines:
|
|
113
|
+
raise ValueError(f"No valid pipeline results found in {output_dir}")
|
|
114
|
+
|
|
115
|
+
# Collect union of categories and metrics
|
|
116
|
+
all_categories: list[str] = []
|
|
117
|
+
seen_cats: set[str] = set()
|
|
118
|
+
for p in pipelines:
|
|
119
|
+
for cat in p["categories"]:
|
|
120
|
+
if cat["name"] not in seen_cats:
|
|
121
|
+
all_categories.append(cat["name"])
|
|
122
|
+
seen_cats.add(cat["name"])
|
|
123
|
+
|
|
124
|
+
# Build scores matrix and collect per-category metrics
|
|
125
|
+
scores: dict[str, dict[str, dict[str, float]]] = {}
|
|
126
|
+
category_files: dict[str, dict[str, int]] = {}
|
|
127
|
+
category_metrics: dict[str, list[str]] = {}
|
|
128
|
+
all_metric_names: set[str] = set()
|
|
129
|
+
|
|
130
|
+
for cat_name in all_categories:
|
|
131
|
+
scores[cat_name] = {}
|
|
132
|
+
category_files[cat_name] = {}
|
|
133
|
+
metric_set: set[str] = set()
|
|
134
|
+
for p in pipelines:
|
|
135
|
+
cat_data = next((c for c in p["categories"] if c["name"] == cat_name), None)
|
|
136
|
+
if cat_data:
|
|
137
|
+
scores[cat_name][p["name"]] = cat_data["metrics"]
|
|
138
|
+
category_files[cat_name][p["name"]] = cat_data["files"]
|
|
139
|
+
metric_set.update(cat_data["metrics"].keys())
|
|
140
|
+
all_metric_names.update(cat_data["metrics"].keys())
|
|
141
|
+
else:
|
|
142
|
+
scores[cat_name][p["name"]] = {}
|
|
143
|
+
category_files[cat_name][p["name"]] = 0
|
|
144
|
+
category_metrics[cat_name] = sorted(metric_set)
|
|
145
|
+
|
|
146
|
+
# Build metric display names
|
|
147
|
+
metric_names_map: dict[str, str] = {}
|
|
148
|
+
for m in all_metric_names:
|
|
149
|
+
metric_names_map[m] = _display_name(m)
|
|
150
|
+
|
|
151
|
+
# Build default metrics per category
|
|
152
|
+
default_metrics: dict[str, str] = {}
|
|
153
|
+
for cat_name in all_categories:
|
|
154
|
+
default = _DEFAULT_METRICS.get(cat_name, "rule_pass_rate")
|
|
155
|
+
if default not in category_metrics.get(cat_name, []):
|
|
156
|
+
if "rule_pass_rate" in category_metrics.get(cat_name, []):
|
|
157
|
+
default = "rule_pass_rate"
|
|
158
|
+
else:
|
|
159
|
+
default = category_metrics[cat_name][0] if category_metrics[cat_name] else ""
|
|
160
|
+
default_metrics[cat_name] = default
|
|
161
|
+
|
|
162
|
+
data_blob = {
|
|
163
|
+
"generatedAt": datetime.now(UTC).strftime("%Y-%m-%d %H:%M:%S UTC"),
|
|
164
|
+
"defaultMetrics": default_metrics,
|
|
165
|
+
"pipelines": [
|
|
166
|
+
{
|
|
167
|
+
"name": p["name"],
|
|
168
|
+
"dirName": p["dirName"],
|
|
169
|
+
"displayName": p["displayName"],
|
|
170
|
+
"provider": p["provider"],
|
|
171
|
+
"productType": p["productType"],
|
|
172
|
+
"config": p["config"],
|
|
173
|
+
"dashboardUrl": p["dirName"] + "/_evaluation_report_dashboard.html",
|
|
174
|
+
}
|
|
175
|
+
for p in pipelines
|
|
176
|
+
],
|
|
177
|
+
"categories": all_categories,
|
|
178
|
+
"categoryDisplayNames": {c: c.replace("_", " ").title() for c in all_categories},
|
|
179
|
+
"categoryFiles": category_files,
|
|
180
|
+
"scores": scores,
|
|
181
|
+
"metricNames": metric_names_map,
|
|
182
|
+
"categoryMetrics": category_metrics,
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
data_json = json.dumps(data_blob, default=str, ensure_ascii=False)
|
|
186
|
+
data_json = data_json.replace("</script>", "<\\/script>")
|
|
187
|
+
data_json = data_json.replace("<!--", "<\\!--")
|
|
188
|
+
|
|
189
|
+
parts: list[str] = []
|
|
190
|
+
parts.append(_HTML_HEAD)
|
|
191
|
+
parts.append(_CSS)
|
|
192
|
+
parts.append("</style>\n</head>\n")
|
|
193
|
+
parts.append(_HTML_BODY)
|
|
194
|
+
parts.append("\n<script>\nconst DATA = ")
|
|
195
|
+
parts.append(data_json)
|
|
196
|
+
parts.append(";\n</script>\n")
|
|
197
|
+
parts.append("<script>\n")
|
|
198
|
+
parts.append(_JS)
|
|
199
|
+
parts.append("\n</script>\n")
|
|
200
|
+
parts.append("</body>\n</html>\n")
|
|
201
|
+
|
|
202
|
+
html = "".join(parts)
|
|
203
|
+
if output_file is None:
|
|
204
|
+
output_file = output_dir / "_leaderboard.html"
|
|
205
|
+
output_file = Path(output_file)
|
|
206
|
+
output_file.parent.mkdir(parents=True, exist_ok=True)
|
|
207
|
+
output_file.write_text(html, encoding="utf-8")
|
|
208
|
+
return output_file
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
# ---------------------------------------------------------------------------
|
|
212
|
+
# HTML template
|
|
213
|
+
# ---------------------------------------------------------------------------
|
|
214
|
+
|
|
215
|
+
_FONT_URL = (
|
|
216
|
+
"https://fonts.googleapis.com/css2?family=Newsreader:ital,opsz,wght@"
|
|
217
|
+
"0,6..72,400;0,6..72,600;0,6..72,700;1,6..72,400"
|
|
218
|
+
"&family=Plus+Jakarta+Sans:wght@400;500;600;700"
|
|
219
|
+
"&family=JetBrains+Mono:wght@400;500&display=swap"
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
_HTML_HEAD = f"""\
|
|
223
|
+
<!DOCTYPE html>
|
|
224
|
+
<html lang="en">
|
|
225
|
+
<head>
|
|
226
|
+
<meta charset="UTF-8">
|
|
227
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
|
228
|
+
<title>Benchmark Leaderboard</title>
|
|
229
|
+
<link rel="preconnect" href="https://fonts.googleapis.com">
|
|
230
|
+
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
|
231
|
+
<link href="{_FONT_URL}" rel="stylesheet">
|
|
232
|
+
<style>
|
|
233
|
+
"""
|
|
234
|
+
|
|
235
|
+
_CSS = """\
|
|
236
|
+
*, *::before, *::after { margin: 0; padding: 0; box-sizing: border-box; }
|
|
237
|
+
:root {
|
|
238
|
+
--bg: #f8f7f4;
|
|
239
|
+
--fg: #1c1917;
|
|
240
|
+
--card: #ffffff;
|
|
241
|
+
--border: #e7e5e4;
|
|
242
|
+
--muted: #78716c;
|
|
243
|
+
--muted-light: #a8a29e;
|
|
244
|
+
--cream: #faf9f6;
|
|
245
|
+
--emerald: #059669;
|
|
246
|
+
--emerald-bg: #ecfdf5;
|
|
247
|
+
--emerald-light: #d1fae5;
|
|
248
|
+
--amber: #d97706;
|
|
249
|
+
--amber-bg: #fffbeb;
|
|
250
|
+
--red: #dc2626;
|
|
251
|
+
--red-bg: #fef2f2;
|
|
252
|
+
--blue: #2563eb;
|
|
253
|
+
--blue-bg: #eff6ff;
|
|
254
|
+
--gold: #b8860b;
|
|
255
|
+
--gold-bg: #fef9e7;
|
|
256
|
+
--gold-border: #e6c547;
|
|
257
|
+
--font-heading: 'Newsreader', Georgia, serif;
|
|
258
|
+
--font-body: 'Plus Jakarta Sans', -apple-system, BlinkMacSystemFont, sans-serif;
|
|
259
|
+
--font-mono: 'JetBrains Mono', 'SF Mono', monospace;
|
|
260
|
+
--shadow-sm: 0 1px 2px rgba(28,25,23,0.05);
|
|
261
|
+
--shadow-md: 0 4px 6px -1px rgba(28,25,23,0.07), 0 2px 4px -2px rgba(28,25,23,0.05);
|
|
262
|
+
--shadow-lg: 0 10px 25px -5px rgba(28,25,23,0.1), 0 4px 10px -4px rgba(28,25,23,0.06);
|
|
263
|
+
--radius: 12px;
|
|
264
|
+
--radius-sm: 6px;
|
|
265
|
+
}
|
|
266
|
+
html { font-size: 15px; }
|
|
267
|
+
body {
|
|
268
|
+
font-family: var(--font-body);
|
|
269
|
+
background: var(--bg);
|
|
270
|
+
color: var(--fg);
|
|
271
|
+
line-height: 1.6;
|
|
272
|
+
-webkit-font-smoothing: antialiased;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
::-webkit-scrollbar { width: 8px; height: 8px; }
|
|
276
|
+
::-webkit-scrollbar-track { background: var(--cream); }
|
|
277
|
+
::-webkit-scrollbar-thumb { background: var(--muted-light); border-radius: 4px; }
|
|
278
|
+
::-webkit-scrollbar-thumb:hover { background: var(--muted); }
|
|
279
|
+
|
|
280
|
+
.report-container {
|
|
281
|
+
max-width: 1600px;
|
|
282
|
+
margin: 0 auto;
|
|
283
|
+
padding: 40px 32px 80px;
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
/* ───── Header ───── */
|
|
287
|
+
.report-header { margin-bottom: 40px; }
|
|
288
|
+
.report-header h1 {
|
|
289
|
+
font-family: var(--font-heading);
|
|
290
|
+
font-size: 2.6rem;
|
|
291
|
+
font-weight: 700;
|
|
292
|
+
letter-spacing: -0.03em;
|
|
293
|
+
color: var(--fg);
|
|
294
|
+
line-height: 1.15;
|
|
295
|
+
}
|
|
296
|
+
.report-header .subtitle {
|
|
297
|
+
font-size: 0.85rem;
|
|
298
|
+
color: var(--muted);
|
|
299
|
+
margin-top: 8px;
|
|
300
|
+
letter-spacing: 0.01em;
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/* ───── Table wrapper ───── */
|
|
304
|
+
.leaderboard-wrap {
|
|
305
|
+
overflow-x: auto;
|
|
306
|
+
background: var(--card);
|
|
307
|
+
border: 1px solid var(--border);
|
|
308
|
+
border-radius: var(--radius);
|
|
309
|
+
box-shadow: var(--shadow-lg);
|
|
310
|
+
}
|
|
311
|
+
.leaderboard-table {
|
|
312
|
+
width: 100%;
|
|
313
|
+
border-collapse: collapse;
|
|
314
|
+
min-width: 600px;
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
/* ───── Cells ───── */
|
|
318
|
+
.leaderboard-table th,
|
|
319
|
+
.leaderboard-table td {
|
|
320
|
+
padding: 18px 24px;
|
|
321
|
+
border-bottom: 1px solid var(--border);
|
|
322
|
+
text-align: center;
|
|
323
|
+
vertical-align: middle;
|
|
324
|
+
transition: background 0.12s ease;
|
|
325
|
+
}
|
|
326
|
+
.leaderboard-table th {
|
|
327
|
+
background: var(--cream);
|
|
328
|
+
font-size: 0.72rem;
|
|
329
|
+
font-weight: 700;
|
|
330
|
+
text-transform: uppercase;
|
|
331
|
+
letter-spacing: 0.05em;
|
|
332
|
+
color: var(--muted);
|
|
333
|
+
position: sticky;
|
|
334
|
+
top: 0;
|
|
335
|
+
z-index: 2;
|
|
336
|
+
padding: 20px 24px;
|
|
337
|
+
border-bottom: 2px solid var(--border);
|
|
338
|
+
vertical-align: bottom;
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
/* Sticky first column */
|
|
342
|
+
.leaderboard-table th:first-child,
|
|
343
|
+
.leaderboard-table td:first-child {
|
|
344
|
+
text-align: left;
|
|
345
|
+
position: sticky;
|
|
346
|
+
left: 0;
|
|
347
|
+
z-index: 1;
|
|
348
|
+
background: var(--card);
|
|
349
|
+
border-right: 1px solid var(--border);
|
|
350
|
+
min-width: 220px;
|
|
351
|
+
padding-left: 28px;
|
|
352
|
+
}
|
|
353
|
+
.leaderboard-table th:first-child {
|
|
354
|
+
background: var(--cream);
|
|
355
|
+
z-index: 3;
|
|
356
|
+
vertical-align: bottom;
|
|
357
|
+
}
|
|
358
|
+
.category-header-label {
|
|
359
|
+
font-family: var(--font-heading);
|
|
360
|
+
font-size: 1rem;
|
|
361
|
+
font-weight: 600;
|
|
362
|
+
color: var(--fg);
|
|
363
|
+
text-transform: none;
|
|
364
|
+
letter-spacing: -0.01em;
|
|
365
|
+
}
|
|
366
|
+
.leaderboard-table tbody tr:last-child td {
|
|
367
|
+
border-bottom: none;
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
/* Row hover (category label column only) */
|
|
371
|
+
.leaderboard-table tbody tr:not(.overall-row):hover td:first-child {
|
|
372
|
+
background: #f5f4f1;
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
/* ───── Pipeline header ───── */
|
|
376
|
+
.pipeline-header {
|
|
377
|
+
display: flex;
|
|
378
|
+
flex-direction: column;
|
|
379
|
+
align-items: center;
|
|
380
|
+
gap: 4px;
|
|
381
|
+
min-width: 140px;
|
|
382
|
+
}
|
|
383
|
+
.pipeline-header .pipeline-crown {
|
|
384
|
+
font-size: 1.3rem;
|
|
385
|
+
line-height: 1;
|
|
386
|
+
filter: drop-shadow(0 1px 2px rgba(184,134,11,0.3));
|
|
387
|
+
}
|
|
388
|
+
.pipeline-header .pipeline-name {
|
|
389
|
+
font-family: var(--font-heading);
|
|
390
|
+
font-size: 1rem;
|
|
391
|
+
font-weight: 700;
|
|
392
|
+
color: var(--fg);
|
|
393
|
+
text-transform: none;
|
|
394
|
+
letter-spacing: -0.01em;
|
|
395
|
+
line-height: 1.3;
|
|
396
|
+
}
|
|
397
|
+
.pipeline-header .pipeline-name a {
|
|
398
|
+
color: inherit;
|
|
399
|
+
text-decoration: none;
|
|
400
|
+
border-bottom: 1px solid transparent;
|
|
401
|
+
transition: border-color 0.15s, color 0.15s;
|
|
402
|
+
}
|
|
403
|
+
.pipeline-header .pipeline-name a:hover {
|
|
404
|
+
color: var(--blue);
|
|
405
|
+
border-bottom-color: var(--blue);
|
|
406
|
+
}
|
|
407
|
+
.pipeline-header .pipeline-tier {
|
|
408
|
+
font-family: var(--font-mono);
|
|
409
|
+
font-size: 0.68rem;
|
|
410
|
+
font-weight: 500;
|
|
411
|
+
color: var(--muted-light);
|
|
412
|
+
text-transform: none;
|
|
413
|
+
letter-spacing: 0.02em;
|
|
414
|
+
background: var(--bg);
|
|
415
|
+
padding: 2px 8px;
|
|
416
|
+
border-radius: 99px;
|
|
417
|
+
border: 1px solid var(--border);
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
/* Winner column header glow */
|
|
421
|
+
.pipeline-header.is-winner .pipeline-name a {
|
|
422
|
+
color: var(--gold);
|
|
423
|
+
}
|
|
424
|
+
.pipeline-header.is-winner .pipeline-tier {
|
|
425
|
+
background: var(--gold-bg);
|
|
426
|
+
border-color: var(--gold-border);
|
|
427
|
+
color: var(--gold);
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
/* ───── Category cell ───── */
|
|
431
|
+
.category-cell {
|
|
432
|
+
display: flex;
|
|
433
|
+
flex-direction: column;
|
|
434
|
+
gap: 6px;
|
|
435
|
+
}
|
|
436
|
+
.category-name {
|
|
437
|
+
font-family: var(--font-heading);
|
|
438
|
+
font-size: 1.05rem;
|
|
439
|
+
font-weight: 600;
|
|
440
|
+
color: var(--fg);
|
|
441
|
+
line-height: 1.3;
|
|
442
|
+
}
|
|
443
|
+
.category-name .file-count {
|
|
444
|
+
font-family: var(--font-body);
|
|
445
|
+
font-size: 0.72rem;
|
|
446
|
+
font-weight: 500;
|
|
447
|
+
color: var(--muted-light);
|
|
448
|
+
margin-left: 2px;
|
|
449
|
+
}
|
|
450
|
+
.category-selector {
|
|
451
|
+
width: 100%;
|
|
452
|
+
max-width: 210px;
|
|
453
|
+
padding: 5px 8px;
|
|
454
|
+
font-family: var(--font-body);
|
|
455
|
+
font-size: 0.73rem;
|
|
456
|
+
border: 1px solid var(--border);
|
|
457
|
+
border-radius: var(--radius-sm);
|
|
458
|
+
background: var(--cream);
|
|
459
|
+
color: var(--muted);
|
|
460
|
+
cursor: pointer;
|
|
461
|
+
outline: none;
|
|
462
|
+
transition: border-color 0.15s;
|
|
463
|
+
}
|
|
464
|
+
.category-selector:focus { border-color: var(--blue); }
|
|
465
|
+
.category-selector:hover { border-color: var(--muted-light); }
|
|
466
|
+
|
|
467
|
+
/* ───── Column hover: simple bounding box ───── */
|
|
468
|
+
.leaderboard-table th[data-col],
|
|
469
|
+
.leaderboard-table td[data-col] {
|
|
470
|
+
cursor: pointer;
|
|
471
|
+
}
|
|
472
|
+
.leaderboard-table th[data-col].col-hover {
|
|
473
|
+
box-shadow: inset 2px 0 0 var(--muted), inset -2px 0 0 var(--muted), inset 0 2px 0 var(--muted);
|
|
474
|
+
}
|
|
475
|
+
.leaderboard-table .overall-row td[data-col].col-hover {
|
|
476
|
+
box-shadow: inset 2px 0 0 var(--muted), inset -2px 0 0 var(--muted), inset 0 -2px 0 var(--muted);
|
|
477
|
+
}
|
|
478
|
+
.leaderboard-table tbody tr:not(.overall-row) td[data-col].col-hover {
|
|
479
|
+
box-shadow: inset 2px 0 0 var(--muted), inset -2px 0 0 var(--muted);
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
/* ───── Score cell ───── */
|
|
483
|
+
.score-wrap {
|
|
484
|
+
display: inline-flex;
|
|
485
|
+
flex-direction: column;
|
|
486
|
+
align-items: center;
|
|
487
|
+
gap: 6px;
|
|
488
|
+
min-width: 90px;
|
|
489
|
+
padding: 6px 10px;
|
|
490
|
+
border-radius: var(--radius-sm);
|
|
491
|
+
transition: background 0.15s ease;
|
|
492
|
+
}
|
|
493
|
+
.score-wrap.is-best {
|
|
494
|
+
background: var(--emerald-bg);
|
|
495
|
+
}
|
|
496
|
+
.score-number {
|
|
497
|
+
font-family: var(--font-mono);
|
|
498
|
+
font-size: 1rem;
|
|
499
|
+
font-weight: 500;
|
|
500
|
+
white-space: nowrap;
|
|
501
|
+
line-height: 1;
|
|
502
|
+
}
|
|
503
|
+
.score-wrap.is-best .score-number {
|
|
504
|
+
font-weight: 700;
|
|
505
|
+
font-size: 1.05rem;
|
|
506
|
+
}
|
|
507
|
+
.score-bar-track {
|
|
508
|
+
width: 100%;
|
|
509
|
+
height: 4px;
|
|
510
|
+
background: var(--border);
|
|
511
|
+
border-radius: 2px;
|
|
512
|
+
overflow: hidden;
|
|
513
|
+
}
|
|
514
|
+
.score-bar-fill {
|
|
515
|
+
height: 100%;
|
|
516
|
+
border-radius: 2px;
|
|
517
|
+
transition: width 0.4s ease;
|
|
518
|
+
}
|
|
519
|
+
.bar-emerald { background: var(--emerald); }
|
|
520
|
+
.bar-amber { background: var(--amber); }
|
|
521
|
+
.bar-red { background: var(--red); }
|
|
522
|
+
.score-badge {
|
|
523
|
+
font-size: 0.65rem;
|
|
524
|
+
font-weight: 700;
|
|
525
|
+
letter-spacing: 0.04em;
|
|
526
|
+
text-transform: uppercase;
|
|
527
|
+
color: var(--emerald);
|
|
528
|
+
line-height: 1;
|
|
529
|
+
}
|
|
530
|
+
.score-na {
|
|
531
|
+
color: var(--muted-light);
|
|
532
|
+
font-size: 0.8rem;
|
|
533
|
+
font-style: italic;
|
|
534
|
+
}
|
|
535
|
+
.color-emerald { color: var(--emerald); }
|
|
536
|
+
.color-amber { color: var(--amber); }
|
|
537
|
+
.color-red { color: var(--red); }
|
|
538
|
+
|
|
539
|
+
/* ───── Overall row ───── */
|
|
540
|
+
.overall-row td {
|
|
541
|
+
border-top: 2px solid var(--border);
|
|
542
|
+
border-bottom: none;
|
|
543
|
+
background: var(--cream);
|
|
544
|
+
padding-top: 20px;
|
|
545
|
+
padding-bottom: 20px;
|
|
546
|
+
}
|
|
547
|
+
.overall-row td:first-child {
|
|
548
|
+
background: var(--cream) !important;
|
|
549
|
+
border-right-color: var(--border);
|
|
550
|
+
}
|
|
551
|
+
.overall-row:hover td,
|
|
552
|
+
.overall-row:hover td:first-child {
|
|
553
|
+
background: #f3f1ec !important;
|
|
554
|
+
}
|
|
555
|
+
.overall-label {
|
|
556
|
+
font-family: var(--font-heading);
|
|
557
|
+
font-size: 1.1rem;
|
|
558
|
+
font-weight: 700;
|
|
559
|
+
color: var(--fg);
|
|
560
|
+
letter-spacing: -0.01em;
|
|
561
|
+
}
|
|
562
|
+
.overall-sublabel {
|
|
563
|
+
font-size: 0.7rem;
|
|
564
|
+
font-weight: 400;
|
|
565
|
+
color: var(--muted);
|
|
566
|
+
display: block;
|
|
567
|
+
margin-top: 2px;
|
|
568
|
+
}
|
|
569
|
+
/* Overall score cells */
|
|
570
|
+
.overall-row .score-wrap {
|
|
571
|
+
background: transparent;
|
|
572
|
+
}
|
|
573
|
+
.overall-row .score-wrap.is-best {
|
|
574
|
+
background: var(--emerald-bg);
|
|
575
|
+
}
|
|
576
|
+
.overall-row .score-number {
|
|
577
|
+
color: var(--fg);
|
|
578
|
+
}
|
|
579
|
+
.overall-row .score-wrap.is-best .score-number {
|
|
580
|
+
color: var(--emerald);
|
|
581
|
+
font-weight: 600;
|
|
582
|
+
}
|
|
583
|
+
.overall-row .score-bar-track {
|
|
584
|
+
background: var(--border);
|
|
585
|
+
}
|
|
586
|
+
.overall-row .score-badge {
|
|
587
|
+
color: var(--emerald);
|
|
588
|
+
}
|
|
589
|
+
.overall-row .score-na {
|
|
590
|
+
color: var(--muted-light);
|
|
591
|
+
}
|
|
592
|
+
|
|
593
|
+
@media (max-width: 768px) {
|
|
594
|
+
.report-container { padding: 20px 16px 48px; }
|
|
595
|
+
.report-header h1 { font-size: 1.8rem; }
|
|
596
|
+
}
|
|
597
|
+
"""
|
|
598
|
+
|
|
599
|
+
_HTML_BODY = """\
|
|
600
|
+
<body>
|
|
601
|
+
<div class="report-container">
|
|
602
|
+
<header class="report-header">
|
|
603
|
+
<h1>Benchmark Leaderboard</h1>
|
|
604
|
+
<p class="subtitle" id="subtitle"></p>
|
|
605
|
+
</header>
|
|
606
|
+
|
|
607
|
+
<div class="leaderboard-wrap">
|
|
608
|
+
<table class="leaderboard-table" id="leaderboard-table">
|
|
609
|
+
<thead id="table-head"></thead>
|
|
610
|
+
<tbody id="table-body"></tbody>
|
|
611
|
+
</table>
|
|
612
|
+
</div>
|
|
613
|
+
</div>
|
|
614
|
+
"""
|
|
615
|
+
|
|
616
|
+
_JS = """\
|
|
617
|
+
(function() {
|
|
618
|
+
function colorClass(rate) {
|
|
619
|
+
if (rate >= 80) return 'emerald';
|
|
620
|
+
if (rate >= 50) return 'amber';
|
|
621
|
+
return 'red';
|
|
622
|
+
}
|
|
623
|
+
|
|
624
|
+
function pct(val) { return val.toFixed(1) + '%'; }
|
|
625
|
+
|
|
626
|
+
function esc(s) {
|
|
627
|
+
if (s == null) return '';
|
|
628
|
+
var d = document.createElement('div');
|
|
629
|
+
d.textContent = String(s);
|
|
630
|
+
return d.innerHTML;
|
|
631
|
+
}
|
|
632
|
+
|
|
633
|
+
// ─── State ───
|
|
634
|
+
var selectedMetrics = {};
|
|
635
|
+
for (var cat in DATA.defaultMetrics) {
|
|
636
|
+
selectedMetrics[cat] = DATA.defaultMetrics[cat];
|
|
637
|
+
}
|
|
638
|
+
|
|
639
|
+
// Subtitle
|
|
640
|
+
document.getElementById('subtitle').textContent =
|
|
641
|
+
DATA.pipelines.length + ' pipeline' + (DATA.pipelines.length !== 1 ? 's' : '') +
|
|
642
|
+
' across ' + DATA.categories.length + ' categories';
|
|
643
|
+
|
|
644
|
+
// ─── Helpers ───
|
|
645
|
+
function getScore(category, pipelineName) {
|
|
646
|
+
var cs = DATA.scores[category];
|
|
647
|
+
if (!cs) return null;
|
|
648
|
+
var ps = cs[pipelineName];
|
|
649
|
+
if (!ps) return null;
|
|
650
|
+
var v = ps[selectedMetrics[category]];
|
|
651
|
+
return (v !== undefined && v !== null) ? v : null;
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
function findBest(category) {
|
|
655
|
+
var bestVal = -1, bestNames = [];
|
|
656
|
+
for (var i = 0; i < DATA.pipelines.length; i++) {
|
|
657
|
+
var v = getScore(category, DATA.pipelines[i].name);
|
|
658
|
+
if (v === null) continue;
|
|
659
|
+
if (v > bestVal) { bestVal = v; bestNames = [DATA.pipelines[i].name]; }
|
|
660
|
+
else if (v === bestVal) { bestNames.push(DATA.pipelines[i].name); }
|
|
661
|
+
}
|
|
662
|
+
return bestNames;
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
function getOverallScore(pipelineName) {
|
|
666
|
+
var sum = 0, count = 0;
|
|
667
|
+
for (var i = 0; i < DATA.categories.length; i++) {
|
|
668
|
+
var v = getScore(DATA.categories[i], pipelineName);
|
|
669
|
+
if (v !== null) { sum += v; count++; }
|
|
670
|
+
}
|
|
671
|
+
return count > 0 ? sum / count : null;
|
|
672
|
+
}
|
|
673
|
+
|
|
674
|
+
function getOverallWinners() {
|
|
675
|
+
var best = -1, names = [];
|
|
676
|
+
for (var i = 0; i < DATA.pipelines.length; i++) {
|
|
677
|
+
var v = getOverallScore(DATA.pipelines[i].name);
|
|
678
|
+
if (v === null) continue;
|
|
679
|
+
if (v > best) { best = v; names = [DATA.pipelines[i].name]; }
|
|
680
|
+
else if (v === best) { names.push(DATA.pipelines[i].name); }
|
|
681
|
+
}
|
|
682
|
+
return names;
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
function getMaxFiles(category) {
|
|
686
|
+
var files = DATA.categoryFiles[category] || {};
|
|
687
|
+
var max = 0;
|
|
688
|
+
for (var p in files) { if (files[p] > max) max = files[p]; }
|
|
689
|
+
return max;
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
// Extract a clean tier/model label from config
|
|
693
|
+
function getTierLabel(p) {
|
|
694
|
+
var cfg = p.config || {};
|
|
695
|
+
if (cfg.tier) return cfg.tier;
|
|
696
|
+
if (cfg.model) return cfg.model;
|
|
697
|
+
if (cfg.ocr_system) return cfg.ocr_system;
|
|
698
|
+
// Fallback: use first config value that's a short string
|
|
699
|
+
for (var k in cfg) {
|
|
700
|
+
var v = cfg[k];
|
|
701
|
+
if (typeof v === 'string' && v.length < 30) return v;
|
|
702
|
+
}
|
|
703
|
+
return p.productType || '';
|
|
704
|
+
}
|
|
705
|
+
|
|
706
|
+
// Build a score cell with progress bar
|
|
707
|
+
function buildScoreCell(v, isBest, isOverall) {
|
|
708
|
+
if (v === null) return '<span class="score-na">N/A</span>';
|
|
709
|
+
var pctVal = v * 100;
|
|
710
|
+
var c = colorClass(pctVal);
|
|
711
|
+
var cls = 'score-wrap' + (isBest ? ' is-best' : '');
|
|
712
|
+
var h = '<div class="' + cls + '">';
|
|
713
|
+
h += '<span class="score-number color-' + c + '">' + pct(pctVal) + '</span>';
|
|
714
|
+
h += '<div class="score-bar-track"><div class="score-bar-fill bar-' + c
|
|
715
|
+
+ '" style="width:' + Math.min(pctVal, 100).toFixed(1) + '%"></div></div>';
|
|
716
|
+
if (isBest) h += '<span class="score-badge">Best</span>';
|
|
717
|
+
h += '</div>';
|
|
718
|
+
return h;
|
|
719
|
+
}
|
|
720
|
+
|
|
721
|
+
// ─── Render ───
|
|
722
|
+
function renderHead() {
|
|
723
|
+
var winners = getOverallWinners();
|
|
724
|
+
var thead = document.getElementById('table-head');
|
|
725
|
+
var html = '<tr><th><span class="category-header-label">Category</span></th>';
|
|
726
|
+
for (var i = 0; i < DATA.pipelines.length; i++) {
|
|
727
|
+
var p = DATA.pipelines[i];
|
|
728
|
+
var isWinner = winners.indexOf(p.name) >= 0;
|
|
729
|
+
var tierLabel = getTierLabel(p);
|
|
730
|
+
html += '<th data-col="' + i + '" data-url="' + esc(p.dashboardUrl)
|
|
731
|
+
+ '"><div class="pipeline-header' + (isWinner ? ' is-winner' : '') + '">';
|
|
732
|
+
if (isWinner) html += '<span class="pipeline-crown">\\ud83d\\udc51</span>';
|
|
733
|
+
html += '<span class="pipeline-name">' + esc(p.displayName) + '</span>';
|
|
734
|
+
var sub = p.provider || '';
|
|
735
|
+
if (tierLabel && tierLabel !== p.provider) sub += (sub ? ' / ' : '') + tierLabel;
|
|
736
|
+
if (sub) html += '<span class="pipeline-tier">' + esc(sub) + '</span>';
|
|
737
|
+
html += '</div></th>';
|
|
738
|
+
}
|
|
739
|
+
html += '</tr>';
|
|
740
|
+
thead.innerHTML = html;
|
|
741
|
+
}
|
|
742
|
+
|
|
743
|
+
function renderBody() {
|
|
744
|
+
var tbody = document.getElementById('table-body');
|
|
745
|
+
var html = '';
|
|
746
|
+
|
|
747
|
+
// Category rows
|
|
748
|
+
for (var ci = 0; ci < DATA.categories.length; ci++) {
|
|
749
|
+
var cat = DATA.categories[ci];
|
|
750
|
+
var bestPipelines = findBest(cat);
|
|
751
|
+
var files = getMaxFiles(cat);
|
|
752
|
+
var metrics = DATA.categoryMetrics[cat] || [];
|
|
753
|
+
|
|
754
|
+
html += '<tr>';
|
|
755
|
+
html += '<td><div class="category-cell">';
|
|
756
|
+
html += '<span class="category-name">' + esc(DATA.categoryDisplayNames[cat] || cat);
|
|
757
|
+
html += ' <span class="file-count">' + files + ' files</span></span>';
|
|
758
|
+
html += '<select class="category-selector" data-cat="' + esc(cat) + '">';
|
|
759
|
+
for (var mi = 0; mi < metrics.length; mi++) {
|
|
760
|
+
var m = metrics[mi];
|
|
761
|
+
var sel = m === selectedMetrics[cat] ? ' selected' : '';
|
|
762
|
+
html += '<option value="' + esc(m) + '"' + sel + '>' + esc(DATA.metricNames[m] || m) + '</option>';
|
|
763
|
+
}
|
|
764
|
+
html += '</select></div></td>';
|
|
765
|
+
|
|
766
|
+
for (var pi = 0; pi < DATA.pipelines.length; pi++) {
|
|
767
|
+
var pName = DATA.pipelines[pi].name;
|
|
768
|
+
var v = getScore(cat, pName);
|
|
769
|
+
var isBest = bestPipelines.indexOf(pName) >= 0;
|
|
770
|
+
html += '<td data-col="' + pi + '" data-url="' + esc(DATA.pipelines[pi].dashboardUrl)
|
|
771
|
+
+ '">' + buildScoreCell(v, isBest, false) + '</td>';
|
|
772
|
+
}
|
|
773
|
+
html += '</tr>';
|
|
774
|
+
}
|
|
775
|
+
|
|
776
|
+
// Overall row
|
|
777
|
+
var overallWinners = getOverallWinners();
|
|
778
|
+
|
|
779
|
+
html += '<tr class="overall-row">';
|
|
780
|
+
html += '<td><span class="overall-label">Overall'
|
|
781
|
+
+ '<span class="overall-sublabel">Average across categories</span></span></td>';
|
|
782
|
+
for (var opi = 0; opi < DATA.pipelines.length; opi++) {
|
|
783
|
+
var opName = DATA.pipelines[opi].name;
|
|
784
|
+
var ov = getOverallScore(opName);
|
|
785
|
+
var oIsBest = overallWinners.indexOf(opName) >= 0;
|
|
786
|
+
html += '<td data-col="' + opi + '" data-url="' + esc(DATA.pipelines[opi].dashboardUrl)
|
|
787
|
+
+ '">' + buildScoreCell(ov, oIsBest, true) + '</td>';
|
|
788
|
+
}
|
|
789
|
+
html += '</tr>';
|
|
790
|
+
|
|
791
|
+
tbody.innerHTML = html;
|
|
792
|
+
|
|
793
|
+
// Bind dropdowns
|
|
794
|
+
var selects = tbody.querySelectorAll('.category-selector');
|
|
795
|
+
for (var si = 0; si < selects.length; si++) {
|
|
796
|
+
selects[si].addEventListener('change', function(e) {
|
|
797
|
+
selectedMetrics[e.target.getAttribute('data-cat')] = e.target.value;
|
|
798
|
+
render();
|
|
799
|
+
});
|
|
800
|
+
}
|
|
801
|
+
}
|
|
802
|
+
|
|
803
|
+
function bindColumnInteractions() {
|
|
804
|
+
var table = document.getElementById('leaderboard-table');
|
|
805
|
+
var lastCol = null;
|
|
806
|
+
|
|
807
|
+
function highlightCol(colIdx) {
|
|
808
|
+
if (colIdx === lastCol) return;
|
|
809
|
+
clearCol();
|
|
810
|
+
if (colIdx === null) return;
|
|
811
|
+
lastCol = colIdx;
|
|
812
|
+
var cells = table.querySelectorAll('[data-col="' + colIdx + '"]');
|
|
813
|
+
for (var i = 0; i < cells.length; i++) cells[i].classList.add('col-hover');
|
|
814
|
+
}
|
|
815
|
+
|
|
816
|
+
function clearCol() {
|
|
817
|
+
if (lastCol === null) return;
|
|
818
|
+
var cells = table.querySelectorAll('[data-col="' + lastCol + '"]');
|
|
819
|
+
for (var i = 0; i < cells.length; i++) cells[i].classList.remove('col-hover');
|
|
820
|
+
lastCol = null;
|
|
821
|
+
}
|
|
822
|
+
|
|
823
|
+
table.addEventListener('mouseover', function(e) {
|
|
824
|
+
var cell = e.target.closest('[data-col]');
|
|
825
|
+
if (cell) {
|
|
826
|
+
highlightCol(cell.getAttribute('data-col'));
|
|
827
|
+
}
|
|
828
|
+
});
|
|
829
|
+
|
|
830
|
+
table.addEventListener('mouseleave', function() {
|
|
831
|
+
clearCol();
|
|
832
|
+
});
|
|
833
|
+
|
|
834
|
+
table.addEventListener('click', function(e) {
|
|
835
|
+
// Don't navigate if clicking a dropdown
|
|
836
|
+
if (e.target.tagName === 'SELECT' || e.target.tagName === 'OPTION') return;
|
|
837
|
+
var cell = e.target.closest('[data-col]');
|
|
838
|
+
if (cell && cell.getAttribute('data-url')) {
|
|
839
|
+
window.location.href = cell.getAttribute('data-url');
|
|
840
|
+
}
|
|
841
|
+
});
|
|
842
|
+
}
|
|
843
|
+
|
|
844
|
+
function render() {
|
|
845
|
+
renderHead();
|
|
846
|
+
renderBody();
|
|
847
|
+
bindColumnInteractions();
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
render();
|
|
851
|
+
})();
|
|
852
|
+
"""
|