parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""LLM normalization for chart evaluation metrics.
|
|
2
|
+
|
|
3
|
+
Uses Claude LLM-as-judge to improve chart benchmark accuracy by handling
|
|
4
|
+
semantic equivalence that deterministic fuzzy matching misses.
|
|
5
|
+
|
|
6
|
+
Off by default. Controlled by env var ``LLAMACLOUD_BENCH_LLM_NORMALIZATION``:
|
|
7
|
+
- unset / ``"off"`` -- no LLM normalization (default; fully deterministic)
|
|
8
|
+
- ``"judge"`` -- opt-in Claude LLM-as-judge normalization
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from parse_bench.evaluation.metrics.parse.llm_normalization.base import (
|
|
12
|
+
BaseNormalizer,
|
|
13
|
+
JudgmentResult,
|
|
14
|
+
LabelMatch,
|
|
15
|
+
NormalizationResult,
|
|
16
|
+
ValueMatch,
|
|
17
|
+
)
|
|
18
|
+
from parse_bench.evaluation.metrics.parse.llm_normalization.config import (
|
|
19
|
+
NormalizationMode,
|
|
20
|
+
get_normalization_mode,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"BaseNormalizer",
|
|
25
|
+
"JudgmentResult",
|
|
26
|
+
"LabelMatch",
|
|
27
|
+
"NormalizationMode",
|
|
28
|
+
"NormalizationResult",
|
|
29
|
+
"ValueMatch",
|
|
30
|
+
"get_normalization_mode",
|
|
31
|
+
"get_normalizer",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def get_normalizer(
|
|
36
|
+
mode: NormalizationMode | None = None,
|
|
37
|
+
) -> BaseNormalizer | None:
|
|
38
|
+
"""Factory function that returns the appropriate normalizer for the given mode.
|
|
39
|
+
|
|
40
|
+
:param mode: Normalization mode. If None, reads from env var.
|
|
41
|
+
:return: A normalizer instance, or None if mode is OFF.
|
|
42
|
+
"""
|
|
43
|
+
if mode is None:
|
|
44
|
+
mode = get_normalization_mode()
|
|
45
|
+
|
|
46
|
+
if mode == NormalizationMode.JUDGE:
|
|
47
|
+
from parse_bench.evaluation.metrics.parse.llm_normalization.strategy_judge import JudgeNormalizer
|
|
48
|
+
|
|
49
|
+
return JudgeNormalizer()
|
|
50
|
+
|
|
51
|
+
return None
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Base classes and shared data structures for LLM normalization."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from abc import ABC, abstractmethod
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class LabelMatch:
|
|
11
|
+
"""Result of comparing an expected label to an actual label via LLM."""
|
|
12
|
+
|
|
13
|
+
expected: str
|
|
14
|
+
actual: str
|
|
15
|
+
is_match: bool
|
|
16
|
+
confidence: float
|
|
17
|
+
reasoning: str
|
|
18
|
+
strategy: str # "structured" or "judge"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class ValueMatch:
|
|
23
|
+
"""Result of comparing an expected value to an actual value via LLM."""
|
|
24
|
+
|
|
25
|
+
expected: str
|
|
26
|
+
actual: str
|
|
27
|
+
is_match: bool
|
|
28
|
+
normalized_expected: str
|
|
29
|
+
normalized_actual: str
|
|
30
|
+
reasoning: str
|
|
31
|
+
strategy: str # "structured" or "judge"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class JudgmentResult:
|
|
36
|
+
"""Result of a direct LLM-as-judge semantic equivalence check."""
|
|
37
|
+
|
|
38
|
+
expected: str
|
|
39
|
+
actual: str
|
|
40
|
+
is_equivalent: bool
|
|
41
|
+
confidence: float
|
|
42
|
+
reasoning: str
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class NormalizationResult:
|
|
47
|
+
"""Aggregated result from a normalization run."""
|
|
48
|
+
|
|
49
|
+
label_matches: list[LabelMatch] = field(default_factory=list)
|
|
50
|
+
value_matches: list[ValueMatch] = field(default_factory=list)
|
|
51
|
+
cost_usd: float = 0.0
|
|
52
|
+
latency_ms: float = 0.0
|
|
53
|
+
api_calls: int = 0
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class BaseNormalizer(ABC):
|
|
57
|
+
"""Abstract base class for LLM normalization strategies."""
|
|
58
|
+
|
|
59
|
+
@abstractmethod
|
|
60
|
+
def normalize_labels(
|
|
61
|
+
self,
|
|
62
|
+
expected_labels: list[str],
|
|
63
|
+
actual_labels: list[str],
|
|
64
|
+
context: str = "",
|
|
65
|
+
) -> list[LabelMatch]:
|
|
66
|
+
"""Determine semantic equivalence between pairs of expected/actual labels.
|
|
67
|
+
|
|
68
|
+
:param expected_labels: Ground truth column/row headers.
|
|
69
|
+
:param actual_labels: Model-produced headers (same length as expected_labels).
|
|
70
|
+
:param context: Optional context (test ID, chart description) for the LLM.
|
|
71
|
+
:return: List of LabelMatch results, one per pair.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
@abstractmethod
|
|
75
|
+
def normalize_values(
|
|
76
|
+
self,
|
|
77
|
+
expected_values: list[str],
|
|
78
|
+
actual_values: list[str],
|
|
79
|
+
context: str = "",
|
|
80
|
+
) -> list[ValueMatch]:
|
|
81
|
+
"""Normalize and compare pairs of expected/actual cell values.
|
|
82
|
+
|
|
83
|
+
:param expected_values: Ground truth cell values.
|
|
84
|
+
:param actual_values: Model-produced values (same length as expected_values).
|
|
85
|
+
:param context: Optional context for the LLM.
|
|
86
|
+
:return: List of ValueMatch results, one per pair.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
@abstractmethod
|
|
90
|
+
def normalize_data_point_labels(
|
|
91
|
+
self,
|
|
92
|
+
missing_labels: list[str],
|
|
93
|
+
table_headers: list[str],
|
|
94
|
+
context: str = "",
|
|
95
|
+
) -> list[LabelMatch]:
|
|
96
|
+
"""Assess whether missing data-point labels could match table headers.
|
|
97
|
+
|
|
98
|
+
Used by ChartDataPointRule when fuzzy matching fails to associate a
|
|
99
|
+
label with any row/column header.
|
|
100
|
+
|
|
101
|
+
:param missing_labels: Labels that fuzzy matching could not find.
|
|
102
|
+
:param table_headers: All headers present in the actual table.
|
|
103
|
+
:param context: Failure explanation or test ID for the LLM.
|
|
104
|
+
:return: List of LabelMatch results, one per missing label.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
@abstractmethod
|
|
109
|
+
def strategy_name(self) -> str:
|
|
110
|
+
"""Return the strategy identifier (e.g. 'structured' or 'judge')."""
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
@abstractmethod
|
|
114
|
+
def total_cost_usd(self) -> float:
|
|
115
|
+
"""Return cumulative USD cost of all API calls made by this normalizer."""
|
|
116
|
+
|
|
117
|
+
@property
|
|
118
|
+
@abstractmethod
|
|
119
|
+
def total_latency_ms(self) -> float:
|
|
120
|
+
"""Return cumulative latency in milliseconds."""
|
|
121
|
+
|
|
122
|
+
@property
|
|
123
|
+
@abstractmethod
|
|
124
|
+
def total_api_calls(self) -> int:
|
|
125
|
+
"""Return cumulative count of API calls."""
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Configuration for LLM normalization of chart evaluation metrics."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from enum import StrEnum
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class NormalizationMode(StrEnum):
|
|
10
|
+
"""LLM normalization mode (off by default; ``judge`` opts in)."""
|
|
11
|
+
|
|
12
|
+
OFF = "off"
|
|
13
|
+
JUDGE = "judge"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
# Anthropic model used for LLM-as-judge normalization
|
|
17
|
+
JUDGE_MODEL = "claude-haiku-4-5-20251001"
|
|
18
|
+
|
|
19
|
+
# Confidence threshold for accepting an LLM label match
|
|
20
|
+
LABEL_CONFIDENCE_THRESHOLD = 0.7
|
|
21
|
+
|
|
22
|
+
# Numeric tolerance for value comparison (relative, e.g. 0.02 = 2%)
|
|
23
|
+
VALUE_RELATIVE_TOLERANCE = 0.02
|
|
24
|
+
|
|
25
|
+
# Maximum number of value pairs to send per API call
|
|
26
|
+
VALUE_BATCH_SIZE = 30
|
|
27
|
+
|
|
28
|
+
# Safety limit: maximum API calls per normalizer instance.
|
|
29
|
+
# Prevents runaway costs on unexpectedly large datasets.
|
|
30
|
+
# A typical charts_core run (49 test cases) needs ~65 calls.
|
|
31
|
+
MAX_API_CALLS_PER_NORMALIZER = 500
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def get_normalization_mode() -> NormalizationMode:
|
|
35
|
+
"""Read LLAMACLOUD_BENCH_LLM_NORMALIZATION; OFF unless set to ``judge``."""
|
|
36
|
+
raw = os.environ.get("LLAMACLOUD_BENCH_LLM_NORMALIZATION", "").strip().lower()
|
|
37
|
+
if raw == NormalizationMode.JUDGE.value:
|
|
38
|
+
return NormalizationMode.JUDGE
|
|
39
|
+
return NormalizationMode.OFF
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def get_anthropic_api_key() -> str | None:
|
|
43
|
+
"""Return ANTHROPIC_API_KEY from environment, or None if not set."""
|
|
44
|
+
return os.environ.get("ANTHROPIC_API_KEY")
|
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
"""Post-processing LLM normalization for chart evaluation metrics.
|
|
2
|
+
|
|
3
|
+
Applies LLM normalization AFTER rules have already been evaluated, by parsing
|
|
4
|
+
the explanation strings from chart rule failures and re-evaluating them with
|
|
5
|
+
an LLM normalizer. This approach requires ZERO modifications to existing
|
|
6
|
+
test rules — it works entirely on the rule_results dicts produced by the
|
|
7
|
+
standard evaluation pipeline.
|
|
8
|
+
|
|
9
|
+
Off by default; opt in with ``LLAMACLOUD_BENCH_LLM_NORMALIZATION=judge``.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import logging
|
|
15
|
+
import re
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from parse_bench.evaluation.metrics.parse.llm_normalization.base import (
|
|
19
|
+
BaseNormalizer,
|
|
20
|
+
)
|
|
21
|
+
from parse_bench.evaluation.metrics.parse.llm_normalization.config import (
|
|
22
|
+
NormalizationMode,
|
|
23
|
+
get_normalization_mode,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger(__name__)
|
|
27
|
+
|
|
28
|
+
# Chart rule types eligible for LLM normalization
|
|
29
|
+
_CHART_RULE_TYPES = {
|
|
30
|
+
"chart_data_point",
|
|
31
|
+
"chart_data_array_labels",
|
|
32
|
+
"chart_data_array_data",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# ---------------------------------------------------------------------------
|
|
37
|
+
# Regex parsers (adapted from scripts/llm_normalization_prototype.py)
|
|
38
|
+
# ---------------------------------------------------------------------------
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _parse_label_pairs(explanation: str) -> list[tuple[str, str, float]]:
|
|
42
|
+
"""Extract (expected, actual, fuzzy_score) label pairs from explanation.
|
|
43
|
+
|
|
44
|
+
Matches patterns like: 'Entity' vs 'Year' (22%)
|
|
45
|
+
"""
|
|
46
|
+
pairs = []
|
|
47
|
+
for m in re.finditer(r"'([^']+)'\s+vs\s+'([^']+)'\s+\((\d+)%\)", explanation):
|
|
48
|
+
pairs.append((m.group(1), m.group(2), int(m.group(3)) / 100.0))
|
|
49
|
+
return pairs
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _parse_value_pairs(
|
|
53
|
+
explanation: str,
|
|
54
|
+
) -> list[tuple[list[str], list[str], float]]:
|
|
55
|
+
"""Extract (expected_vals, actual_vals, row_score) from data explanation rows.
|
|
56
|
+
|
|
57
|
+
Matches patterns like:
|
|
58
|
+
Row 1: 99% | Expected: ['Nigeria', '38.75'] | Actual: ['Nigeria', '39%']
|
|
59
|
+
"""
|
|
60
|
+
rows = []
|
|
61
|
+
for m in re.finditer(
|
|
62
|
+
r"Row \d+: (\d+)% \| Expected: \[([^\]]+)\] \| Actual: \[([^\]]+)\]",
|
|
63
|
+
explanation,
|
|
64
|
+
):
|
|
65
|
+
score = int(m.group(1)) / 100.0
|
|
66
|
+
exp = [v.strip().strip("'\"") for v in m.group(2).split(",")]
|
|
67
|
+
act = [v.strip().strip("'\"") for v in m.group(3).split(",")]
|
|
68
|
+
rows.append((exp, act, score))
|
|
69
|
+
return rows
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _parse_data_point_labels(explanation: str) -> list[str]:
|
|
73
|
+
"""Extract missing label names from data point failure explanation.
|
|
74
|
+
|
|
75
|
+
Matches patterns like: missing labels: ['number of major events']
|
|
76
|
+
"""
|
|
77
|
+
labels: list[str] = []
|
|
78
|
+
seen: set[str] = set()
|
|
79
|
+
for m in re.finditer(r"missing labels: \[([^\]]+)\]", explanation):
|
|
80
|
+
for label_m in re.finditer(r"'([^']+)'", m.group(1)):
|
|
81
|
+
label = label_m.group(1)
|
|
82
|
+
if label not in seen:
|
|
83
|
+
seen.add(label)
|
|
84
|
+
labels.append(label)
|
|
85
|
+
return labels
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _extract_score_parts(explanation: str) -> tuple[float | None, float | None]:
|
|
89
|
+
"""Extract (achieved, total) from score notation like '(3.61/4.00)'."""
|
|
90
|
+
m = re.search(r"\(([\d.]+)/([\d.]+)\)", explanation)
|
|
91
|
+
if m:
|
|
92
|
+
return float(m.group(1)), float(m.group(2))
|
|
93
|
+
return None, None
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _has_dimension_mismatch(explanation: str) -> bool:
|
|
97
|
+
"""Check if explanation contains a dimension mismatch (structural issue)."""
|
|
98
|
+
return bool(re.search(r"Dimension mismatch:", explanation))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# ---------------------------------------------------------------------------
|
|
102
|
+
# Per-rule-type normalization
|
|
103
|
+
# ---------------------------------------------------------------------------
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def _normalize_labels_rule(
|
|
107
|
+
result: dict[str, Any],
|
|
108
|
+
normalizer: BaseNormalizer,
|
|
109
|
+
) -> dict[str, Any] | None:
|
|
110
|
+
"""Try to upgrade a failed chart_data_array_labels rule via LLM."""
|
|
111
|
+
explanation = result.get("explanation", "")
|
|
112
|
+
pairs = _parse_label_pairs(explanation)
|
|
113
|
+
if not pairs:
|
|
114
|
+
return None
|
|
115
|
+
|
|
116
|
+
expected = [p[0] for p in pairs]
|
|
117
|
+
actual = [p[1] for p in pairs]
|
|
118
|
+
|
|
119
|
+
matches = normalizer.normalize_labels(expected, actual, context=explanation[:300])
|
|
120
|
+
|
|
121
|
+
upgraded = sum(1 for m in matches if m.is_match)
|
|
122
|
+
if upgraded == 0:
|
|
123
|
+
return None
|
|
124
|
+
|
|
125
|
+
# Recalculate score: each upgraded label pair was contributing its fuzzy
|
|
126
|
+
# score to achieved. Upgrading it to 1.0 adds (1.0 - fuzzy_score).
|
|
127
|
+
achieved, total = _extract_score_parts(explanation)
|
|
128
|
+
if achieved is not None and total is not None and total > 0:
|
|
129
|
+
improvement = sum(1.0 - pairs[i][2] for i, m in enumerate(matches) if m.is_match and i < len(pairs))
|
|
130
|
+
new_achieved = min(achieved + improvement, total)
|
|
131
|
+
new_score = new_achieved / total
|
|
132
|
+
else:
|
|
133
|
+
new_score = 1.0 if upgraded == len(pairs) else result["score"]
|
|
134
|
+
|
|
135
|
+
reasons = "; ".join(f"'{m.expected}'~'{m.actual}' ({m.reasoning})" for m in matches if m.is_match)
|
|
136
|
+
|
|
137
|
+
return {
|
|
138
|
+
**result,
|
|
139
|
+
"score": new_score,
|
|
140
|
+
"passed": new_score >= 1.0 - 1e-9,
|
|
141
|
+
"explanation": f"{explanation} [LLM normalized {upgraded}/{len(pairs)} labels: {reasons}]",
|
|
142
|
+
"llm_normalized": True,
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _normalize_data_rule(
|
|
147
|
+
result: dict[str, Any],
|
|
148
|
+
normalizer: BaseNormalizer,
|
|
149
|
+
) -> dict[str, Any] | None:
|
|
150
|
+
"""Try to upgrade a failed chart_data_array_data rule via LLM."""
|
|
151
|
+
explanation = result.get("explanation", "")
|
|
152
|
+
|
|
153
|
+
if _has_dimension_mismatch(explanation):
|
|
154
|
+
return None # Can't fix structural mismatches with LLM
|
|
155
|
+
|
|
156
|
+
rows = _parse_value_pairs(explanation)
|
|
157
|
+
if not rows:
|
|
158
|
+
return None
|
|
159
|
+
|
|
160
|
+
# Flatten mismatched values across all failing rows
|
|
161
|
+
all_exp: list[str] = []
|
|
162
|
+
all_act: list[str] = []
|
|
163
|
+
for exp_vals, act_vals, _row_score in rows:
|
|
164
|
+
for e, a in zip(exp_vals, act_vals, strict=False):
|
|
165
|
+
if e != a:
|
|
166
|
+
all_exp.append(e)
|
|
167
|
+
all_act.append(a)
|
|
168
|
+
|
|
169
|
+
if not all_exp:
|
|
170
|
+
return None
|
|
171
|
+
|
|
172
|
+
matches = normalizer.normalize_values(all_exp, all_act, context=explanation[:300])
|
|
173
|
+
|
|
174
|
+
upgraded = sum(1 for m in matches if m.is_match)
|
|
175
|
+
if upgraded == 0:
|
|
176
|
+
return None
|
|
177
|
+
|
|
178
|
+
# Each upgraded value adds ~1 matched cell to the achieved count
|
|
179
|
+
achieved, total = _extract_score_parts(explanation)
|
|
180
|
+
if achieved is not None and total is not None and total > 0:
|
|
181
|
+
new_score = min((achieved + upgraded) / total, 1.0)
|
|
182
|
+
else:
|
|
183
|
+
new_score = 1.0 if upgraded == len(all_exp) else result["score"]
|
|
184
|
+
|
|
185
|
+
reasons = "; ".join(f"'{m.expected}'~'{m.actual}' ({m.reasoning})" for m in matches if m.is_match)
|
|
186
|
+
|
|
187
|
+
return {
|
|
188
|
+
**result,
|
|
189
|
+
"score": new_score,
|
|
190
|
+
"passed": new_score >= 1.0 - 1e-9,
|
|
191
|
+
"explanation": (f"{explanation} [LLM normalized {upgraded}/{len(all_exp)} values: {reasons}]"),
|
|
192
|
+
"llm_normalized": True,
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _normalize_data_point_rule(
|
|
197
|
+
result: dict[str, Any],
|
|
198
|
+
normalizer: BaseNormalizer,
|
|
199
|
+
) -> dict[str, Any] | None:
|
|
200
|
+
"""Try to upgrade a failed chart_data_point rule via LLM."""
|
|
201
|
+
explanation = result.get("explanation", "")
|
|
202
|
+
|
|
203
|
+
missing = _parse_data_point_labels(explanation)
|
|
204
|
+
if not missing:
|
|
205
|
+
return None
|
|
206
|
+
|
|
207
|
+
# We don't have direct access to table headers from the explanation,
|
|
208
|
+
# but the normalizer can work with the explanation context alone.
|
|
209
|
+
matches = normalizer.normalize_data_point_labels(missing, table_headers=[], context=explanation[:500])
|
|
210
|
+
|
|
211
|
+
upgraded = sum(1 for m in matches if m.is_match)
|
|
212
|
+
if upgraded == 0:
|
|
213
|
+
return None
|
|
214
|
+
|
|
215
|
+
all_matched = upgraded == len(missing)
|
|
216
|
+
|
|
217
|
+
reasons = "; ".join(f"'{m.expected}' matched ({m.reasoning})" for m in matches if m.is_match)
|
|
218
|
+
|
|
219
|
+
return {
|
|
220
|
+
**result,
|
|
221
|
+
"score": 1.0 if all_matched else result["score"],
|
|
222
|
+
"passed": all_matched,
|
|
223
|
+
"explanation": f"{explanation} [LLM matched {upgraded}/{len(missing)} labels: {reasons}]",
|
|
224
|
+
"llm_normalized": True,
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
# ---------------------------------------------------------------------------
|
|
229
|
+
# Core normalization pipeline
|
|
230
|
+
# ---------------------------------------------------------------------------
|
|
231
|
+
|
|
232
|
+
_RULE_HANDLERS = {
|
|
233
|
+
"chart_data_array_labels": _normalize_labels_rule,
|
|
234
|
+
"chart_data_array_data": _normalize_data_rule,
|
|
235
|
+
"chart_data_point": _normalize_data_point_rule,
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _normalizer_stats(normalizer: BaseNormalizer) -> dict[str, Any]:
|
|
240
|
+
"""Extract cost/latency/api_calls stats from a normalizer."""
|
|
241
|
+
return {
|
|
242
|
+
"strategy": normalizer.strategy_name,
|
|
243
|
+
"cost_usd": normalizer.total_cost_usd,
|
|
244
|
+
"latency_ms": normalizer.total_latency_ms,
|
|
245
|
+
"api_calls": normalizer.total_api_calls,
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _apply_normalization(
|
|
250
|
+
rule_results: list[dict[str, Any]],
|
|
251
|
+
normalizer: BaseNormalizer,
|
|
252
|
+
) -> list[dict[str, Any]]:
|
|
253
|
+
"""Apply LLM normalization to chart rule failures, returning updated list."""
|
|
254
|
+
updated: list[dict[str, Any]] = []
|
|
255
|
+
for result in rule_results:
|
|
256
|
+
rtype = result.get("type", "")
|
|
257
|
+
|
|
258
|
+
# Only process failed chart rules
|
|
259
|
+
if rtype not in _CHART_RULE_TYPES or result.get("passed", True):
|
|
260
|
+
updated.append(result)
|
|
261
|
+
continue
|
|
262
|
+
|
|
263
|
+
handler = _RULE_HANDLERS.get(rtype)
|
|
264
|
+
normalized = handler(result, normalizer) if handler else None
|
|
265
|
+
updated.append(normalized if normalized is not None else result)
|
|
266
|
+
|
|
267
|
+
return updated
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
# ---------------------------------------------------------------------------
|
|
271
|
+
# Public API
|
|
272
|
+
# ---------------------------------------------------------------------------
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def maybe_apply_llm_normalization(
|
|
276
|
+
rule_results: list[dict[str, Any]],
|
|
277
|
+
) -> tuple[list[dict[str, Any]], dict[str, Any] | None]:
|
|
278
|
+
"""Post-process chart rule results with LLM normalization.
|
|
279
|
+
|
|
280
|
+
Returns ``(updated_results, llm_metadata_or_None)``.
|
|
281
|
+
When mode is OFF or initialization fails, returns results unchanged
|
|
282
|
+
with ``None`` metadata.
|
|
283
|
+
"""
|
|
284
|
+
# Lazy import to avoid circular dependency (__init__.py defines get_normalizer)
|
|
285
|
+
from parse_bench.evaluation.metrics.parse.llm_normalization import (
|
|
286
|
+
get_normalizer,
|
|
287
|
+
)
|
|
288
|
+
|
|
289
|
+
mode = get_normalization_mode()
|
|
290
|
+
if mode == NormalizationMode.OFF:
|
|
291
|
+
return rule_results, None
|
|
292
|
+
|
|
293
|
+
try:
|
|
294
|
+
normalizer_or_list = get_normalizer(mode)
|
|
295
|
+
except Exception:
|
|
296
|
+
logger.warning(
|
|
297
|
+
"Failed to initialize LLM normalizer (mode=%s), skipping",
|
|
298
|
+
mode.value,
|
|
299
|
+
exc_info=True,
|
|
300
|
+
)
|
|
301
|
+
return rule_results, None
|
|
302
|
+
|
|
303
|
+
if normalizer_or_list is None:
|
|
304
|
+
return rule_results, None
|
|
305
|
+
|
|
306
|
+
n_failures = sum(1 for r in rule_results if r.get("type", "") in _CHART_RULE_TYPES and not r.get("passed", True))
|
|
307
|
+
logger.info(
|
|
308
|
+
"LLM normalization mode=%s, post-processing %d chart failures out of %d rules",
|
|
309
|
+
mode.value,
|
|
310
|
+
n_failures,
|
|
311
|
+
len(rule_results),
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
if not isinstance(normalizer_or_list, BaseNormalizer):
|
|
315
|
+
return rule_results, None
|
|
316
|
+
|
|
317
|
+
normalizer = normalizer_or_list
|
|
318
|
+
|
|
319
|
+
logger.info("Post-processing with normalizer: %s", normalizer.strategy_name)
|
|
320
|
+
updated = _apply_normalization(rule_results, normalizer)
|
|
321
|
+
|
|
322
|
+
return updated, {"mode": mode.value, **_normalizer_stats(normalizer)}
|