parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,51 @@
1
+ """LLM normalization for chart evaluation metrics.
2
+
3
+ Uses Claude LLM-as-judge to improve chart benchmark accuracy by handling
4
+ semantic equivalence that deterministic fuzzy matching misses.
5
+
6
+ Off by default. Controlled by env var ``LLAMACLOUD_BENCH_LLM_NORMALIZATION``:
7
+ - unset / ``"off"`` -- no LLM normalization (default; fully deterministic)
8
+ - ``"judge"`` -- opt-in Claude LLM-as-judge normalization
9
+ """
10
+
11
+ from parse_bench.evaluation.metrics.parse.llm_normalization.base import (
12
+ BaseNormalizer,
13
+ JudgmentResult,
14
+ LabelMatch,
15
+ NormalizationResult,
16
+ ValueMatch,
17
+ )
18
+ from parse_bench.evaluation.metrics.parse.llm_normalization.config import (
19
+ NormalizationMode,
20
+ get_normalization_mode,
21
+ )
22
+
23
+ __all__ = [
24
+ "BaseNormalizer",
25
+ "JudgmentResult",
26
+ "LabelMatch",
27
+ "NormalizationMode",
28
+ "NormalizationResult",
29
+ "ValueMatch",
30
+ "get_normalization_mode",
31
+ "get_normalizer",
32
+ ]
33
+
34
+
35
+ def get_normalizer(
36
+ mode: NormalizationMode | None = None,
37
+ ) -> BaseNormalizer | None:
38
+ """Factory function that returns the appropriate normalizer for the given mode.
39
+
40
+ :param mode: Normalization mode. If None, reads from env var.
41
+ :return: A normalizer instance, or None if mode is OFF.
42
+ """
43
+ if mode is None:
44
+ mode = get_normalization_mode()
45
+
46
+ if mode == NormalizationMode.JUDGE:
47
+ from parse_bench.evaluation.metrics.parse.llm_normalization.strategy_judge import JudgeNormalizer
48
+
49
+ return JudgeNormalizer()
50
+
51
+ return None
@@ -0,0 +1,125 @@
1
+ """Base classes and shared data structures for LLM normalization."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from abc import ABC, abstractmethod
6
+ from dataclasses import dataclass, field
7
+
8
+
9
+ @dataclass
10
+ class LabelMatch:
11
+ """Result of comparing an expected label to an actual label via LLM."""
12
+
13
+ expected: str
14
+ actual: str
15
+ is_match: bool
16
+ confidence: float
17
+ reasoning: str
18
+ strategy: str # "structured" or "judge"
19
+
20
+
21
+ @dataclass
22
+ class ValueMatch:
23
+ """Result of comparing an expected value to an actual value via LLM."""
24
+
25
+ expected: str
26
+ actual: str
27
+ is_match: bool
28
+ normalized_expected: str
29
+ normalized_actual: str
30
+ reasoning: str
31
+ strategy: str # "structured" or "judge"
32
+
33
+
34
+ @dataclass
35
+ class JudgmentResult:
36
+ """Result of a direct LLM-as-judge semantic equivalence check."""
37
+
38
+ expected: str
39
+ actual: str
40
+ is_equivalent: bool
41
+ confidence: float
42
+ reasoning: str
43
+
44
+
45
+ @dataclass
46
+ class NormalizationResult:
47
+ """Aggregated result from a normalization run."""
48
+
49
+ label_matches: list[LabelMatch] = field(default_factory=list)
50
+ value_matches: list[ValueMatch] = field(default_factory=list)
51
+ cost_usd: float = 0.0
52
+ latency_ms: float = 0.0
53
+ api_calls: int = 0
54
+
55
+
56
+ class BaseNormalizer(ABC):
57
+ """Abstract base class for LLM normalization strategies."""
58
+
59
+ @abstractmethod
60
+ def normalize_labels(
61
+ self,
62
+ expected_labels: list[str],
63
+ actual_labels: list[str],
64
+ context: str = "",
65
+ ) -> list[LabelMatch]:
66
+ """Determine semantic equivalence between pairs of expected/actual labels.
67
+
68
+ :param expected_labels: Ground truth column/row headers.
69
+ :param actual_labels: Model-produced headers (same length as expected_labels).
70
+ :param context: Optional context (test ID, chart description) for the LLM.
71
+ :return: List of LabelMatch results, one per pair.
72
+ """
73
+
74
+ @abstractmethod
75
+ def normalize_values(
76
+ self,
77
+ expected_values: list[str],
78
+ actual_values: list[str],
79
+ context: str = "",
80
+ ) -> list[ValueMatch]:
81
+ """Normalize and compare pairs of expected/actual cell values.
82
+
83
+ :param expected_values: Ground truth cell values.
84
+ :param actual_values: Model-produced values (same length as expected_values).
85
+ :param context: Optional context for the LLM.
86
+ :return: List of ValueMatch results, one per pair.
87
+ """
88
+
89
+ @abstractmethod
90
+ def normalize_data_point_labels(
91
+ self,
92
+ missing_labels: list[str],
93
+ table_headers: list[str],
94
+ context: str = "",
95
+ ) -> list[LabelMatch]:
96
+ """Assess whether missing data-point labels could match table headers.
97
+
98
+ Used by ChartDataPointRule when fuzzy matching fails to associate a
99
+ label with any row/column header.
100
+
101
+ :param missing_labels: Labels that fuzzy matching could not find.
102
+ :param table_headers: All headers present in the actual table.
103
+ :param context: Failure explanation or test ID for the LLM.
104
+ :return: List of LabelMatch results, one per missing label.
105
+ """
106
+
107
+ @property
108
+ @abstractmethod
109
+ def strategy_name(self) -> str:
110
+ """Return the strategy identifier (e.g. 'structured' or 'judge')."""
111
+
112
+ @property
113
+ @abstractmethod
114
+ def total_cost_usd(self) -> float:
115
+ """Return cumulative USD cost of all API calls made by this normalizer."""
116
+
117
+ @property
118
+ @abstractmethod
119
+ def total_latency_ms(self) -> float:
120
+ """Return cumulative latency in milliseconds."""
121
+
122
+ @property
123
+ @abstractmethod
124
+ def total_api_calls(self) -> int:
125
+ """Return cumulative count of API calls."""
@@ -0,0 +1,44 @@
1
+ """Configuration for LLM normalization of chart evaluation metrics."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ from enum import StrEnum
7
+
8
+
9
+ class NormalizationMode(StrEnum):
10
+ """LLM normalization mode (off by default; ``judge`` opts in)."""
11
+
12
+ OFF = "off"
13
+ JUDGE = "judge"
14
+
15
+
16
+ # Anthropic model used for LLM-as-judge normalization
17
+ JUDGE_MODEL = "claude-haiku-4-5-20251001"
18
+
19
+ # Confidence threshold for accepting an LLM label match
20
+ LABEL_CONFIDENCE_THRESHOLD = 0.7
21
+
22
+ # Numeric tolerance for value comparison (relative, e.g. 0.02 = 2%)
23
+ VALUE_RELATIVE_TOLERANCE = 0.02
24
+
25
+ # Maximum number of value pairs to send per API call
26
+ VALUE_BATCH_SIZE = 30
27
+
28
+ # Safety limit: maximum API calls per normalizer instance.
29
+ # Prevents runaway costs on unexpectedly large datasets.
30
+ # A typical charts_core run (49 test cases) needs ~65 calls.
31
+ MAX_API_CALLS_PER_NORMALIZER = 500
32
+
33
+
34
+ def get_normalization_mode() -> NormalizationMode:
35
+ """Read LLAMACLOUD_BENCH_LLM_NORMALIZATION; OFF unless set to ``judge``."""
36
+ raw = os.environ.get("LLAMACLOUD_BENCH_LLM_NORMALIZATION", "").strip().lower()
37
+ if raw == NormalizationMode.JUDGE.value:
38
+ return NormalizationMode.JUDGE
39
+ return NormalizationMode.OFF
40
+
41
+
42
+ def get_anthropic_api_key() -> str | None:
43
+ """Return ANTHROPIC_API_KEY from environment, or None if not set."""
44
+ return os.environ.get("ANTHROPIC_API_KEY")
@@ -0,0 +1,322 @@
1
+ """Post-processing LLM normalization for chart evaluation metrics.
2
+
3
+ Applies LLM normalization AFTER rules have already been evaluated, by parsing
4
+ the explanation strings from chart rule failures and re-evaluating them with
5
+ an LLM normalizer. This approach requires ZERO modifications to existing
6
+ test rules — it works entirely on the rule_results dicts produced by the
7
+ standard evaluation pipeline.
8
+
9
+ Off by default; opt in with ``LLAMACLOUD_BENCH_LLM_NORMALIZATION=judge``.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import logging
15
+ import re
16
+ from typing import Any
17
+
18
+ from parse_bench.evaluation.metrics.parse.llm_normalization.base import (
19
+ BaseNormalizer,
20
+ )
21
+ from parse_bench.evaluation.metrics.parse.llm_normalization.config import (
22
+ NormalizationMode,
23
+ get_normalization_mode,
24
+ )
25
+
26
+ logger = logging.getLogger(__name__)
27
+
28
+ # Chart rule types eligible for LLM normalization
29
+ _CHART_RULE_TYPES = {
30
+ "chart_data_point",
31
+ "chart_data_array_labels",
32
+ "chart_data_array_data",
33
+ }
34
+
35
+
36
+ # ---------------------------------------------------------------------------
37
+ # Regex parsers (adapted from scripts/llm_normalization_prototype.py)
38
+ # ---------------------------------------------------------------------------
39
+
40
+
41
+ def _parse_label_pairs(explanation: str) -> list[tuple[str, str, float]]:
42
+ """Extract (expected, actual, fuzzy_score) label pairs from explanation.
43
+
44
+ Matches patterns like: 'Entity' vs 'Year' (22%)
45
+ """
46
+ pairs = []
47
+ for m in re.finditer(r"'([^']+)'\s+vs\s+'([^']+)'\s+\((\d+)%\)", explanation):
48
+ pairs.append((m.group(1), m.group(2), int(m.group(3)) / 100.0))
49
+ return pairs
50
+
51
+
52
+ def _parse_value_pairs(
53
+ explanation: str,
54
+ ) -> list[tuple[list[str], list[str], float]]:
55
+ """Extract (expected_vals, actual_vals, row_score) from data explanation rows.
56
+
57
+ Matches patterns like:
58
+ Row 1: 99% | Expected: ['Nigeria', '38.75'] | Actual: ['Nigeria', '39%']
59
+ """
60
+ rows = []
61
+ for m in re.finditer(
62
+ r"Row \d+: (\d+)% \| Expected: \[([^\]]+)\] \| Actual: \[([^\]]+)\]",
63
+ explanation,
64
+ ):
65
+ score = int(m.group(1)) / 100.0
66
+ exp = [v.strip().strip("'\"") for v in m.group(2).split(",")]
67
+ act = [v.strip().strip("'\"") for v in m.group(3).split(",")]
68
+ rows.append((exp, act, score))
69
+ return rows
70
+
71
+
72
+ def _parse_data_point_labels(explanation: str) -> list[str]:
73
+ """Extract missing label names from data point failure explanation.
74
+
75
+ Matches patterns like: missing labels: ['number of major events']
76
+ """
77
+ labels: list[str] = []
78
+ seen: set[str] = set()
79
+ for m in re.finditer(r"missing labels: \[([^\]]+)\]", explanation):
80
+ for label_m in re.finditer(r"'([^']+)'", m.group(1)):
81
+ label = label_m.group(1)
82
+ if label not in seen:
83
+ seen.add(label)
84
+ labels.append(label)
85
+ return labels
86
+
87
+
88
+ def _extract_score_parts(explanation: str) -> tuple[float | None, float | None]:
89
+ """Extract (achieved, total) from score notation like '(3.61/4.00)'."""
90
+ m = re.search(r"\(([\d.]+)/([\d.]+)\)", explanation)
91
+ if m:
92
+ return float(m.group(1)), float(m.group(2))
93
+ return None, None
94
+
95
+
96
+ def _has_dimension_mismatch(explanation: str) -> bool:
97
+ """Check if explanation contains a dimension mismatch (structural issue)."""
98
+ return bool(re.search(r"Dimension mismatch:", explanation))
99
+
100
+
101
+ # ---------------------------------------------------------------------------
102
+ # Per-rule-type normalization
103
+ # ---------------------------------------------------------------------------
104
+
105
+
106
+ def _normalize_labels_rule(
107
+ result: dict[str, Any],
108
+ normalizer: BaseNormalizer,
109
+ ) -> dict[str, Any] | None:
110
+ """Try to upgrade a failed chart_data_array_labels rule via LLM."""
111
+ explanation = result.get("explanation", "")
112
+ pairs = _parse_label_pairs(explanation)
113
+ if not pairs:
114
+ return None
115
+
116
+ expected = [p[0] for p in pairs]
117
+ actual = [p[1] for p in pairs]
118
+
119
+ matches = normalizer.normalize_labels(expected, actual, context=explanation[:300])
120
+
121
+ upgraded = sum(1 for m in matches if m.is_match)
122
+ if upgraded == 0:
123
+ return None
124
+
125
+ # Recalculate score: each upgraded label pair was contributing its fuzzy
126
+ # score to achieved. Upgrading it to 1.0 adds (1.0 - fuzzy_score).
127
+ achieved, total = _extract_score_parts(explanation)
128
+ if achieved is not None and total is not None and total > 0:
129
+ improvement = sum(1.0 - pairs[i][2] for i, m in enumerate(matches) if m.is_match and i < len(pairs))
130
+ new_achieved = min(achieved + improvement, total)
131
+ new_score = new_achieved / total
132
+ else:
133
+ new_score = 1.0 if upgraded == len(pairs) else result["score"]
134
+
135
+ reasons = "; ".join(f"'{m.expected}'~'{m.actual}' ({m.reasoning})" for m in matches if m.is_match)
136
+
137
+ return {
138
+ **result,
139
+ "score": new_score,
140
+ "passed": new_score >= 1.0 - 1e-9,
141
+ "explanation": f"{explanation} [LLM normalized {upgraded}/{len(pairs)} labels: {reasons}]",
142
+ "llm_normalized": True,
143
+ }
144
+
145
+
146
+ def _normalize_data_rule(
147
+ result: dict[str, Any],
148
+ normalizer: BaseNormalizer,
149
+ ) -> dict[str, Any] | None:
150
+ """Try to upgrade a failed chart_data_array_data rule via LLM."""
151
+ explanation = result.get("explanation", "")
152
+
153
+ if _has_dimension_mismatch(explanation):
154
+ return None # Can't fix structural mismatches with LLM
155
+
156
+ rows = _parse_value_pairs(explanation)
157
+ if not rows:
158
+ return None
159
+
160
+ # Flatten mismatched values across all failing rows
161
+ all_exp: list[str] = []
162
+ all_act: list[str] = []
163
+ for exp_vals, act_vals, _row_score in rows:
164
+ for e, a in zip(exp_vals, act_vals, strict=False):
165
+ if e != a:
166
+ all_exp.append(e)
167
+ all_act.append(a)
168
+
169
+ if not all_exp:
170
+ return None
171
+
172
+ matches = normalizer.normalize_values(all_exp, all_act, context=explanation[:300])
173
+
174
+ upgraded = sum(1 for m in matches if m.is_match)
175
+ if upgraded == 0:
176
+ return None
177
+
178
+ # Each upgraded value adds ~1 matched cell to the achieved count
179
+ achieved, total = _extract_score_parts(explanation)
180
+ if achieved is not None and total is not None and total > 0:
181
+ new_score = min((achieved + upgraded) / total, 1.0)
182
+ else:
183
+ new_score = 1.0 if upgraded == len(all_exp) else result["score"]
184
+
185
+ reasons = "; ".join(f"'{m.expected}'~'{m.actual}' ({m.reasoning})" for m in matches if m.is_match)
186
+
187
+ return {
188
+ **result,
189
+ "score": new_score,
190
+ "passed": new_score >= 1.0 - 1e-9,
191
+ "explanation": (f"{explanation} [LLM normalized {upgraded}/{len(all_exp)} values: {reasons}]"),
192
+ "llm_normalized": True,
193
+ }
194
+
195
+
196
+ def _normalize_data_point_rule(
197
+ result: dict[str, Any],
198
+ normalizer: BaseNormalizer,
199
+ ) -> dict[str, Any] | None:
200
+ """Try to upgrade a failed chart_data_point rule via LLM."""
201
+ explanation = result.get("explanation", "")
202
+
203
+ missing = _parse_data_point_labels(explanation)
204
+ if not missing:
205
+ return None
206
+
207
+ # We don't have direct access to table headers from the explanation,
208
+ # but the normalizer can work with the explanation context alone.
209
+ matches = normalizer.normalize_data_point_labels(missing, table_headers=[], context=explanation[:500])
210
+
211
+ upgraded = sum(1 for m in matches if m.is_match)
212
+ if upgraded == 0:
213
+ return None
214
+
215
+ all_matched = upgraded == len(missing)
216
+
217
+ reasons = "; ".join(f"'{m.expected}' matched ({m.reasoning})" for m in matches if m.is_match)
218
+
219
+ return {
220
+ **result,
221
+ "score": 1.0 if all_matched else result["score"],
222
+ "passed": all_matched,
223
+ "explanation": f"{explanation} [LLM matched {upgraded}/{len(missing)} labels: {reasons}]",
224
+ "llm_normalized": True,
225
+ }
226
+
227
+
228
+ # ---------------------------------------------------------------------------
229
+ # Core normalization pipeline
230
+ # ---------------------------------------------------------------------------
231
+
232
+ _RULE_HANDLERS = {
233
+ "chart_data_array_labels": _normalize_labels_rule,
234
+ "chart_data_array_data": _normalize_data_rule,
235
+ "chart_data_point": _normalize_data_point_rule,
236
+ }
237
+
238
+
239
+ def _normalizer_stats(normalizer: BaseNormalizer) -> dict[str, Any]:
240
+ """Extract cost/latency/api_calls stats from a normalizer."""
241
+ return {
242
+ "strategy": normalizer.strategy_name,
243
+ "cost_usd": normalizer.total_cost_usd,
244
+ "latency_ms": normalizer.total_latency_ms,
245
+ "api_calls": normalizer.total_api_calls,
246
+ }
247
+
248
+
249
+ def _apply_normalization(
250
+ rule_results: list[dict[str, Any]],
251
+ normalizer: BaseNormalizer,
252
+ ) -> list[dict[str, Any]]:
253
+ """Apply LLM normalization to chart rule failures, returning updated list."""
254
+ updated: list[dict[str, Any]] = []
255
+ for result in rule_results:
256
+ rtype = result.get("type", "")
257
+
258
+ # Only process failed chart rules
259
+ if rtype not in _CHART_RULE_TYPES or result.get("passed", True):
260
+ updated.append(result)
261
+ continue
262
+
263
+ handler = _RULE_HANDLERS.get(rtype)
264
+ normalized = handler(result, normalizer) if handler else None
265
+ updated.append(normalized if normalized is not None else result)
266
+
267
+ return updated
268
+
269
+
270
+ # ---------------------------------------------------------------------------
271
+ # Public API
272
+ # ---------------------------------------------------------------------------
273
+
274
+
275
+ def maybe_apply_llm_normalization(
276
+ rule_results: list[dict[str, Any]],
277
+ ) -> tuple[list[dict[str, Any]], dict[str, Any] | None]:
278
+ """Post-process chart rule results with LLM normalization.
279
+
280
+ Returns ``(updated_results, llm_metadata_or_None)``.
281
+ When mode is OFF or initialization fails, returns results unchanged
282
+ with ``None`` metadata.
283
+ """
284
+ # Lazy import to avoid circular dependency (__init__.py defines get_normalizer)
285
+ from parse_bench.evaluation.metrics.parse.llm_normalization import (
286
+ get_normalizer,
287
+ )
288
+
289
+ mode = get_normalization_mode()
290
+ if mode == NormalizationMode.OFF:
291
+ return rule_results, None
292
+
293
+ try:
294
+ normalizer_or_list = get_normalizer(mode)
295
+ except Exception:
296
+ logger.warning(
297
+ "Failed to initialize LLM normalizer (mode=%s), skipping",
298
+ mode.value,
299
+ exc_info=True,
300
+ )
301
+ return rule_results, None
302
+
303
+ if normalizer_or_list is None:
304
+ return rule_results, None
305
+
306
+ n_failures = sum(1 for r in rule_results if r.get("type", "") in _CHART_RULE_TYPES and not r.get("passed", True))
307
+ logger.info(
308
+ "LLM normalization mode=%s, post-processing %d chart failures out of %d rules",
309
+ mode.value,
310
+ n_failures,
311
+ len(rule_results),
312
+ )
313
+
314
+ if not isinstance(normalizer_or_list, BaseNormalizer):
315
+ return rule_results, None
316
+
317
+ normalizer = normalizer_or_list
318
+
319
+ logger.info("Post-processing with normalizer: %s", normalizer.strategy_name)
320
+ updated = _apply_normalization(rule_results, normalizer)
321
+
322
+ return updated, {"mode": mode.value, **_normalizer_stats(normalizer)}