parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,276 @@
1
+ """Page-decoration rule: running header, running footer and printed page number, per page.
2
+
3
+ LlamaParse lifts page furniture out of the body markdown into per-page structured fields
4
+ (``page_header_markdown``, ``page_footer_markdown``, ``printed_page_number``); the prompts ask
5
+ for ``<page_header>`` / ``<page_footer>`` / ``<page_number>`` tags which post-processing removes.
6
+ This rule scores that contract on one page with four axes:
7
+
8
+ * ``header`` / ``footer`` — predicted text vs annotated text after normalisation (markdown and
9
+ punctuation stripped, whitespace collapsed, lowercase) with ``token_set_ratio`` so that the
10
+ left / centre / right pieces of a header may come in any order. An annotated ``None`` means
11
+ the slot must be empty: a hallucinated header on a cover page fails.
12
+ * ``page_number`` — both sides canonicalised (arabic digits, lowercase roman, ``Page 3 of 10`` →
13
+ ``3``, ``– 3 –`` → ``3``, compound ids such as ``A-3`` kept) and compared exactly; a number returned
14
+ inside the header/footer text instead of the dedicated field also counts.
15
+ * ``leak`` — none of the annotated furniture strings may remain in the body markdown; the body
16
+ is the page markdown with any ``<page_*>`` tags removed.
17
+
18
+ Prediction source: the structured page fields when ``parse_output.layout_pages`` is available,
19
+ else the markdown tags — the same preference order as the ``is_header`` / ``is_footer`` rules.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import re
25
+ from typing import Any, cast
26
+
27
+ import numpy as np
28
+ from rapidfuzz import fuzz
29
+
30
+ from parse_bench.evaluation.metrics.parse.rules_base import ParseTestRule
31
+ from parse_bench.evaluation.metrics.parse.test_types import TestType
32
+ from parse_bench.evaluation.metrics.parse.utils import normalize_text
33
+ from parse_bench.test_cases.parse_rule_schemas import ParsePageDecorationRule
34
+
35
+ _TAG_RE = {
36
+ "header": re.compile(r"<page_header>(.*?)</page_header>", re.S | re.I),
37
+ "footer": re.compile(r"<page_footer>(.*?)</page_footer>", re.S | re.I),
38
+ "page_number": re.compile(r"<page_number>(.*?)</page_number>", re.S | re.I),
39
+ }
40
+ _ANY_TAG_RE = re.compile(r"</?page_(?:header|footer|number)>", re.I)
41
+ _ROMAN_RE = re.compile(r"^[ivxlcdm]+$")
42
+ _ROMAN_VALUES = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100, "d": 500, "m": 1000}
43
+
44
+
45
+ def norm_furniture(text: str | None) -> str:
46
+ """Comparison form of a header/footer: bench text normalisation, no punctuation, one space."""
47
+ s = normalize_text(text or "").lower()
48
+ s = re.sub(r"[|•·–—\-_/\\,:;.()\[\]{}\"'`*#>]+", " ", s)
49
+ return re.sub(r"\s+", " ", s).strip()
50
+
51
+
52
+ def roman_to_int(s: str) -> int | None:
53
+ s = s.lower()
54
+ if not s or not _ROMAN_RE.match(s):
55
+ return None
56
+ total, prev = 0, 0
57
+ for ch in reversed(s):
58
+ v = _ROMAN_VALUES[ch]
59
+ total = total - v if v < prev else total + v
60
+ prev = max(prev, v)
61
+ return total
62
+
63
+
64
+ def canonical_page_number(raw: str | None) -> str | None:
65
+ """``Page 12 of 48`` → ``12``; ``– iv –`` → ``iv``; ``A-3`` → ``a-3``; ``3 / 10`` → ``3``."""
66
+ if raw is None:
67
+ return None
68
+ s = normalize_text(str(raw)).strip().lower()
69
+ s = re.sub(r"^(page|p\.?|pg\.?|seite|página|pagina)\s*", "", s)
70
+ s = re.sub(r"\s*(of|/|sur|de|von)\s*\d+\s*$", "", s)
71
+ s = s.strip(" -–—|.·•")
72
+ if not s:
73
+ return None
74
+ m = re.fullmatch(r"0*(\d+)", s)
75
+ if m:
76
+ return str(int(m.group(1)))
77
+ if _ROMAN_RE.match(s):
78
+ return s
79
+ m = re.fullmatch(r"([a-z]{1,3})[\s\-–.]?0*(\d+)", s)
80
+ if m:
81
+ return f"{m.group(1)}-{int(m.group(2))}"
82
+ return s
83
+
84
+
85
+ def page_number_equal(a: str | None, b: str | None) -> bool:
86
+ ca, cb = canonical_page_number(a), canonical_page_number(b)
87
+ if ca is None or cb is None:
88
+ return ca == cb
89
+ if ca == cb:
90
+ return True
91
+ ra, rb = roman_to_int(ca), roman_to_int(cb)
92
+ # ``iv`` printed vs ``4`` predicted (or the reverse) counts as the same number.
93
+ return ra is not None and str(ra) == cb or rb is not None and str(rb) == ca
94
+
95
+
96
+ def _page_markdown(md_content: str, page: int | None) -> str:
97
+ if page is None or "\f" not in md_content:
98
+ return md_content
99
+ parts = md_content.split("\f")
100
+ if 1 <= page <= len(parts):
101
+ return parts[page - 1]
102
+ return md_content
103
+
104
+
105
+ class PageDecorationRule(ParseTestRule):
106
+ """Header / footer / printed page number for one page, plus leakage into the body."""
107
+
108
+ def __init__(self, rule_data: ParsePageDecorationRule | dict):
109
+ super().__init__(rule_data)
110
+ if self.type != TestType.PAGE_DECORATION.value:
111
+ raise ValueError(f"Invalid type for PageDecorationRule: {self.type}")
112
+ self.rule = cast(ParsePageDecorationRule, self._rule_data)
113
+
114
+ def _predicted(self, page_md: str) -> tuple[dict[str, str | None], str]:
115
+ """Predicted slots and the source they came from (``structured`` or ``markdown_tags``)."""
116
+ if self.parse_output is not None and self.parse_output.layout_pages:
117
+ pages = self.parse_output.layout_pages
118
+ if self.page is not None:
119
+ pages = [p for p in pages if p.page_number == self.page] or pages[:1]
120
+ if pages:
121
+ p = pages[0]
122
+ return {
123
+ "header": p.page_header_markdown or None,
124
+ "footer": p.page_footer_markdown or None,
125
+ "page_number": p.printed_page_number or None,
126
+ }, "structured"
127
+ out: dict[str, str | None] = {}
128
+ for slot, pattern in _TAG_RE.items():
129
+ found = [m.group(1).strip() for m in pattern.finditer(page_md)]
130
+ out[slot] = " | ".join(f for f in found if f) or None
131
+ return out, "markdown_tags"
132
+
133
+ def _text_axis(
134
+ self, expected: str | None, predicted: str | None, threshold: int, ignore: set[str] | None = None
135
+ ) -> dict[str, Any]:
136
+ ne, npd = norm_furniture(expected), norm_furniture(predicted)
137
+ if ignore:
138
+ # The printed page number may sit in the header/footer line on either side; it is scored by
139
+ # its own axis and must not cost precision or recall here.
140
+ ne = " ".join(t for t in ne.split() if t not in ignore)
141
+ npd = " ".join(t for t in npd.split() if t not in ignore)
142
+ if not ne and not npd:
143
+ return {
144
+ "passed": True,
145
+ "score": 1.0,
146
+ "expected": expected,
147
+ "predicted": predicted,
148
+ "reason": "none expected, none predicted",
149
+ }
150
+ if not ne:
151
+ return {
152
+ "passed": False,
153
+ "score": 0.0,
154
+ "expected": expected,
155
+ "predicted": predicted,
156
+ "reason": "hallucinated",
157
+ }
158
+ if not npd:
159
+ return {"passed": False, "score": 0.0, "expected": expected, "predicted": predicted, "reason": "missing"}
160
+ # Token-level F1 with fuzzy token matching: order-free (header pieces may be reordered), but a
161
+ # dropped piece lowers recall and body text swept into the header lowers precision. A plain
162
+ # token_set_ratio would score a strict subset 100 and hide missing pieces.
163
+ exp_tokens, pred_tokens = ne.split(), npd.split()
164
+ matched_pred: set[int] = set()
165
+ hits = 0
166
+ for tok in exp_tokens:
167
+ best, best_j = 0.0, -1
168
+ for j, cand in enumerate(pred_tokens):
169
+ if j in matched_pred:
170
+ continue
171
+ score = 100.0 if tok == cand else float(fuzz.ratio(tok, cand))
172
+ if score > best:
173
+ best, best_j = score, j
174
+ if best >= 85 and best_j >= 0:
175
+ matched_pred.add(best_j)
176
+ hits += 1
177
+ recall = hits / len(exp_tokens)
178
+ precision = hits / len(pred_tokens) if pred_tokens else 0.0
179
+ f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0
180
+ # Both directions must clear the threshold: F1 alone would forgive one dropped piece in four.
181
+ return {
182
+ "passed": recall * 100 >= threshold and precision * 100 >= threshold,
183
+ "score": round(f1, 4),
184
+ "expected": expected,
185
+ "predicted": predicted,
186
+ "recall": round(recall, 3),
187
+ "precision": round(precision, 3),
188
+ }
189
+
190
+ def run(self, md_content: str, normalized_content: str | None = None) -> tuple[bool, str, float]:
191
+ page_md = _page_markdown(md_content, self.page)
192
+ predicted, source = self._predicted(page_md)
193
+ r = self.rule
194
+ page_tokens = {
195
+ t for t in (canonical_page_number(r.page_number), canonical_page_number(predicted["page_number"])) if t
196
+ }
197
+ axes: dict[str, dict[str, Any]] = {
198
+ "header": self._text_axis(r.header, predicted["header"], r.text_threshold, page_tokens),
199
+ "footer": self._text_axis(r.footer, predicted["footer"], r.text_threshold, page_tokens),
200
+ }
201
+ pn_field_ok = page_number_equal(r.page_number, predicted["page_number"])
202
+ # A page number printed inside the running header or footer is legitimately returned as part of
203
+ # that text; the dedicated field is preferred but not required. It counts when the canonical
204
+ # number appears as a whole token in the predicted header/footer.
205
+ in_furniture = False
206
+ if not pn_field_ok and r.page_number and not predicted["page_number"]:
207
+ furniture = norm_furniture(" ".join(v for v in (predicted["header"], predicted["footer"]) if v))
208
+ canon = canonical_page_number(r.page_number) or ""
209
+ in_furniture = bool(canon) and re.search(rf"(?<!\w){re.escape(canon)}(?!\w)", furniture) is not None
210
+ pn_ok = pn_field_ok or in_furniture
211
+ axes["page_number"] = {
212
+ "passed": pn_ok,
213
+ "score": 1.0 if pn_ok else 0.0,
214
+ "expected": r.page_number,
215
+ "predicted": predicted["page_number"],
216
+ "expected_canonical": canonical_page_number(r.page_number),
217
+ "predicted_canonical": canonical_page_number(predicted["page_number"]),
218
+ "in_furniture_text": in_furniture,
219
+ }
220
+
221
+ # leak: annotated furniture must not remain in the body
222
+ body = page_md
223
+ for pattern in _TAG_RE.values(): # tagged furniture is not body
224
+ body = pattern.sub(" ", body)
225
+ body = _ANY_TAG_RE.sub(" ", body) # then any unbalanced stray tag
226
+ body_norm = " " + norm_furniture(body) + " "
227
+ leaked: list[str] = []
228
+ checked = 0
229
+ for slot in ("header", "footer"):
230
+ value = getattr(r, slot)
231
+ if value:
232
+ pieces = [p for p in re.split(r"\s*\|\s*", value) if len(norm_furniture(p)) >= r.leak_min_chars]
233
+ for piece in pieces:
234
+ checked += 1
235
+ if norm_furniture(piece) in body_norm:
236
+ leaked.append(piece)
237
+ if r.page_number_raw or r.page_number:
238
+ raw = norm_furniture(r.page_number_raw or r.page_number)
239
+ if raw:
240
+ checked += 1
241
+ if re.search(rf"(?<!\S){re.escape(raw)}(?!\S)", body_norm):
242
+ leaked.append(r.page_number_raw or r.page_number or "")
243
+ if checked == 0:
244
+ axes["leak"] = {"passed": None, "score": None, "reason": "no furniture expected"}
245
+ else:
246
+ axes["leak"] = {
247
+ "passed": not leaked,
248
+ "score": round(1.0 - len(leaked) / checked, 4),
249
+ "leaked": leaked,
250
+ "checked": checked,
251
+ }
252
+
253
+ self.result_details = {
254
+ "source": source,
255
+ "expected": {
256
+ "header": r.header,
257
+ "footer": r.footer,
258
+ "page_number": r.page_number,
259
+ "page_number_raw": r.page_number_raw,
260
+ },
261
+ "predicted": predicted,
262
+ "axes": axes,
263
+ }
264
+ applicable = [a for a in axes.values() if a.get("passed") is not None]
265
+ passed = all(a["passed"] for a in applicable)
266
+ score = float(np.mean([a["score"] for a in applicable])) if applicable else 1.0
267
+ failed = [k for k, a in axes.items() if a.get("passed") is False]
268
+
269
+ def verdict(k: str) -> str:
270
+ if axes[k].get("passed"):
271
+ return "ok (in header/footer text)" if k == "page_number" and axes[k].get("in_furniture_text") else "ok"
272
+ return axes[k].get("reason") or "mismatch"
273
+
274
+ summary = ", ".join(f"{k}:{verdict(k)}" for k in ("header", "footer", "page_number"))
275
+ expl = f"[{source}] {summary}" + (f"; failed: {', '.join(failed)}" if failed else "; all axes pass")
276
+ return passed, expl, round(score, 4)