parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,1666 @@
1
+ """Table structure and hierarchy test rules."""
2
+
3
+ import json
4
+ import re
5
+ import unicodedata
6
+ from collections import Counter
7
+ from html import unescape
8
+ from typing import Any, cast
9
+
10
+ import pandas as pd
11
+ from bs4 import BeautifulSoup
12
+ from rapidfuzz import fuzz
13
+ from unidecode import unidecode
14
+
15
+ from parse_bench.evaluation.metrics.parse.rules_base import (
16
+ CELL_FUZZY_MATCH_THRESHOLD,
17
+ AdjacentTableRuleData,
18
+ NoBorderTableRuleData,
19
+ ParseTestRule,
20
+ )
21
+ from parse_bench.evaluation.metrics.parse.table_parsing import (
22
+ ResolvedGrid,
23
+ TableData,
24
+ find_all_html_tables,
25
+ find_cell_in_grids,
26
+ find_table_by_anchors,
27
+ parse_html_tables,
28
+ parse_markdown_tables,
29
+ )
30
+ from parse_bench.evaluation.metrics.parse.test_types import TestType
31
+ from parse_bench.evaluation.metrics.parse.utils import normalize_text
32
+ from parse_bench.test_cases.parse_rule_schemas import (
33
+ ParseTableAdjacentDownRule,
34
+ ParseTableAdjacentLeftRule,
35
+ ParseTableAdjacentRightRule,
36
+ ParseTableAdjacentUpRule,
37
+ ParseTableColspanRule,
38
+ ParseTableHeaderChainRule,
39
+ ParseTableLeftHeaderRule,
40
+ ParseTableMarkerCellsRule,
41
+ ParseTableNoAboveRule,
42
+ ParseTableNoBelowRule,
43
+ ParseTableNoLeftRule,
44
+ ParseTableNoRightRule,
45
+ ParseTableRowspanRule,
46
+ ParseTableRule,
47
+ ParseTableSameColumnRule,
48
+ ParseTableSameRowRule,
49
+ ParseTablesNumColsRule,
50
+ ParseTablesNumRowsRule,
51
+ ParseTablesValuesRule,
52
+ ParseTableTopHeaderRule,
53
+ )
54
+
55
+ _DEFAULT_TABLE_MARKER_ALIASES = {
56
+ "x",
57
+ "y",
58
+ "n",
59
+ "yes",
60
+ "no",
61
+ "check",
62
+ "checked",
63
+ "✓",
64
+ "✔",
65
+ "☑",
66
+ "✅",
67
+ "✗",
68
+ "✘",
69
+ "×",
70
+ "•",
71
+ "●",
72
+ "∙",
73
+ "○",
74
+ "◉",
75
+ "◦",
76
+ "■",
77
+ "□",
78
+ "▪",
79
+ "▫",
80
+ "◼",
81
+ "◻",
82
+ "◆",
83
+ "◇",
84
+ "♦",
85
+ "★",
86
+ "☆",
87
+ "→",
88
+ "←",
89
+ "↑",
90
+ "↓",
91
+ "▲",
92
+ "▼",
93
+ "+",
94
+ "!",
95
+ "selected",
96
+ "green",
97
+ "red",
98
+ "yellow",
99
+ "orange",
100
+ "grey",
101
+ "gray",
102
+ "green square",
103
+ "red square",
104
+ "yellow square",
105
+ "orange square",
106
+ "grey square",
107
+ "gray square",
108
+ "expert knowledge",
109
+ "good knowledge",
110
+ "basic knowledge",
111
+ }
112
+ _MARKDOWN_IMAGE_RE = re.compile(r"!\[[^\]]*\]\([^)]*\)", re.DOTALL)
113
+ _BRACKETED_IMAGE_LABEL_RE = re.compile(r"\[\s*(?:icon|image)(?::[^\]]+)?\s*\]", re.IGNORECASE)
114
+ _HTML_IMAGE_RE = re.compile(r"<img\b[^>]*>", re.IGNORECASE)
115
+ _HTML_TAG_RE = re.compile(r"<[^>]+>")
116
+ _HTML_IMAGE_CELL_SENTINEL = "llamacloud-bench-image-cell"
117
+ _MARKER_LIST_TOKEN = "<marker-list>"
118
+
119
+
120
+ def _preserve_html_image_cells(content: str) -> str:
121
+ """Replace images inside HTML cells with text visible to ``parse_html_tables``.
122
+
123
+ The generic HTML table parser intentionally extracts text with
124
+ ``BeautifulSoup.get_text()``, which drops ``<img>`` elements. Work on a
125
+ private copy here so image presence is retained for this rule without
126
+ changing cell values seen by every other table evaluator.
127
+ """
128
+
129
+ if not _HTML_IMAGE_RE.search(content):
130
+ return content
131
+ soup = BeautifulSoup(content, "lxml")
132
+ changed = False
133
+ for cell in soup.find_all(["td", "th"]):
134
+ if cell.find("img") is None:
135
+ continue
136
+ cell.clear()
137
+ cell.append(_HTML_IMAGE_CELL_SENTINEL)
138
+ changed = True
139
+ return str(soup) if changed else content
140
+
141
+
142
+ def _marker_cell_token(value: object, aliases: set[str], allow_ocr_glyphs: bool) -> str | None:
143
+ raw = unescape(str(value)).strip()
144
+ if not raw:
145
+ return None
146
+ if (
147
+ _HTML_IMAGE_CELL_SENTINEL in raw
148
+ or _MARKDOWN_IMAGE_RE.search(raw)
149
+ or _BRACKETED_IMAGE_LABEL_RE.search(raw)
150
+ or _HTML_IMAGE_RE.search(raw)
151
+ ):
152
+ return "<image>"
153
+
154
+ token = " ".join(_HTML_TAG_RE.sub("", raw).split()).casefold()
155
+ if not token:
156
+ return None
157
+ if token in aliases:
158
+ return token
159
+ marker_list = [part.strip() for part in re.split(r"[,;/]", token)]
160
+ if len(marker_list) > 1 and all(part in aliases for part in marker_list):
161
+ return _MARKER_LIST_TOKEN
162
+ # Markers are sometimes serialized beside their numeric or textual value,
163
+ # for example ``▲ 512.40`` or ``✓ Announced``. Only accept a leading
164
+ # alias when it is punctuation/symbol-like; ordinary word aliases must
165
+ # still occupy the whole cell to avoid matching prose.
166
+ leading_token = token.split(maxsplit=1)[0]
167
+ if leading_token in aliases and all(unicodedata.category(char)[0] in {"P", "S"} for char in leading_token):
168
+ return leading_token
169
+ if len(token) >= 3 and (token[0], token[-1]) in {("[", "]"), ("(", ")"), ("{", "}")}:
170
+ inner = token[1:-1].strip()
171
+ if inner in aliases:
172
+ return inner
173
+ if not allow_ocr_glyphs or len(token) > 3 or any(char.isspace() or char.isdigit() for char in token):
174
+ return None
175
+
176
+ # Visual markers are frequently OCR'd as a non-ASCII glyph (for example
177
+ # Harley-Davidson's badge becomes Cyrillic Ө). Accept short non-ASCII
178
+ # glyphs, plus Unicode symbols/punctuation, but leave ordinary ASCII words
179
+ # and punctuation out so abbreviations and missing-value dashes do not
180
+ # become false markers. ASCII symbols must be configured as aliases.
181
+ if any(ord(char) > 127 for char in token):
182
+ return token
183
+ return None
184
+
185
+
186
+ class TableMarkerCellsRule(ParseTestRule):
187
+ """Check that repeated icon-like values remain inside a table grid."""
188
+
189
+ def __init__(self, rule_data: ParseTableMarkerCellsRule | dict):
190
+ super().__init__(rule_data)
191
+ rule_data = cast(ParseTableMarkerCellsRule, self._rule_data)
192
+ if self.type != TestType.TABLE_MARKER_CELLS.value:
193
+ raise ValueError(f"Invalid type for TableMarkerCellsRule: {self.type}")
194
+ self.min_count = rule_data.min_count
195
+ self.min_distinct_rows = rule_data.min_distinct_rows
196
+ self.min_distinct_columns = rule_data.min_distinct_columns
197
+ self.marker_aliases = {
198
+ " ".join(unescape(alias).split()).casefold()
199
+ for alias in [*_DEFAULT_TABLE_MARKER_ALIASES, *rule_data.marker_aliases]
200
+ if alias.strip()
201
+ }
202
+ self.allow_repeated_ocr_glyphs = rule_data.allow_repeated_ocr_glyphs
203
+
204
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
205
+ tables = [*parse_markdown_tables(content), *parse_html_tables(_preserve_html_image_cells(content))]
206
+ if not tables:
207
+ self.result_details = {
208
+ "requirement": self._requirement_summary(),
209
+ "tables_inspected": 0,
210
+ "diagnosis": "The parser emitted no recognizable Markdown or HTML table.",
211
+ }
212
+ return False, "No tables found; icon-valued cells could not be evaluated"
213
+
214
+ best = (0, 0, 0)
215
+ best_table: dict[str, Any] = {}
216
+ for table_index, table in enumerate(tables, start=1):
217
+ positions_by_token: dict[str, list[tuple[int, int]]] = {}
218
+ for row in range(table.data.shape[0]):
219
+ for column in range(table.data.shape[1]):
220
+ token = _marker_cell_token(
221
+ table.data[row, column],
222
+ self.marker_aliases,
223
+ self.allow_repeated_ocr_glyphs,
224
+ )
225
+ if token is not None:
226
+ positions_by_token.setdefault(token, []).append((row, column))
227
+
228
+ # A short OCR glyph only counts as a marker when it repeats. Known
229
+ # aliases and image-only cells are already semantically explicit.
230
+ positions = [
231
+ position
232
+ for token, token_positions in positions_by_token.items()
233
+ if token in self.marker_aliases or token in {"<image>", _MARKER_LIST_TOKEN} or len(token_positions) >= 2
234
+ for position in token_positions
235
+ ]
236
+ score = (
237
+ len(positions),
238
+ len({row for row, _ in positions}),
239
+ len({column for _, column in positions}),
240
+ )
241
+ table_details = {
242
+ "index": table_index,
243
+ "shape": f"{table.data.shape[0]} rows x {table.data.shape[1]} columns",
244
+ "marker_cells": score[0],
245
+ "marker_rows": score[1],
246
+ "marker_columns": score[2],
247
+ "recognized_tokens": {
248
+ token: len(token_positions) for token, token_positions in sorted(positions_by_token.items())
249
+ },
250
+ }
251
+ if score > best or not best_table:
252
+ best = score
253
+ best_table = table_details
254
+ if (
255
+ score[0] >= self.min_count
256
+ and score[1] >= self.min_distinct_rows
257
+ and score[2] >= self.min_distinct_columns
258
+ ):
259
+ self.result_details = {
260
+ "requirement": self._requirement_summary(),
261
+ "tables_inspected": len(tables),
262
+ "matching_table": table_details,
263
+ }
264
+ return True, ""
265
+
266
+ diagnosis = self._failure_diagnosis(best)
267
+ self.result_details = {
268
+ "requirement": self._requirement_summary(),
269
+ "tables_inspected": len(tables),
270
+ "best_table": best_table,
271
+ "diagnosis": diagnosis,
272
+ }
273
+
274
+ return (
275
+ False,
276
+ f"Icon-table rule failed: {diagnosis} "
277
+ f"Best table had {best[0]} marker cells across {best[1]} rows and {best[2]} columns; "
278
+ f"required {self._requirement_summary()}.",
279
+ )
280
+
281
+ def _requirement_summary(self) -> str:
282
+ return (
283
+ f">={self.min_count} marker cells across >={self.min_distinct_rows} rows "
284
+ f"and >={self.min_distinct_columns} columns"
285
+ )
286
+
287
+ def _failure_diagnosis(self, best: tuple[int, int, int]) -> str:
288
+ if best[0] < self.min_count:
289
+ if best[0] == 0:
290
+ return "tables were found, but no repeated recognized marker-only values remained in their cells."
291
+ return f"only {best[0]} recognized marker cells remained; at least {self.min_count} are required."
292
+ if best[1] < self.min_distinct_rows:
293
+ return f"markers occupied only {best[1]} table rows; at least {self.min_distinct_rows} are required."
294
+ return f"markers occupied only {best[2]} table column(s); at least {self.min_distinct_columns} are required."
295
+
296
+
297
+ class TableRule(ParseTestRule):
298
+ """Test rule to verify table cell relationships."""
299
+
300
+ def __init__(self, rule_data: ParseTableRule | dict):
301
+ super().__init__(rule_data)
302
+ rule_data = cast(ParseTableRule, self._rule_data)
303
+
304
+ if self.type != TestType.TABLE.value:
305
+ raise ValueError(f"Invalid type for TableRule: {self.type}")
306
+
307
+ # Normalize the search text
308
+ self.cell = normalize_text(rule_data.cell)
309
+ self.up = normalize_text(rule_data.up or "")
310
+ self.down = normalize_text(rule_data.down or "")
311
+ self.left = normalize_text(rule_data.left or "")
312
+ self.right = normalize_text(rule_data.right or "")
313
+ self.top_heading = normalize_text(rule_data.top_heading or "")
314
+ self.left_heading = normalize_text(rule_data.left_heading or "")
315
+ self.ignore_markdown_tables = rule_data.ignore_markdown_tables
316
+
317
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
318
+ """Check if table cell relationships are satisfied."""
319
+ tables_to_check = []
320
+ failed_reasons = []
321
+
322
+ # Threshold for fuzzy matching derived from max_diffs
323
+ threshold = 1.0 - (self.max_diffs / (len(self.cell) if len(self.cell) > 0 else 1))
324
+ threshold = max(0.5, threshold)
325
+
326
+ # Parse tables
327
+ if not self.ignore_markdown_tables:
328
+ md_tables = parse_markdown_tables(content)
329
+ tables_to_check.extend(md_tables)
330
+
331
+ html_tables = parse_html_tables(content)
332
+ tables_to_check.extend(html_tables)
333
+
334
+ # If no tables found, return failure
335
+ if not tables_to_check:
336
+ return False, "No tables found in the content"
337
+
338
+ # Check each table
339
+ for table_data in tables_to_check:
340
+ table_array = table_data.data
341
+ header_rows = table_data.header_rows
342
+ header_cols = table_data.header_cols
343
+
344
+ # Find all cells that match the target cell using fuzzy matching
345
+ matches = []
346
+ for i in range(table_array.shape[0]):
347
+ for j in range(table_array.shape[1]):
348
+ cell_content = normalize_text(str(table_array[i, j]))
349
+ similarity = fuzz.ratio(self.cell, cell_content) / 100.0
350
+
351
+ if similarity >= threshold:
352
+ matches.append((i, j))
353
+
354
+ # If no matches found in this table, continue to the next table
355
+ if not matches:
356
+ continue
357
+
358
+ # Check the relationships for each matching cell
359
+ for row_idx, col_idx in matches:
360
+ all_relationships_satisfied = True
361
+ current_failed_reasons = []
362
+
363
+ # Check up relationship
364
+ if self.up and row_idx > 0:
365
+ up_cell = normalize_text(str(table_array[row_idx - 1, col_idx]))
366
+ up_similarity = fuzz.ratio(self.up, up_cell) / 100.0
367
+ up_threshold = max(0.5, 1.0 - (self.max_diffs / (len(self.up) if len(self.up) > 0 else 1)))
368
+ if up_similarity < up_threshold:
369
+ all_relationships_satisfied = False
370
+ current_failed_reasons.append(
371
+ f"Cell above '{up_cell}' doesn't match "
372
+ f"expected '{self.up}' "
373
+ f"(similarity: {up_similarity:.2f})"
374
+ )
375
+
376
+ # Check down relationship
377
+ if self.down and row_idx < table_array.shape[0] - 1:
378
+ down_cell = normalize_text(str(table_array[row_idx + 1, col_idx]))
379
+ down_similarity = fuzz.ratio(self.down, down_cell) / 100.0
380
+ down_threshold = max(0.5, 1.0 - (self.max_diffs / (len(self.down) if len(self.down) > 0 else 1)))
381
+ if down_similarity < down_threshold:
382
+ all_relationships_satisfied = False
383
+ current_failed_reasons.append(
384
+ f"Cell below '{down_cell}' doesn't match "
385
+ f"expected '{self.down}' "
386
+ f"(similarity: {down_similarity:.2f})"
387
+ )
388
+
389
+ # Check left relationship
390
+ if self.left and col_idx > 0:
391
+ left_cell = normalize_text(str(table_array[row_idx, col_idx - 1]))
392
+ left_similarity = fuzz.ratio(self.left, left_cell) / 100.0
393
+ left_threshold = max(0.5, 1.0 - (self.max_diffs / (len(self.left) if len(self.left) > 0 else 1)))
394
+ if left_similarity < left_threshold:
395
+ all_relationships_satisfied = False
396
+ current_failed_reasons.append(
397
+ f"Cell to the left '{left_cell}' doesn't "
398
+ f"match expected '{self.left}' "
399
+ f"(similarity: {left_similarity:.2f})"
400
+ )
401
+
402
+ # Check right relationship
403
+ if self.right and col_idx < table_array.shape[1] - 1:
404
+ right_cell = normalize_text(str(table_array[row_idx, col_idx + 1]))
405
+ right_similarity = fuzz.ratio(self.right, right_cell) / 100.0
406
+ right_threshold = max(
407
+ 0.5,
408
+ 1.0 - (self.max_diffs / (len(self.right) if len(self.right) > 0 else 1)),
409
+ )
410
+ if right_similarity < right_threshold:
411
+ all_relationships_satisfied = False
412
+ current_failed_reasons.append(
413
+ f"Cell to the right '{right_cell}' doesn't "
414
+ f"match expected '{self.right}' "
415
+ f"(similarity: {right_similarity:.2f})"
416
+ )
417
+
418
+ # Check top heading relationship
419
+ if self.top_heading:
420
+ top_heading_found = False
421
+ best_match = ""
422
+ best_similarity = 0.0
423
+
424
+ # Check the col_headers dictionary first
425
+ if col_idx in table_data.col_headers:
426
+ for _, header_text in table_data.col_headers[col_idx]:
427
+ header_text = normalize_text(header_text)
428
+ similarity = fuzz.ratio(self.top_heading, header_text) / 100.0
429
+ if similarity > best_similarity:
430
+ best_similarity = similarity
431
+ best_match = header_text
432
+ top_threshold = max(
433
+ 0.5,
434
+ 1.0
435
+ - (self.max_diffs / (len(self.top_heading) if len(self.top_heading) > 0 else 1)),
436
+ )
437
+ if best_similarity >= top_threshold:
438
+ top_heading_found = True
439
+ break
440
+
441
+ # If no match found in col_headers, fall back to checking header rows
442
+ if not top_heading_found and header_rows:
443
+ for i in sorted(header_rows):
444
+ if i < row_idx and str(table_array[i, col_idx]).strip():
445
+ header_text = normalize_text(str(table_array[i, col_idx]))
446
+ similarity = fuzz.ratio(self.top_heading, header_text) / 100.0
447
+ if similarity > best_similarity:
448
+ best_similarity = similarity
449
+ best_match = header_text
450
+ top_threshold = max(
451
+ 0.5,
452
+ 1.0
453
+ - (
454
+ self.max_diffs / (len(self.top_heading) if len(self.top_heading) > 0 else 1)
455
+ ),
456
+ )
457
+ if best_similarity >= top_threshold:
458
+ top_heading_found = True
459
+ break
460
+
461
+ # If still no match, use any non-empty cell above as a last resort
462
+ if not top_heading_found and not best_match and row_idx > 0:
463
+ for i in range(row_idx):
464
+ if str(table_array[i, col_idx]).strip():
465
+ header_text = normalize_text(str(table_array[i, col_idx]))
466
+ similarity = fuzz.ratio(self.top_heading, header_text) / 100.0
467
+ if similarity > best_similarity:
468
+ best_similarity = similarity
469
+ best_match = header_text
470
+
471
+ if not best_match:
472
+ all_relationships_satisfied = False
473
+ current_failed_reasons.append(f"No top heading found for cell at ({row_idx}, {col_idx})")
474
+ else:
475
+ top_threshold = max(
476
+ 0.5,
477
+ 1.0 - (self.max_diffs / (len(self.top_heading) if len(self.top_heading) > 0 else 1)),
478
+ )
479
+ if best_similarity < top_threshold:
480
+ all_relationships_satisfied = False
481
+ current_failed_reasons.append(
482
+ f"Top heading '{best_match}' doesn't "
483
+ f"match expected '{self.top_heading}' "
484
+ f"(similarity: {best_similarity:.2f})"
485
+ )
486
+
487
+ # Check left heading relationship
488
+ if self.left_heading:
489
+ left_heading_found = False
490
+ best_match = ""
491
+ best_similarity = 0.0
492
+
493
+ # Check the row_headers dictionary first
494
+ if row_idx in table_data.row_headers:
495
+ for _, header_text in table_data.row_headers[row_idx]:
496
+ header_text = normalize_text(header_text)
497
+ similarity = fuzz.ratio(self.left_heading, header_text) / 100.0
498
+ if similarity > best_similarity:
499
+ best_similarity = similarity
500
+ best_match = header_text
501
+ left_threshold = max(
502
+ 0.5,
503
+ 1.0
504
+ - (self.max_diffs / (len(self.left_heading) if len(self.left_heading) > 0 else 1)),
505
+ )
506
+ if best_similarity >= left_threshold:
507
+ left_heading_found = True
508
+ break
509
+
510
+ # If no match found in row_headers, fall back to checking header columns
511
+ if not left_heading_found and header_cols:
512
+ for j in sorted(header_cols):
513
+ if j < col_idx and str(table_array[row_idx, j]).strip():
514
+ header_text = normalize_text(str(table_array[row_idx, j]))
515
+ similarity = fuzz.ratio(self.left_heading, header_text) / 100.0
516
+ if similarity > best_similarity:
517
+ best_similarity = similarity
518
+ best_match = header_text
519
+ left_threshold = max(
520
+ 0.5,
521
+ 1.0
522
+ - (
523
+ self.max_diffs
524
+ / (len(self.left_heading) if len(self.left_heading) > 0 else 1)
525
+ ),
526
+ )
527
+ if best_similarity >= left_threshold:
528
+ left_heading_found = True
529
+ break
530
+
531
+ # If still no match, use any non-empty cell to the left as a last resort
532
+ if not left_heading_found and not best_match and col_idx > 0:
533
+ for j in range(col_idx):
534
+ if str(table_array[row_idx, j]).strip():
535
+ header_text = normalize_text(str(table_array[row_idx, j]))
536
+ similarity = fuzz.ratio(self.left_heading, header_text) / 100.0
537
+ if similarity > best_similarity:
538
+ best_similarity = similarity
539
+ best_match = header_text
540
+
541
+ if not best_match:
542
+ all_relationships_satisfied = False
543
+ current_failed_reasons.append(f"No left heading found for cell at ({row_idx}, {col_idx})")
544
+ else:
545
+ left_threshold = max(
546
+ 0.5,
547
+ 1.0 - (self.max_diffs / (len(self.left_heading) if len(self.left_heading) > 0 else 1)),
548
+ )
549
+ if best_similarity < left_threshold:
550
+ all_relationships_satisfied = False
551
+ current_failed_reasons.append(
552
+ f"Left heading '{best_match}' doesn't "
553
+ f"match expected '{self.left_heading}' "
554
+ f"(similarity: {best_similarity:.2f})"
555
+ )
556
+
557
+ # If all relationships are satisfied for this cell, the test passes
558
+ if all_relationships_satisfied:
559
+ return True, ""
560
+ else:
561
+ failed_reasons.extend(current_failed_reasons)
562
+
563
+ if not failed_reasons:
564
+ return (
565
+ False,
566
+ f"No cell matching '{self.cell}' found in any table with threshold {threshold}",
567
+ )
568
+ else:
569
+ return (
570
+ False,
571
+ f"Found cells matching '{self.cell}' but relationships were not satisfied: {'; '.join(failed_reasons)}",
572
+ )
573
+
574
+
575
+ class TablesValuesRule(ParseTestRule):
576
+ """Test rule to verify that tables match ground truth tables."""
577
+
578
+ def __init__(self, rule_data: ParseTablesValuesRule | dict):
579
+ super().__init__(rule_data)
580
+ rule_data = cast(ParseTablesValuesRule, self._rule_data)
581
+
582
+ if self.type != TestType.TABLES_VALUES.value:
583
+ raise ValueError(f"Invalid type for TablesValuesRule: {self.type}")
584
+
585
+ self.table_variations = rule_data.table_variations
586
+ self.json_path = rule_data.json_path
587
+ self.table_match_threshold = rule_data.table_match_threshold
588
+ self.table_values_match_threshold = rule_data.table_values_match_threshold
589
+ self.add_check_num_rows_test = rule_data.add_check_num_rows_test
590
+ self.add_check_num_cols_test = rule_data.add_check_num_cols_test
591
+
592
+ # Must have either table_variations or json_path
593
+ if not self.table_variations and not self.json_path:
594
+ raise ValueError("Either table_variations or json_path must be provided")
595
+
596
+ self.relevant_gt: pd.DataFrame | None = None
597
+ self.relevant_pred: pd.DataFrame | None = None
598
+
599
+ def _load_json_table(self, json_file_path: str) -> dict[str, Any]:
600
+ """Load the ground truth table JSON file."""
601
+ with open(json_file_path, encoding="utf-8") as f:
602
+ data = json.load(f)
603
+
604
+ # Validate the structure
605
+ required_keys = ["pdf", "page", "table_variations", "id", "table_match_threshold"]
606
+ for key in required_keys:
607
+ if key not in data:
608
+ raise ValueError(f"JSON file missing required key: {key}")
609
+
610
+ return data # type: ignore[no-any-return]
611
+
612
+ def _tabledata_to_dataframe(self, table_data: TableData) -> pd.DataFrame:
613
+ """Convert a TableData object to a pandas DataFrame."""
614
+ return pd.DataFrame(table_data.data)
615
+
616
+ def _table_schema_to_dataframe(self, table_schema: dict[str, Any]) -> pd.DataFrame:
617
+ """Convert a table schema (with rowspan/colspan) to a pandas DataFrame."""
618
+ if "rows" not in table_schema:
619
+ raise ValueError("table_schema missing 'rows' key")
620
+
621
+ rows_data = table_schema["rows"]
622
+
623
+ # First pass: determine the grid size and build a cell position map
624
+ grid = {} # (row_idx, col_idx) -> cell_text
625
+ max_cols = 0
626
+
627
+ for row_idx, row_dict in enumerate(rows_data):
628
+ if "cells" not in row_dict:
629
+ raise ValueError(f"Row {row_idx} missing 'cells' key")
630
+
631
+ cells = row_dict["cells"]
632
+ col_idx = 0
633
+
634
+ for cell_dict in cells:
635
+ # Skip columns that are already filled by previous rowspan/colspan
636
+ while (row_idx, col_idx) in grid:
637
+ col_idx += 1
638
+
639
+ # Extract cell properties
640
+ text = cell_dict.get("text", "")
641
+ colspan = cell_dict.get("colspan", 1)
642
+ rowspan = cell_dict.get("rowspan", 1)
643
+
644
+ # Fill the grid for this cell and all its spans
645
+ for r_offset in range(rowspan):
646
+ for c_offset in range(colspan):
647
+ grid[(row_idx + r_offset, col_idx + c_offset)] = text
648
+
649
+ col_idx += colspan
650
+ max_cols = max(max_cols, col_idx)
651
+
652
+ # Second pass: build the DataFrame from the grid
653
+ num_rows = max(r for r, c in grid.keys()) + 1 if grid else 0
654
+ num_cols = max_cols
655
+
656
+ # Create a 2D list for the DataFrame
657
+ data_array = []
658
+ for r in range(num_rows):
659
+ row = []
660
+ for c in range(num_cols):
661
+ cell_value = grid.get((r, c), "")
662
+ row.append(cell_value)
663
+ data_array.append(row)
664
+
665
+ # Convert to DataFrame
666
+ df = pd.DataFrame(data_array)
667
+
668
+ return df
669
+
670
+ def _normalize_cell(self, cell: str) -> str:
671
+ """Normalize a cell value for comparison."""
672
+ text = unidecode(str(cell)).lower()
673
+ # Remove all whitespace
674
+ text = re.sub(r"\s+", "", text)
675
+ # Remove zero-width characters
676
+ text = re.sub(r"[\u200B-\u200D\uFEFF\u00AD]", "", text)
677
+
678
+ # Handle numbers with commas
679
+ number_pattern = r"(-?\d{1,3}(?:,\d{3})*\.?\d*)"
680
+ match = re.search(number_pattern, text)
681
+
682
+ if match:
683
+ number_str = match.group(1)
684
+ try:
685
+ clean_number = number_str.replace(",", "")
686
+ float_val = float(clean_number)
687
+
688
+ if "," in number_str:
689
+ if float_val >= 1000:
690
+ normalized_number = f"{float_val:,.10g}".rstrip("0").rstrip(".")
691
+ else:
692
+ normalized_number = f"{float_val:g}"
693
+ else:
694
+ normalized_number = f"{float_val:g}"
695
+
696
+ return text.replace(number_str, normalized_number)
697
+ except ValueError:
698
+ pass
699
+
700
+ return text
701
+
702
+ def _compute_single_table_similarity(self, gt_df: pd.DataFrame, pred_df: pd.DataFrame) -> float:
703
+ # Extract and normalize all words from ground truth table
704
+ gt_words = []
705
+ for row_idx in range(gt_df.shape[0]):
706
+ for col_idx in range(gt_df.shape[1]):
707
+ cell_value = str(gt_df.iloc[row_idx, col_idx])
708
+ normalized = self._normalize_cell(cell_value)
709
+ if normalized:
710
+ gt_words.append(normalized)
711
+
712
+ gt_counter = Counter(gt_words)
713
+
714
+ # If ground truth is empty, return 0
715
+ if not gt_words:
716
+ return 0.0
717
+
718
+ # Extract and normalize all words from predicted table
719
+ pred_words = []
720
+ for row_idx in range(pred_df.shape[0]):
721
+ for col_idx in range(pred_df.shape[1]):
722
+ cell_value = str(pred_df.iloc[row_idx, col_idx])
723
+ normalized = self._normalize_cell(cell_value)
724
+ if normalized:
725
+ pred_words.append(normalized)
726
+
727
+ pred_counter = Counter(pred_words)
728
+
729
+ # If predicted table is empty, return 0
730
+ if not pred_words:
731
+ return 0.0
732
+
733
+ # Compute intersection (minimum counts for each word)
734
+ intersection = sum((gt_counter & pred_counter).values())
735
+
736
+ # Compute union (maximum counts for each word)
737
+ union = sum((gt_counter | pred_counter).values())
738
+
739
+ # Compute Jaccard similarity with counts
740
+ if union > 0:
741
+ return intersection / union
742
+
743
+ return 0.0
744
+
745
+ def _compare_cells_exactly(self, gt_df: pd.DataFrame, pred_df: pd.DataFrame) -> tuple[int, int, float]:
746
+ """Compare cells one-by-one between ground truth and predicted DataFrames."""
747
+ # Compare only the overlapping region
748
+ min_rows = min(len(gt_df), len(pred_df))
749
+ min_cols = min(len(gt_df.columns), len(pred_df.columns))
750
+
751
+ matching_cells = 0
752
+ total_cells = min_rows * min_cols
753
+
754
+ if total_cells == 0:
755
+ return 0, 0, 0.0
756
+
757
+ for row_idx in range(min_rows):
758
+ for col_idx in range(min_cols):
759
+ gt_cell = str(gt_df.iloc[row_idx, col_idx])
760
+ pred_cell = str(pred_df.iloc[row_idx, col_idx])
761
+
762
+ # Normalize both cells
763
+ gt_normalized = self._normalize_cell(gt_cell)
764
+ pred_normalized = self._normalize_cell(pred_cell)
765
+
766
+ # Exact match after normalization
767
+ if gt_normalized == pred_normalized:
768
+ matching_cells += 1
769
+
770
+ match_ratio = matching_cells / total_cells if total_cells > 0 else 0.0
771
+ return matching_cells, total_cells, match_ratio
772
+
773
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
774
+ """Run the table values test on provided content."""
775
+ # Extract tables from content
776
+ pred_tables = []
777
+
778
+ # Parse markdown tables
779
+ md_tables = parse_markdown_tables(content)
780
+ pred_tables.extend(md_tables)
781
+
782
+ # Parse HTML tables
783
+ html_tables = parse_html_tables(content)
784
+ pred_tables.extend(html_tables)
785
+
786
+ if not pred_tables:
787
+ return False, "No tables found in the content"
788
+
789
+ pred_tables = [self._tabledata_to_dataframe(table) for table in pred_tables]
790
+
791
+ # Load table variations either from embedded data or external JSON
792
+ try:
793
+ if self.table_variations:
794
+ table_variations = self.table_variations
795
+ elif self.json_path:
796
+ gt_data = self._load_json_table(self.json_path)
797
+ table_variations = gt_data["table_variations"]
798
+ else:
799
+ return False, "No table variations available (neither embedded nor in json_path)"
800
+
801
+ if not table_variations:
802
+ return False, "No table variations found"
803
+
804
+ except Exception as e:
805
+ return False, f"Error loading ground truth data: {e}"
806
+
807
+ # Track the best variation and its score
808
+ best_variation_idx = -1
809
+ best_score = 0.0
810
+ best_pred_idx = -1
811
+
812
+ for var_idx, table_schema in enumerate(table_variations):
813
+ try:
814
+ # Convert the table schema to a DataFrame
815
+ gt_df = self._table_schema_to_dataframe(table_schema)
816
+
817
+ # Compare this GT variation with each predicted table
818
+ for pred_idx, pred_table in enumerate(pred_tables):
819
+ # Compute similarity between this specific GT and this specific pred table
820
+ similarity = self._compute_single_table_similarity(gt_df, pred_table)
821
+
822
+ # Track the overall best score across all GT-pred pairs
823
+ if similarity > best_score:
824
+ best_score = similarity
825
+ best_variation_idx = var_idx
826
+ best_pred_idx = pred_idx
827
+ self.relevant_pred = pred_table
828
+ self.relevant_gt = gt_df
829
+
830
+ except Exception:
831
+ # If conversion fails, continue to next variation
832
+ continue
833
+
834
+ # Check if the best variation passes the threshold
835
+ threshold = self.table_match_threshold
836
+
837
+ if best_score < threshold:
838
+ return (
839
+ False,
840
+ f"Best match: GT variation {best_variation_idx} with pred table {best_pred_idx} "
841
+ f"scored {best_score:.3f}, below threshold {threshold:.3f}",
842
+ )
843
+
844
+ # Perform exact cell-by-cell comparison on the best match
845
+ if self.relevant_gt is None or self.relevant_pred is None:
846
+ return False, "No relevant GT or pred table found for cell comparison"
847
+
848
+ matching_cells, total_cells, match_ratio = self._compare_cells_exactly(self.relevant_gt, self.relevant_pred)
849
+ cell_threshold = self.table_values_match_threshold
850
+
851
+ if match_ratio < cell_threshold:
852
+ return ( # type: ignore[return-value]
853
+ False,
854
+ f"Best match: GT variation {best_variation_idx} with pred table {best_pred_idx} "
855
+ f"scored {best_score:.3f} (>= {threshold:.3f}), "
856
+ f"but cell exact match "
857
+ f"{matching_cells}/{total_cells} "
858
+ f"({match_ratio:.3f}) below threshold "
859
+ f"{cell_threshold:.3f}",
860
+ f"({match_ratio:.3f}) below threshold {cell_threshold:.3f}",
861
+ )
862
+
863
+ return (
864
+ True,
865
+ f"Best match: GT variation {best_variation_idx} with pred table {best_pred_idx} "
866
+ f"scored {best_score:.3f} (>= {threshold:.3f}), "
867
+ f"cell exact match {matching_cells}/{total_cells} "
868
+ f"({match_ratio:.3f})",
869
+ )
870
+
871
+
872
+ class TablesNumRowsRule(ParseTestRule):
873
+ """Test rule to verify that predicted table has the correct number of rows."""
874
+
875
+ def __init__(self, rule_data: ParseTablesNumRowsRule | dict):
876
+ super().__init__(rule_data)
877
+ rule_data = cast(ParseTablesNumRowsRule, self._rule_data)
878
+
879
+ if self.type != TestType.TABLES_NUM_ROWS.value:
880
+ raise ValueError(f"Invalid type for TablesNumRowsRule: {self.type}")
881
+
882
+ self.expected_num_rows = rule_data.expected_num_rows
883
+ self.actual_num_rows = rule_data.actual_num_rows
884
+
885
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
886
+ """Check if row count matches."""
887
+ if self.actual_num_rows is None:
888
+ return False, "Row count not populated"
889
+
890
+ if self.actual_num_rows == self.expected_num_rows:
891
+ return True, f"Row count matches: {self.actual_num_rows}"
892
+ else:
893
+ return (
894
+ False,
895
+ f"Row count mismatch: expected {self.expected_num_rows}, got {self.actual_num_rows}",
896
+ )
897
+
898
+
899
+ class TablesNumColsRule(ParseTestRule):
900
+ """Test rule to verify that predicted table has the correct number of columns."""
901
+
902
+ def __init__(self, rule_data: ParseTablesNumColsRule | dict):
903
+ super().__init__(rule_data)
904
+ rule_data = cast(ParseTablesNumColsRule, self._rule_data)
905
+
906
+ if self.type != TestType.TABLES_NUM_COLS.value:
907
+ raise ValueError(f"Invalid type for TablesNumColsRule: {self.type}")
908
+
909
+ self.expected_num_cols = rule_data.expected_num_cols
910
+ self.actual_num_cols = rule_data.actual_num_cols
911
+
912
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
913
+ """Check if column count matches."""
914
+ if self.actual_num_cols is None:
915
+ return False, "Column count not populated"
916
+
917
+ if self.actual_num_cols == self.expected_num_cols:
918
+ return True, f"Column count matches: {self.actual_num_cols}"
919
+ else:
920
+ return (
921
+ False,
922
+ f"Column count mismatch: expected {self.expected_num_cols}, got {self.actual_num_cols}",
923
+ )
924
+
925
+
926
+ # =============================================================================
927
+ # Table Hierarchy Rules
928
+ # =============================================================================
929
+
930
+
931
+ class TableColspanRule(ParseTestRule):
932
+ """Test rule to verify a cell has the expected colspan attribute."""
933
+
934
+ def __init__(self, rule_data: ParseTableColspanRule | dict):
935
+ super().__init__(rule_data)
936
+ rule_data = cast(ParseTableColspanRule, self._rule_data)
937
+
938
+ if self.type != TestType.TABLE_COLSPAN.value:
939
+ raise ValueError(f"Invalid type for TableColspanRule: {self.type}")
940
+
941
+ self.cell = normalize_text(rule_data.cell)
942
+ self.expected_colspan = rule_data.expected_colspan
943
+ self.table_anchor_cells = rule_data.table_anchor_cells
944
+
945
+ if not self.cell:
946
+ raise ValueError("cell must be provided")
947
+ if self.expected_colspan < 1:
948
+ raise ValueError("expected_colspan must be >= 1")
949
+
950
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
951
+ """Check if cell has expected colspan attribute."""
952
+ grids = find_all_html_tables(content)
953
+ if not grids:
954
+ return False, "No HTML tables found in content"
955
+
956
+ # Step 1: Find the correct table using anchor cells if provided
957
+ if self.table_anchor_cells:
958
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
959
+ if anchor_result.grid is not None:
960
+ grids = [anchor_result.grid]
961
+ elif anchor_result.is_ambiguous:
962
+ return False, (
963
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
964
+ f"tables - could not uniquely identify target table"
965
+ )
966
+ else:
967
+ return False, "Table anchor cells not found in any table"
968
+
969
+ # Step 2: Find cell and check colspan
970
+ match = find_cell_in_grids(grids, self.cell)
971
+ if not match:
972
+ return False, f"Cell '{self.cell}' not found in target table"
973
+
974
+ grid, cell, row_idx, col_idx = match
975
+
976
+ if cell.colspan == self.expected_colspan:
977
+ return True, f"Cell '{self.cell}' has correct colspan={cell.colspan}"
978
+ else:
979
+ return (
980
+ False,
981
+ f"Cell '{self.cell}' has colspan={cell.colspan}, expected {self.expected_colspan}",
982
+ )
983
+
984
+
985
+ class TableRowspanRule(ParseTestRule):
986
+ """Test rule to verify a cell has the expected rowspan attribute."""
987
+
988
+ def __init__(self, rule_data: ParseTableRowspanRule | dict):
989
+ super().__init__(rule_data)
990
+ rule_data = cast(ParseTableRowspanRule, self._rule_data)
991
+
992
+ if self.type != TestType.TABLE_ROWSPAN.value:
993
+ raise ValueError(f"Invalid type for TableRowspanRule: {self.type}")
994
+
995
+ self.cell = normalize_text(rule_data.cell)
996
+ self.expected_rowspan = rule_data.expected_rowspan
997
+ self.table_anchor_cells = rule_data.table_anchor_cells
998
+
999
+ if not self.cell:
1000
+ raise ValueError("cell must be provided")
1001
+ if self.expected_rowspan < 1:
1002
+ raise ValueError("expected_rowspan must be >= 1")
1003
+
1004
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1005
+ """Check if cell has expected rowspan attribute."""
1006
+ grids = find_all_html_tables(content)
1007
+ if not grids:
1008
+ return False, "No HTML tables found in content"
1009
+
1010
+ # Step 1: Find the correct table using anchor cells if provided
1011
+ if self.table_anchor_cells:
1012
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1013
+ if anchor_result.grid is not None:
1014
+ grids = [anchor_result.grid]
1015
+ elif anchor_result.is_ambiguous:
1016
+ return False, (
1017
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1018
+ f"tables - could not uniquely identify target table"
1019
+ )
1020
+ else:
1021
+ return False, "Table anchor cells not found in any table"
1022
+
1023
+ # Step 2: Find cell and check rowspan
1024
+ match = find_cell_in_grids(grids, self.cell)
1025
+ if not match:
1026
+ return False, f"Cell '{self.cell}' not found in target table"
1027
+
1028
+ grid, cell, row_idx, col_idx = match
1029
+
1030
+ if cell.rowspan == self.expected_rowspan:
1031
+ return True, f"Cell '{self.cell}' has correct rowspan={cell.rowspan}"
1032
+ else:
1033
+ return (
1034
+ False,
1035
+ f"Cell '{self.cell}' has rowspan={cell.rowspan}, expected {self.expected_rowspan}",
1036
+ )
1037
+
1038
+
1039
+ class TableSameRowRule(ParseTestRule):
1040
+ """Test rule to verify two cells share a logical row (considering rowspan)."""
1041
+
1042
+ def __init__(self, rule_data: ParseTableSameRowRule | dict):
1043
+ super().__init__(rule_data)
1044
+ rule_data = cast(ParseTableSameRowRule, self._rule_data)
1045
+
1046
+ if self.type != TestType.TABLE_SAME_ROW.value:
1047
+ raise ValueError(f"Invalid type for TableSameRowRule: {self.type}")
1048
+
1049
+ self.cell_a = normalize_text(rule_data.cell_a)
1050
+ self.cell_b = normalize_text(rule_data.cell_b)
1051
+ self.table_anchor_cells = rule_data.table_anchor_cells
1052
+
1053
+ if not self.cell_a or not self.cell_b:
1054
+ raise ValueError("Both cell_a and cell_b must be provided")
1055
+
1056
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1057
+ """Check if two cells share a logical row."""
1058
+ grids = find_all_html_tables(content)
1059
+ if not grids:
1060
+ return False, "No HTML tables found in content"
1061
+
1062
+ # Step 1: Find the correct table using anchor cells if provided
1063
+ if self.table_anchor_cells:
1064
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1065
+ if anchor_result.grid is not None:
1066
+ grids = [anchor_result.grid]
1067
+ elif anchor_result.is_ambiguous:
1068
+ return False, (
1069
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1070
+ f"tables - could not uniquely identify target table"
1071
+ )
1072
+ else:
1073
+ return False, "Table anchor cells not found in any table"
1074
+
1075
+ match_a = find_cell_in_grids(grids, self.cell_a)
1076
+ if not match_a:
1077
+ return False, f"Cell '{self.cell_a}' not found in target table"
1078
+
1079
+ match_b = find_cell_in_grids(grids, self.cell_b)
1080
+ if not match_b:
1081
+ return False, f"Cell '{self.cell_b}' not found in target table"
1082
+
1083
+ grid_a, cell_a, row_a, col_a = match_a
1084
+ grid_b, cell_b, row_b, col_b = match_b
1085
+
1086
+ # Must be in the same table
1087
+ if grid_a is not grid_b:
1088
+ return False, "Cells are in different tables"
1089
+
1090
+ # Calculate row ranges for each cell (considering rowspan)
1091
+ rows_a = set(range(cell_a.original_row, cell_a.original_row + cell_a.rowspan))
1092
+ rows_b = set(range(cell_b.original_row, cell_b.original_row + cell_b.rowspan))
1093
+
1094
+ if rows_a & rows_b: # Intersection
1095
+ return True, f"Cells share rows: {rows_a & rows_b}"
1096
+ else:
1097
+ return False, f"Cells do not share any row. A: rows {rows_a}, B: rows {rows_b}"
1098
+
1099
+
1100
+ class TableSameColumnRule(ParseTestRule):
1101
+ """Test rule to verify two cells share a logical column (considering colspan)."""
1102
+
1103
+ def __init__(self, rule_data: ParseTableSameColumnRule | dict):
1104
+ super().__init__(rule_data)
1105
+ rule_data = cast(ParseTableSameColumnRule, self._rule_data)
1106
+
1107
+ if self.type != TestType.TABLE_SAME_COLUMN.value:
1108
+ raise ValueError(f"Invalid type for TableSameColumnRule: {self.type}")
1109
+
1110
+ self.cell_a = normalize_text(rule_data.cell_a)
1111
+ self.cell_b = normalize_text(rule_data.cell_b)
1112
+ self.table_anchor_cells = rule_data.table_anchor_cells
1113
+
1114
+ if not self.cell_a or not self.cell_b:
1115
+ raise ValueError("Both cell_a and cell_b must be provided")
1116
+
1117
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1118
+ """Check if two cells share a logical column."""
1119
+ grids = find_all_html_tables(content)
1120
+ if not grids:
1121
+ return False, "No HTML tables found in content"
1122
+
1123
+ # Step 1: Find the correct table using anchor cells if provided
1124
+ if self.table_anchor_cells:
1125
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1126
+ if anchor_result.grid is not None:
1127
+ grids = [anchor_result.grid]
1128
+ elif anchor_result.is_ambiguous:
1129
+ return False, (
1130
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1131
+ f"tables - could not uniquely identify target table"
1132
+ )
1133
+ else:
1134
+ return False, "Table anchor cells not found in any table"
1135
+
1136
+ match_a = find_cell_in_grids(grids, self.cell_a)
1137
+ if not match_a:
1138
+ return False, f"Cell '{self.cell_a}' not found in target table"
1139
+
1140
+ match_b = find_cell_in_grids(grids, self.cell_b)
1141
+ if not match_b:
1142
+ return False, f"Cell '{self.cell_b}' not found in target table"
1143
+
1144
+ grid_a, cell_a, row_a, col_a = match_a
1145
+ grid_b, cell_b, row_b, col_b = match_b
1146
+
1147
+ # Must be in the same table
1148
+ if grid_a is not grid_b:
1149
+ return False, "Cells are in different tables"
1150
+
1151
+ # Calculate column ranges for each cell (considering colspan)
1152
+ cols_a = set(range(cell_a.original_col, cell_a.original_col + cell_a.colspan))
1153
+ cols_b = set(range(cell_b.original_col, cell_b.original_col + cell_b.colspan))
1154
+
1155
+ if cols_a & cols_b: # Intersection
1156
+ return True, f"Cells share columns: {cols_a & cols_b}"
1157
+ else:
1158
+ return False, f"Cells do not share any column. A: cols {cols_a}, B: cols {cols_b}"
1159
+
1160
+
1161
+ class TableHeaderChainRule(ParseTestRule):
1162
+ """Test rule to verify a data cell has the correct header chain."""
1163
+
1164
+ def __init__(self, rule_data: ParseTableHeaderChainRule | dict):
1165
+ super().__init__(rule_data)
1166
+ rule_data = cast(ParseTableHeaderChainRule, self._rule_data)
1167
+
1168
+ if self.type != TestType.TABLE_HEADER_CHAIN.value:
1169
+ raise ValueError(f"Invalid type for TableHeaderChainRule: {self.type}")
1170
+
1171
+ self.data_cell = normalize_text(rule_data.data_cell)
1172
+ self.column_headers = rule_data.column_headers
1173
+ self.row_headers = rule_data.row_headers
1174
+ self.table_anchor_cells = rule_data.table_anchor_cells
1175
+
1176
+ if not self.data_cell:
1177
+ raise ValueError("data_cell must be provided")
1178
+ if not self.column_headers and not self.row_headers:
1179
+ raise ValueError("At least one of column_headers or row_headers must be provided")
1180
+
1181
+ def _get_column_headers(self, grid: ResolvedGrid, data_row: int, data_col: int) -> list[str]:
1182
+ """Get all column headers above the data cell."""
1183
+ headers = []
1184
+ seen_cells: set[tuple[int, int]] = set()
1185
+
1186
+ for row_idx in range(data_row):
1187
+ cell = grid.cells[row_idx][data_col]
1188
+ if cell is None:
1189
+ continue
1190
+ # Use original position as key to avoid duplicates
1191
+ cell_key = (cell.original_row, cell.original_col)
1192
+ if cell_key in seen_cells:
1193
+ continue
1194
+ seen_cells.add(cell_key)
1195
+
1196
+ if cell.text:
1197
+ headers.append(cell.text)
1198
+
1199
+ return headers
1200
+
1201
+ def _get_row_headers(self, grid: ResolvedGrid, data_row: int, data_col: int) -> list[str]:
1202
+ """Get all row headers to the left of the data cell."""
1203
+ headers = []
1204
+ seen_cells: set[tuple[int, int]] = set()
1205
+
1206
+ for col_idx in range(data_col):
1207
+ cell = grid.cells[data_row][col_idx]
1208
+ if cell is None:
1209
+ continue
1210
+ # Use original position as key to avoid duplicates
1211
+ cell_key = (cell.original_row, cell.original_col)
1212
+ if cell_key in seen_cells:
1213
+ continue
1214
+ seen_cells.add(cell_key)
1215
+
1216
+ if cell.text:
1217
+ headers.append(cell.text)
1218
+
1219
+ return headers
1220
+
1221
+ def _fuzzy_list_match(self, expected: list[str], actual: list[str], threshold: float = 0.8) -> tuple[bool, str]:
1222
+ """Check if two lists match using fuzzy matching."""
1223
+ if len(expected) != len(actual):
1224
+ return (
1225
+ False,
1226
+ f"Length mismatch: expected {len(expected)} headers, got {len(actual)}",
1227
+ )
1228
+
1229
+ for i, (exp, act) in enumerate(zip(expected, actual, strict=False)):
1230
+ exp_norm = normalize_text(exp)
1231
+ act_norm = normalize_text(act)
1232
+ similarity = fuzz.ratio(exp_norm, act_norm) / 100.0
1233
+ if similarity < threshold:
1234
+ return (
1235
+ False,
1236
+ f"Header {i} mismatch: expected '{exp}', got '{act}' (similarity: {similarity:.2f})",
1237
+ )
1238
+
1239
+ return True, ""
1240
+
1241
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1242
+ """Check if data cell has correct header chain."""
1243
+ grids = find_all_html_tables(content)
1244
+ if not grids:
1245
+ return False, "No HTML tables found in content"
1246
+
1247
+ # Step 1: Find the correct table using anchor cells if provided
1248
+ if self.table_anchor_cells:
1249
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1250
+ if anchor_result.grid is not None:
1251
+ grids = [anchor_result.grid]
1252
+ elif anchor_result.is_ambiguous:
1253
+ return False, (
1254
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1255
+ f"tables - could not uniquely identify target table"
1256
+ )
1257
+ else:
1258
+ return False, "Table anchor cells not found in any table"
1259
+
1260
+ match = find_cell_in_grids(grids, self.data_cell)
1261
+ if not match:
1262
+ return False, f"Data cell '{self.data_cell}' not found in target table"
1263
+
1264
+ grid, cell, row_idx, col_idx = match
1265
+
1266
+ errors = []
1267
+
1268
+ # Check column headers if expected
1269
+ if self.column_headers:
1270
+ actual_col_headers = self._get_column_headers(grid, row_idx, col_idx)
1271
+ passed, err = self._fuzzy_list_match(self.column_headers, actual_col_headers)
1272
+ if not passed:
1273
+ errors.append(f"Column headers: {err}. Expected: {self.column_headers}, Got: {actual_col_headers}")
1274
+
1275
+ # Check row headers if expected
1276
+ if self.row_headers:
1277
+ actual_row_headers = self._get_row_headers(grid, row_idx, col_idx)
1278
+ passed, err = self._fuzzy_list_match(self.row_headers, actual_row_headers)
1279
+ if not passed:
1280
+ errors.append(f"Row headers: {err}. Expected: {self.row_headers}, Got: {actual_row_headers}")
1281
+
1282
+ if errors:
1283
+ return False, "; ".join(errors)
1284
+ else:
1285
+ return True, f"Header chain verified for '{self.data_cell}'"
1286
+
1287
+
1288
+ # =============================================================================
1289
+ # Table Adjacency and Header Rules
1290
+ # =============================================================================
1291
+
1292
+
1293
+ class TableAdjacentRule(ParseTestRule):
1294
+ """
1295
+ Base class for table adjacency rules.
1296
+
1297
+ Tests that anchor_cell has expected_neighbor in a specific direction.
1298
+ Handles duplicate anchor cells by checking ALL occurrences.
1299
+ """
1300
+
1301
+ def __init__(self, rule_data: AdjacentTableRuleData | dict):
1302
+ super().__init__(rule_data)
1303
+ rule_data = cast(AdjacentTableRuleData, self._rule_data)
1304
+
1305
+ self.anchor_cell = normalize_text(rule_data.anchor_cell)
1306
+ self.expected_neighbor = normalize_text(rule_data.expected_neighbor)
1307
+ self.table_anchor_cells = rule_data.table_anchor_cells
1308
+ self.direction = "" # Set by subclass
1309
+
1310
+ if not self.anchor_cell:
1311
+ raise ValueError("anchor_cell must be provided")
1312
+ if not self.expected_neighbor:
1313
+ raise ValueError("expected_neighbor must be provided")
1314
+
1315
+ def _get_neighbor_position(self, row: int, col: int, grid: ResolvedGrid) -> tuple[int, int] | None:
1316
+ """Get neighbor position based on direction."""
1317
+ if self.direction == "up" and row > 0:
1318
+ return (row - 1, col)
1319
+ elif self.direction == "down" and row < grid.num_rows - 1:
1320
+ return (row + 1, col)
1321
+ elif self.direction == "left" and col > 0:
1322
+ return (row, col - 1)
1323
+ elif self.direction == "right" and col < grid.num_cols - 1:
1324
+ return (row, col + 1)
1325
+ return None
1326
+
1327
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1328
+ grids = find_all_html_tables(content)
1329
+ if not grids:
1330
+ return False, "No HTML tables found"
1331
+
1332
+ # Step 1: Find the correct table using anchor cells if provided
1333
+ if self.table_anchor_cells:
1334
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1335
+ if anchor_result.grid is not None:
1336
+ grids = [anchor_result.grid]
1337
+ elif anchor_result.is_ambiguous:
1338
+ return False, (
1339
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1340
+ f"tables - could not uniquely identify target table"
1341
+ )
1342
+ else:
1343
+ return False, "Table anchor cells not found in any table"
1344
+
1345
+ for grid in grids:
1346
+ for row_idx, row in enumerate(grid.cells):
1347
+ for col_idx, cell in enumerate(row):
1348
+ if cell is None:
1349
+ continue
1350
+ if cell.original_row != row_idx or cell.original_col != col_idx:
1351
+ continue
1352
+
1353
+ similarity = fuzz.ratio(self.anchor_cell, cell.text) / 100.0
1354
+ if similarity < CELL_FUZZY_MATCH_THRESHOLD:
1355
+ continue
1356
+
1357
+ neighbor_pos = self._get_neighbor_position(row_idx, col_idx, grid)
1358
+ if neighbor_pos is None:
1359
+ continue
1360
+
1361
+ neighbor = grid.cells[neighbor_pos[0]][neighbor_pos[1]]
1362
+ if neighbor is None:
1363
+ continue
1364
+
1365
+ neighbor_sim = fuzz.ratio(self.expected_neighbor, neighbor.text) / 100.0
1366
+ if neighbor_sim >= CELL_FUZZY_MATCH_THRESHOLD:
1367
+ return True, ""
1368
+
1369
+ return False, f"No '{self.anchor_cell}' has '{self.expected_neighbor}' {self.direction}"
1370
+
1371
+
1372
+ class TableAdjacentUpRule(TableAdjacentRule):
1373
+ def __init__(self, rule_data: ParseTableAdjacentUpRule | dict):
1374
+ super().__init__(rule_data)
1375
+ rule_data = cast(ParseTableAdjacentUpRule, self._rule_data)
1376
+ if self.type != TestType.TABLE_ADJACENT_UP.value:
1377
+ raise ValueError(f"Invalid type: {self.type}")
1378
+ self.direction = "up"
1379
+
1380
+
1381
+ class TableAdjacentDownRule(TableAdjacentRule):
1382
+ def __init__(self, rule_data: ParseTableAdjacentDownRule | dict):
1383
+ super().__init__(rule_data)
1384
+ rule_data = cast(ParseTableAdjacentDownRule, self._rule_data)
1385
+ if self.type != TestType.TABLE_ADJACENT_DOWN.value:
1386
+ raise ValueError(f"Invalid type: {self.type}")
1387
+ self.direction = "down"
1388
+
1389
+
1390
+ class TableAdjacentLeftRule(TableAdjacentRule):
1391
+ def __init__(self, rule_data: ParseTableAdjacentLeftRule | dict):
1392
+ super().__init__(rule_data)
1393
+ rule_data = cast(ParseTableAdjacentLeftRule, self._rule_data)
1394
+ if self.type != TestType.TABLE_ADJACENT_LEFT.value:
1395
+ raise ValueError(f"Invalid type: {self.type}")
1396
+ self.direction = "left"
1397
+
1398
+
1399
+ class TableAdjacentRightRule(TableAdjacentRule):
1400
+ def __init__(self, rule_data: ParseTableAdjacentRightRule | dict):
1401
+ super().__init__(rule_data)
1402
+ rule_data = cast(ParseTableAdjacentRightRule, self._rule_data)
1403
+ if self.type != TestType.TABLE_ADJACENT_RIGHT.value:
1404
+ raise ValueError(f"Invalid type: {self.type}")
1405
+ self.direction = "right"
1406
+
1407
+
1408
+ class TableTopHeaderRule(ParseTestRule):
1409
+ """
1410
+ Test that a data cell has a specific column header above it.
1411
+
1412
+ Handles duplicate data cells by checking ALL occurrences.
1413
+ """
1414
+
1415
+ def __init__(self, rule_data: ParseTableTopHeaderRule | dict):
1416
+ super().__init__(rule_data)
1417
+ rule_data = cast(ParseTableTopHeaderRule, self._rule_data)
1418
+
1419
+ if self.type != TestType.TABLE_TOP_HEADER.value:
1420
+ raise ValueError(f"Invalid type: {self.type}")
1421
+ self.data_cell = normalize_text(rule_data.data_cell)
1422
+ self.expected_header = normalize_text(rule_data.expected_header)
1423
+ self.table_anchor_cells = rule_data.table_anchor_cells
1424
+
1425
+ if not self.data_cell:
1426
+ raise ValueError("data_cell must be provided")
1427
+ if not self.expected_header:
1428
+ raise ValueError("expected_header must be provided")
1429
+
1430
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1431
+ grids = find_all_html_tables(content)
1432
+ if not grids:
1433
+ return False, "No HTML tables found"
1434
+
1435
+ # Step 1: Find the correct table using anchor cells if provided
1436
+ if self.table_anchor_cells:
1437
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1438
+ if anchor_result.grid is not None:
1439
+ grids = [anchor_result.grid]
1440
+ elif anchor_result.is_ambiguous:
1441
+ return False, (
1442
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1443
+ f"tables - could not uniquely identify target table"
1444
+ )
1445
+ else:
1446
+ return False, "Table anchor cells not found in any table"
1447
+
1448
+ for grid in grids:
1449
+ for row_idx, row in enumerate(grid.cells):
1450
+ for col_idx, cell in enumerate(row):
1451
+ if cell is None:
1452
+ continue
1453
+ if cell.original_row != row_idx or cell.original_col != col_idx:
1454
+ continue
1455
+
1456
+ similarity = fuzz.ratio(self.data_cell, cell.text) / 100.0
1457
+ if similarity < CELL_FUZZY_MATCH_THRESHOLD:
1458
+ continue
1459
+
1460
+ # Look above for header
1461
+ for header_row in range(row_idx):
1462
+ header_cell = grid.cells[header_row][col_idx]
1463
+ if header_cell is None:
1464
+ continue
1465
+
1466
+ header_sim = fuzz.ratio(self.expected_header, header_cell.text) / 100.0
1467
+ if header_sim >= CELL_FUZZY_MATCH_THRESHOLD:
1468
+ return True, ""
1469
+
1470
+ return False, f"No '{self.data_cell}' has header '{self.expected_header}' above"
1471
+
1472
+
1473
+ class TableLeftHeaderRule(ParseTestRule):
1474
+ """
1475
+ Test that a data cell has a specific row header to its left.
1476
+
1477
+ Handles duplicate data cells by checking ALL occurrences.
1478
+ """
1479
+
1480
+ def __init__(self, rule_data: ParseTableLeftHeaderRule | dict):
1481
+ super().__init__(rule_data)
1482
+ rule_data = cast(ParseTableLeftHeaderRule, self._rule_data)
1483
+
1484
+ if self.type != TestType.TABLE_LEFT_HEADER.value:
1485
+ raise ValueError(f"Invalid type: {self.type}")
1486
+ self.data_cell = normalize_text(rule_data.data_cell)
1487
+ self.expected_header = normalize_text(rule_data.expected_header)
1488
+ self.table_anchor_cells = rule_data.table_anchor_cells
1489
+
1490
+ if not self.data_cell:
1491
+ raise ValueError("data_cell must be provided")
1492
+ if not self.expected_header:
1493
+ raise ValueError("expected_header must be provided")
1494
+
1495
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1496
+ grids = find_all_html_tables(content)
1497
+ if not grids:
1498
+ return False, "No HTML tables found"
1499
+
1500
+ # Step 1: Find the correct table using anchor cells if provided
1501
+ if self.table_anchor_cells:
1502
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1503
+ if anchor_result.grid is not None:
1504
+ grids = [anchor_result.grid]
1505
+ elif anchor_result.is_ambiguous:
1506
+ return False, (
1507
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1508
+ f"tables - could not uniquely identify target table"
1509
+ )
1510
+ else:
1511
+ return False, "Table anchor cells not found in any table"
1512
+
1513
+ for grid in grids:
1514
+ for row_idx, row in enumerate(grid.cells):
1515
+ for col_idx, cell in enumerate(row):
1516
+ if cell is None:
1517
+ continue
1518
+ if cell.original_row != row_idx or cell.original_col != col_idx:
1519
+ continue
1520
+
1521
+ similarity = fuzz.ratio(self.data_cell, cell.text) / 100.0
1522
+ if similarity < CELL_FUZZY_MATCH_THRESHOLD:
1523
+ continue
1524
+
1525
+ # Look left for header
1526
+ for header_col in range(col_idx):
1527
+ header_cell = grid.cells[row_idx][header_col]
1528
+ if header_cell is None:
1529
+ continue
1530
+
1531
+ header_sim = fuzz.ratio(self.expected_header, header_cell.text) / 100.0
1532
+ if header_sim >= CELL_FUZZY_MATCH_THRESHOLD:
1533
+ return True, ""
1534
+
1535
+ return False, f"No '{self.data_cell}' has header '{self.expected_header}' to left"
1536
+
1537
+
1538
+ # =============================================================================
1539
+ # Table Border Rules (Negative Tests)
1540
+ # =============================================================================
1541
+
1542
+
1543
+ class TableNoBorderRule(ParseTestRule):
1544
+ """
1545
+ Base class for table border rules that verify absence of cells.
1546
+
1547
+ These are "negative tests" that ensure predicted tables don't have
1548
+ extra rows/columns beyond the ground truth boundaries.
1549
+ """
1550
+
1551
+ def __init__(self, rule_data: NoBorderTableRuleData | dict):
1552
+ super().__init__(rule_data)
1553
+ rule_data = cast(NoBorderTableRuleData, self._rule_data)
1554
+
1555
+ self.cell = normalize_text(rule_data.cell)
1556
+ self.table_anchor_cells = rule_data.table_anchor_cells
1557
+ self.direction = "" # Set by subclass: "left", "right", "up", "down"
1558
+
1559
+ if not self.cell:
1560
+ raise ValueError("cell must be provided")
1561
+
1562
+ def _get_neighbor_position(self, row: int, col: int, grid: ResolvedGrid) -> tuple[int, int] | None:
1563
+ """Get neighbor position based on direction. Returns None if out of bounds."""
1564
+ if self.direction == "up":
1565
+ return (row - 1, col) if row > 0 else None
1566
+ elif self.direction == "down":
1567
+ return (row + 1, col) if row < grid.num_rows - 1 else None
1568
+ elif self.direction == "left":
1569
+ return (row, col - 1) if col > 0 else None
1570
+ elif self.direction == "right":
1571
+ return (row, col + 1) if col < grid.num_cols - 1 else None
1572
+ return None
1573
+
1574
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
1575
+ grids = find_all_html_tables(content)
1576
+ if not grids:
1577
+ return False, "No HTML tables found"
1578
+
1579
+ # Find the correct table using anchor cells if provided
1580
+ if self.table_anchor_cells:
1581
+ anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
1582
+ if anchor_result.grid is not None:
1583
+ grids = [anchor_result.grid]
1584
+ elif anchor_result.is_ambiguous:
1585
+ return False, (
1586
+ f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
1587
+ f"tables - could not uniquely identify target table"
1588
+ )
1589
+ else:
1590
+ return False, "Table anchor cells not found in any table"
1591
+
1592
+ for grid in grids:
1593
+ for row_idx, row in enumerate(grid.cells):
1594
+ for col_idx, cell in enumerate(row):
1595
+ if cell is None:
1596
+ continue
1597
+ if cell.original_row != row_idx or cell.original_col != col_idx:
1598
+ continue
1599
+
1600
+ similarity = fuzz.ratio(self.cell, cell.text) / 100.0
1601
+ if similarity < CELL_FUZZY_MATCH_THRESHOLD:
1602
+ continue
1603
+
1604
+ # Found the cell - now check if there's NO neighbor in the direction
1605
+ neighbor_pos = self._get_neighbor_position(row_idx, col_idx, grid)
1606
+
1607
+ if neighbor_pos is None:
1608
+ # No neighbor position possible (at grid boundary) - PASS
1609
+ return True, ""
1610
+
1611
+ neighbor = grid.cells[neighbor_pos[0]][neighbor_pos[1]]
1612
+ if neighbor is None or not neighbor.text.strip():
1613
+ # No neighbor cell or empty cell - PASS
1614
+ return True, ""
1615
+
1616
+ # There IS a neighbor - this is a FAILURE for border tests
1617
+ return (
1618
+ False,
1619
+ f"Cell '{self.cell}' has unexpected neighbor '{neighbor.text}' to {self.direction}",
1620
+ )
1621
+
1622
+ return False, f"Could not find cell '{self.cell}' in any table"
1623
+
1624
+
1625
+ class TableNoLeftRule(TableNoBorderRule):
1626
+ """Test that a cell has no cell to its left (leftmost column boundary)."""
1627
+
1628
+ def __init__(self, rule_data: ParseTableNoLeftRule | dict):
1629
+ super().__init__(rule_data)
1630
+ rule_data = cast(ParseTableNoLeftRule, self._rule_data)
1631
+ if self.type != TestType.TABLE_NO_LEFT.value:
1632
+ raise ValueError(f"Invalid type: {self.type}")
1633
+ self.direction = "left"
1634
+
1635
+
1636
+ class TableNoRightRule(TableNoBorderRule):
1637
+ """Test that a cell has no cell to its right (rightmost column boundary)."""
1638
+
1639
+ def __init__(self, rule_data: ParseTableNoRightRule | dict):
1640
+ super().__init__(rule_data)
1641
+ rule_data = cast(ParseTableNoRightRule, self._rule_data)
1642
+ if self.type != TestType.TABLE_NO_RIGHT.value:
1643
+ raise ValueError(f"Invalid type: {self.type}")
1644
+ self.direction = "right"
1645
+
1646
+
1647
+ class TableNoAboveRule(TableNoBorderRule):
1648
+ """Test that a cell has no cell above it (top row boundary)."""
1649
+
1650
+ def __init__(self, rule_data: ParseTableNoAboveRule | dict):
1651
+ super().__init__(rule_data)
1652
+ rule_data = cast(ParseTableNoAboveRule, self._rule_data)
1653
+ if self.type != TestType.TABLE_NO_ABOVE.value:
1654
+ raise ValueError(f"Invalid type: {self.type}")
1655
+ self.direction = "up"
1656
+
1657
+
1658
+ class TableNoBelowRule(TableNoBorderRule):
1659
+ """Test that a cell has no cell below it (bottom row boundary)."""
1660
+
1661
+ def __init__(self, rule_data: ParseTableNoBelowRule | dict):
1662
+ super().__init__(rule_data)
1663
+ rule_data = cast(ParseTableNoBelowRule, self._rule_data)
1664
+ if self.type != TestType.TABLE_NO_BELOW.value:
1665
+ raise ValueError(f"Invalid type: {self.type}")
1666
+ self.direction = "down"