parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,1666 @@
|
|
|
1
|
+
"""Table structure and hierarchy test rules."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
import unicodedata
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from html import unescape
|
|
8
|
+
from typing import Any, cast
|
|
9
|
+
|
|
10
|
+
import pandas as pd
|
|
11
|
+
from bs4 import BeautifulSoup
|
|
12
|
+
from rapidfuzz import fuzz
|
|
13
|
+
from unidecode import unidecode
|
|
14
|
+
|
|
15
|
+
from parse_bench.evaluation.metrics.parse.rules_base import (
|
|
16
|
+
CELL_FUZZY_MATCH_THRESHOLD,
|
|
17
|
+
AdjacentTableRuleData,
|
|
18
|
+
NoBorderTableRuleData,
|
|
19
|
+
ParseTestRule,
|
|
20
|
+
)
|
|
21
|
+
from parse_bench.evaluation.metrics.parse.table_parsing import (
|
|
22
|
+
ResolvedGrid,
|
|
23
|
+
TableData,
|
|
24
|
+
find_all_html_tables,
|
|
25
|
+
find_cell_in_grids,
|
|
26
|
+
find_table_by_anchors,
|
|
27
|
+
parse_html_tables,
|
|
28
|
+
parse_markdown_tables,
|
|
29
|
+
)
|
|
30
|
+
from parse_bench.evaluation.metrics.parse.test_types import TestType
|
|
31
|
+
from parse_bench.evaluation.metrics.parse.utils import normalize_text
|
|
32
|
+
from parse_bench.test_cases.parse_rule_schemas import (
|
|
33
|
+
ParseTableAdjacentDownRule,
|
|
34
|
+
ParseTableAdjacentLeftRule,
|
|
35
|
+
ParseTableAdjacentRightRule,
|
|
36
|
+
ParseTableAdjacentUpRule,
|
|
37
|
+
ParseTableColspanRule,
|
|
38
|
+
ParseTableHeaderChainRule,
|
|
39
|
+
ParseTableLeftHeaderRule,
|
|
40
|
+
ParseTableMarkerCellsRule,
|
|
41
|
+
ParseTableNoAboveRule,
|
|
42
|
+
ParseTableNoBelowRule,
|
|
43
|
+
ParseTableNoLeftRule,
|
|
44
|
+
ParseTableNoRightRule,
|
|
45
|
+
ParseTableRowspanRule,
|
|
46
|
+
ParseTableRule,
|
|
47
|
+
ParseTableSameColumnRule,
|
|
48
|
+
ParseTableSameRowRule,
|
|
49
|
+
ParseTablesNumColsRule,
|
|
50
|
+
ParseTablesNumRowsRule,
|
|
51
|
+
ParseTablesValuesRule,
|
|
52
|
+
ParseTableTopHeaderRule,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
_DEFAULT_TABLE_MARKER_ALIASES = {
|
|
56
|
+
"x",
|
|
57
|
+
"y",
|
|
58
|
+
"n",
|
|
59
|
+
"yes",
|
|
60
|
+
"no",
|
|
61
|
+
"check",
|
|
62
|
+
"checked",
|
|
63
|
+
"✓",
|
|
64
|
+
"✔",
|
|
65
|
+
"☑",
|
|
66
|
+
"✅",
|
|
67
|
+
"✗",
|
|
68
|
+
"✘",
|
|
69
|
+
"×",
|
|
70
|
+
"•",
|
|
71
|
+
"●",
|
|
72
|
+
"∙",
|
|
73
|
+
"○",
|
|
74
|
+
"◉",
|
|
75
|
+
"◦",
|
|
76
|
+
"■",
|
|
77
|
+
"□",
|
|
78
|
+
"▪",
|
|
79
|
+
"▫",
|
|
80
|
+
"◼",
|
|
81
|
+
"◻",
|
|
82
|
+
"◆",
|
|
83
|
+
"◇",
|
|
84
|
+
"♦",
|
|
85
|
+
"★",
|
|
86
|
+
"☆",
|
|
87
|
+
"→",
|
|
88
|
+
"←",
|
|
89
|
+
"↑",
|
|
90
|
+
"↓",
|
|
91
|
+
"▲",
|
|
92
|
+
"▼",
|
|
93
|
+
"+",
|
|
94
|
+
"!",
|
|
95
|
+
"selected",
|
|
96
|
+
"green",
|
|
97
|
+
"red",
|
|
98
|
+
"yellow",
|
|
99
|
+
"orange",
|
|
100
|
+
"grey",
|
|
101
|
+
"gray",
|
|
102
|
+
"green square",
|
|
103
|
+
"red square",
|
|
104
|
+
"yellow square",
|
|
105
|
+
"orange square",
|
|
106
|
+
"grey square",
|
|
107
|
+
"gray square",
|
|
108
|
+
"expert knowledge",
|
|
109
|
+
"good knowledge",
|
|
110
|
+
"basic knowledge",
|
|
111
|
+
}
|
|
112
|
+
_MARKDOWN_IMAGE_RE = re.compile(r"!\[[^\]]*\]\([^)]*\)", re.DOTALL)
|
|
113
|
+
_BRACKETED_IMAGE_LABEL_RE = re.compile(r"\[\s*(?:icon|image)(?::[^\]]+)?\s*\]", re.IGNORECASE)
|
|
114
|
+
_HTML_IMAGE_RE = re.compile(r"<img\b[^>]*>", re.IGNORECASE)
|
|
115
|
+
_HTML_TAG_RE = re.compile(r"<[^>]+>")
|
|
116
|
+
_HTML_IMAGE_CELL_SENTINEL = "llamacloud-bench-image-cell"
|
|
117
|
+
_MARKER_LIST_TOKEN = "<marker-list>"
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _preserve_html_image_cells(content: str) -> str:
|
|
121
|
+
"""Replace images inside HTML cells with text visible to ``parse_html_tables``.
|
|
122
|
+
|
|
123
|
+
The generic HTML table parser intentionally extracts text with
|
|
124
|
+
``BeautifulSoup.get_text()``, which drops ``<img>`` elements. Work on a
|
|
125
|
+
private copy here so image presence is retained for this rule without
|
|
126
|
+
changing cell values seen by every other table evaluator.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
if not _HTML_IMAGE_RE.search(content):
|
|
130
|
+
return content
|
|
131
|
+
soup = BeautifulSoup(content, "lxml")
|
|
132
|
+
changed = False
|
|
133
|
+
for cell in soup.find_all(["td", "th"]):
|
|
134
|
+
if cell.find("img") is None:
|
|
135
|
+
continue
|
|
136
|
+
cell.clear()
|
|
137
|
+
cell.append(_HTML_IMAGE_CELL_SENTINEL)
|
|
138
|
+
changed = True
|
|
139
|
+
return str(soup) if changed else content
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _marker_cell_token(value: object, aliases: set[str], allow_ocr_glyphs: bool) -> str | None:
|
|
143
|
+
raw = unescape(str(value)).strip()
|
|
144
|
+
if not raw:
|
|
145
|
+
return None
|
|
146
|
+
if (
|
|
147
|
+
_HTML_IMAGE_CELL_SENTINEL in raw
|
|
148
|
+
or _MARKDOWN_IMAGE_RE.search(raw)
|
|
149
|
+
or _BRACKETED_IMAGE_LABEL_RE.search(raw)
|
|
150
|
+
or _HTML_IMAGE_RE.search(raw)
|
|
151
|
+
):
|
|
152
|
+
return "<image>"
|
|
153
|
+
|
|
154
|
+
token = " ".join(_HTML_TAG_RE.sub("", raw).split()).casefold()
|
|
155
|
+
if not token:
|
|
156
|
+
return None
|
|
157
|
+
if token in aliases:
|
|
158
|
+
return token
|
|
159
|
+
marker_list = [part.strip() for part in re.split(r"[,;/]", token)]
|
|
160
|
+
if len(marker_list) > 1 and all(part in aliases for part in marker_list):
|
|
161
|
+
return _MARKER_LIST_TOKEN
|
|
162
|
+
# Markers are sometimes serialized beside their numeric or textual value,
|
|
163
|
+
# for example ``▲ 512.40`` or ``✓ Announced``. Only accept a leading
|
|
164
|
+
# alias when it is punctuation/symbol-like; ordinary word aliases must
|
|
165
|
+
# still occupy the whole cell to avoid matching prose.
|
|
166
|
+
leading_token = token.split(maxsplit=1)[0]
|
|
167
|
+
if leading_token in aliases and all(unicodedata.category(char)[0] in {"P", "S"} for char in leading_token):
|
|
168
|
+
return leading_token
|
|
169
|
+
if len(token) >= 3 and (token[0], token[-1]) in {("[", "]"), ("(", ")"), ("{", "}")}:
|
|
170
|
+
inner = token[1:-1].strip()
|
|
171
|
+
if inner in aliases:
|
|
172
|
+
return inner
|
|
173
|
+
if not allow_ocr_glyphs or len(token) > 3 or any(char.isspace() or char.isdigit() for char in token):
|
|
174
|
+
return None
|
|
175
|
+
|
|
176
|
+
# Visual markers are frequently OCR'd as a non-ASCII glyph (for example
|
|
177
|
+
# Harley-Davidson's badge becomes Cyrillic Ө). Accept short non-ASCII
|
|
178
|
+
# glyphs, plus Unicode symbols/punctuation, but leave ordinary ASCII words
|
|
179
|
+
# and punctuation out so abbreviations and missing-value dashes do not
|
|
180
|
+
# become false markers. ASCII symbols must be configured as aliases.
|
|
181
|
+
if any(ord(char) > 127 for char in token):
|
|
182
|
+
return token
|
|
183
|
+
return None
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class TableMarkerCellsRule(ParseTestRule):
|
|
187
|
+
"""Check that repeated icon-like values remain inside a table grid."""
|
|
188
|
+
|
|
189
|
+
def __init__(self, rule_data: ParseTableMarkerCellsRule | dict):
|
|
190
|
+
super().__init__(rule_data)
|
|
191
|
+
rule_data = cast(ParseTableMarkerCellsRule, self._rule_data)
|
|
192
|
+
if self.type != TestType.TABLE_MARKER_CELLS.value:
|
|
193
|
+
raise ValueError(f"Invalid type for TableMarkerCellsRule: {self.type}")
|
|
194
|
+
self.min_count = rule_data.min_count
|
|
195
|
+
self.min_distinct_rows = rule_data.min_distinct_rows
|
|
196
|
+
self.min_distinct_columns = rule_data.min_distinct_columns
|
|
197
|
+
self.marker_aliases = {
|
|
198
|
+
" ".join(unescape(alias).split()).casefold()
|
|
199
|
+
for alias in [*_DEFAULT_TABLE_MARKER_ALIASES, *rule_data.marker_aliases]
|
|
200
|
+
if alias.strip()
|
|
201
|
+
}
|
|
202
|
+
self.allow_repeated_ocr_glyphs = rule_data.allow_repeated_ocr_glyphs
|
|
203
|
+
|
|
204
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
205
|
+
tables = [*parse_markdown_tables(content), *parse_html_tables(_preserve_html_image_cells(content))]
|
|
206
|
+
if not tables:
|
|
207
|
+
self.result_details = {
|
|
208
|
+
"requirement": self._requirement_summary(),
|
|
209
|
+
"tables_inspected": 0,
|
|
210
|
+
"diagnosis": "The parser emitted no recognizable Markdown or HTML table.",
|
|
211
|
+
}
|
|
212
|
+
return False, "No tables found; icon-valued cells could not be evaluated"
|
|
213
|
+
|
|
214
|
+
best = (0, 0, 0)
|
|
215
|
+
best_table: dict[str, Any] = {}
|
|
216
|
+
for table_index, table in enumerate(tables, start=1):
|
|
217
|
+
positions_by_token: dict[str, list[tuple[int, int]]] = {}
|
|
218
|
+
for row in range(table.data.shape[0]):
|
|
219
|
+
for column in range(table.data.shape[1]):
|
|
220
|
+
token = _marker_cell_token(
|
|
221
|
+
table.data[row, column],
|
|
222
|
+
self.marker_aliases,
|
|
223
|
+
self.allow_repeated_ocr_glyphs,
|
|
224
|
+
)
|
|
225
|
+
if token is not None:
|
|
226
|
+
positions_by_token.setdefault(token, []).append((row, column))
|
|
227
|
+
|
|
228
|
+
# A short OCR glyph only counts as a marker when it repeats. Known
|
|
229
|
+
# aliases and image-only cells are already semantically explicit.
|
|
230
|
+
positions = [
|
|
231
|
+
position
|
|
232
|
+
for token, token_positions in positions_by_token.items()
|
|
233
|
+
if token in self.marker_aliases or token in {"<image>", _MARKER_LIST_TOKEN} or len(token_positions) >= 2
|
|
234
|
+
for position in token_positions
|
|
235
|
+
]
|
|
236
|
+
score = (
|
|
237
|
+
len(positions),
|
|
238
|
+
len({row for row, _ in positions}),
|
|
239
|
+
len({column for _, column in positions}),
|
|
240
|
+
)
|
|
241
|
+
table_details = {
|
|
242
|
+
"index": table_index,
|
|
243
|
+
"shape": f"{table.data.shape[0]} rows x {table.data.shape[1]} columns",
|
|
244
|
+
"marker_cells": score[0],
|
|
245
|
+
"marker_rows": score[1],
|
|
246
|
+
"marker_columns": score[2],
|
|
247
|
+
"recognized_tokens": {
|
|
248
|
+
token: len(token_positions) for token, token_positions in sorted(positions_by_token.items())
|
|
249
|
+
},
|
|
250
|
+
}
|
|
251
|
+
if score > best or not best_table:
|
|
252
|
+
best = score
|
|
253
|
+
best_table = table_details
|
|
254
|
+
if (
|
|
255
|
+
score[0] >= self.min_count
|
|
256
|
+
and score[1] >= self.min_distinct_rows
|
|
257
|
+
and score[2] >= self.min_distinct_columns
|
|
258
|
+
):
|
|
259
|
+
self.result_details = {
|
|
260
|
+
"requirement": self._requirement_summary(),
|
|
261
|
+
"tables_inspected": len(tables),
|
|
262
|
+
"matching_table": table_details,
|
|
263
|
+
}
|
|
264
|
+
return True, ""
|
|
265
|
+
|
|
266
|
+
diagnosis = self._failure_diagnosis(best)
|
|
267
|
+
self.result_details = {
|
|
268
|
+
"requirement": self._requirement_summary(),
|
|
269
|
+
"tables_inspected": len(tables),
|
|
270
|
+
"best_table": best_table,
|
|
271
|
+
"diagnosis": diagnosis,
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
return (
|
|
275
|
+
False,
|
|
276
|
+
f"Icon-table rule failed: {diagnosis} "
|
|
277
|
+
f"Best table had {best[0]} marker cells across {best[1]} rows and {best[2]} columns; "
|
|
278
|
+
f"required {self._requirement_summary()}.",
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
def _requirement_summary(self) -> str:
|
|
282
|
+
return (
|
|
283
|
+
f">={self.min_count} marker cells across >={self.min_distinct_rows} rows "
|
|
284
|
+
f"and >={self.min_distinct_columns} columns"
|
|
285
|
+
)
|
|
286
|
+
|
|
287
|
+
def _failure_diagnosis(self, best: tuple[int, int, int]) -> str:
|
|
288
|
+
if best[0] < self.min_count:
|
|
289
|
+
if best[0] == 0:
|
|
290
|
+
return "tables were found, but no repeated recognized marker-only values remained in their cells."
|
|
291
|
+
return f"only {best[0]} recognized marker cells remained; at least {self.min_count} are required."
|
|
292
|
+
if best[1] < self.min_distinct_rows:
|
|
293
|
+
return f"markers occupied only {best[1]} table rows; at least {self.min_distinct_rows} are required."
|
|
294
|
+
return f"markers occupied only {best[2]} table column(s); at least {self.min_distinct_columns} are required."
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
class TableRule(ParseTestRule):
|
|
298
|
+
"""Test rule to verify table cell relationships."""
|
|
299
|
+
|
|
300
|
+
def __init__(self, rule_data: ParseTableRule | dict):
|
|
301
|
+
super().__init__(rule_data)
|
|
302
|
+
rule_data = cast(ParseTableRule, self._rule_data)
|
|
303
|
+
|
|
304
|
+
if self.type != TestType.TABLE.value:
|
|
305
|
+
raise ValueError(f"Invalid type for TableRule: {self.type}")
|
|
306
|
+
|
|
307
|
+
# Normalize the search text
|
|
308
|
+
self.cell = normalize_text(rule_data.cell)
|
|
309
|
+
self.up = normalize_text(rule_data.up or "")
|
|
310
|
+
self.down = normalize_text(rule_data.down or "")
|
|
311
|
+
self.left = normalize_text(rule_data.left or "")
|
|
312
|
+
self.right = normalize_text(rule_data.right or "")
|
|
313
|
+
self.top_heading = normalize_text(rule_data.top_heading or "")
|
|
314
|
+
self.left_heading = normalize_text(rule_data.left_heading or "")
|
|
315
|
+
self.ignore_markdown_tables = rule_data.ignore_markdown_tables
|
|
316
|
+
|
|
317
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
318
|
+
"""Check if table cell relationships are satisfied."""
|
|
319
|
+
tables_to_check = []
|
|
320
|
+
failed_reasons = []
|
|
321
|
+
|
|
322
|
+
# Threshold for fuzzy matching derived from max_diffs
|
|
323
|
+
threshold = 1.0 - (self.max_diffs / (len(self.cell) if len(self.cell) > 0 else 1))
|
|
324
|
+
threshold = max(0.5, threshold)
|
|
325
|
+
|
|
326
|
+
# Parse tables
|
|
327
|
+
if not self.ignore_markdown_tables:
|
|
328
|
+
md_tables = parse_markdown_tables(content)
|
|
329
|
+
tables_to_check.extend(md_tables)
|
|
330
|
+
|
|
331
|
+
html_tables = parse_html_tables(content)
|
|
332
|
+
tables_to_check.extend(html_tables)
|
|
333
|
+
|
|
334
|
+
# If no tables found, return failure
|
|
335
|
+
if not tables_to_check:
|
|
336
|
+
return False, "No tables found in the content"
|
|
337
|
+
|
|
338
|
+
# Check each table
|
|
339
|
+
for table_data in tables_to_check:
|
|
340
|
+
table_array = table_data.data
|
|
341
|
+
header_rows = table_data.header_rows
|
|
342
|
+
header_cols = table_data.header_cols
|
|
343
|
+
|
|
344
|
+
# Find all cells that match the target cell using fuzzy matching
|
|
345
|
+
matches = []
|
|
346
|
+
for i in range(table_array.shape[0]):
|
|
347
|
+
for j in range(table_array.shape[1]):
|
|
348
|
+
cell_content = normalize_text(str(table_array[i, j]))
|
|
349
|
+
similarity = fuzz.ratio(self.cell, cell_content) / 100.0
|
|
350
|
+
|
|
351
|
+
if similarity >= threshold:
|
|
352
|
+
matches.append((i, j))
|
|
353
|
+
|
|
354
|
+
# If no matches found in this table, continue to the next table
|
|
355
|
+
if not matches:
|
|
356
|
+
continue
|
|
357
|
+
|
|
358
|
+
# Check the relationships for each matching cell
|
|
359
|
+
for row_idx, col_idx in matches:
|
|
360
|
+
all_relationships_satisfied = True
|
|
361
|
+
current_failed_reasons = []
|
|
362
|
+
|
|
363
|
+
# Check up relationship
|
|
364
|
+
if self.up and row_idx > 0:
|
|
365
|
+
up_cell = normalize_text(str(table_array[row_idx - 1, col_idx]))
|
|
366
|
+
up_similarity = fuzz.ratio(self.up, up_cell) / 100.0
|
|
367
|
+
up_threshold = max(0.5, 1.0 - (self.max_diffs / (len(self.up) if len(self.up) > 0 else 1)))
|
|
368
|
+
if up_similarity < up_threshold:
|
|
369
|
+
all_relationships_satisfied = False
|
|
370
|
+
current_failed_reasons.append(
|
|
371
|
+
f"Cell above '{up_cell}' doesn't match "
|
|
372
|
+
f"expected '{self.up}' "
|
|
373
|
+
f"(similarity: {up_similarity:.2f})"
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
# Check down relationship
|
|
377
|
+
if self.down and row_idx < table_array.shape[0] - 1:
|
|
378
|
+
down_cell = normalize_text(str(table_array[row_idx + 1, col_idx]))
|
|
379
|
+
down_similarity = fuzz.ratio(self.down, down_cell) / 100.0
|
|
380
|
+
down_threshold = max(0.5, 1.0 - (self.max_diffs / (len(self.down) if len(self.down) > 0 else 1)))
|
|
381
|
+
if down_similarity < down_threshold:
|
|
382
|
+
all_relationships_satisfied = False
|
|
383
|
+
current_failed_reasons.append(
|
|
384
|
+
f"Cell below '{down_cell}' doesn't match "
|
|
385
|
+
f"expected '{self.down}' "
|
|
386
|
+
f"(similarity: {down_similarity:.2f})"
|
|
387
|
+
)
|
|
388
|
+
|
|
389
|
+
# Check left relationship
|
|
390
|
+
if self.left and col_idx > 0:
|
|
391
|
+
left_cell = normalize_text(str(table_array[row_idx, col_idx - 1]))
|
|
392
|
+
left_similarity = fuzz.ratio(self.left, left_cell) / 100.0
|
|
393
|
+
left_threshold = max(0.5, 1.0 - (self.max_diffs / (len(self.left) if len(self.left) > 0 else 1)))
|
|
394
|
+
if left_similarity < left_threshold:
|
|
395
|
+
all_relationships_satisfied = False
|
|
396
|
+
current_failed_reasons.append(
|
|
397
|
+
f"Cell to the left '{left_cell}' doesn't "
|
|
398
|
+
f"match expected '{self.left}' "
|
|
399
|
+
f"(similarity: {left_similarity:.2f})"
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
# Check right relationship
|
|
403
|
+
if self.right and col_idx < table_array.shape[1] - 1:
|
|
404
|
+
right_cell = normalize_text(str(table_array[row_idx, col_idx + 1]))
|
|
405
|
+
right_similarity = fuzz.ratio(self.right, right_cell) / 100.0
|
|
406
|
+
right_threshold = max(
|
|
407
|
+
0.5,
|
|
408
|
+
1.0 - (self.max_diffs / (len(self.right) if len(self.right) > 0 else 1)),
|
|
409
|
+
)
|
|
410
|
+
if right_similarity < right_threshold:
|
|
411
|
+
all_relationships_satisfied = False
|
|
412
|
+
current_failed_reasons.append(
|
|
413
|
+
f"Cell to the right '{right_cell}' doesn't "
|
|
414
|
+
f"match expected '{self.right}' "
|
|
415
|
+
f"(similarity: {right_similarity:.2f})"
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
# Check top heading relationship
|
|
419
|
+
if self.top_heading:
|
|
420
|
+
top_heading_found = False
|
|
421
|
+
best_match = ""
|
|
422
|
+
best_similarity = 0.0
|
|
423
|
+
|
|
424
|
+
# Check the col_headers dictionary first
|
|
425
|
+
if col_idx in table_data.col_headers:
|
|
426
|
+
for _, header_text in table_data.col_headers[col_idx]:
|
|
427
|
+
header_text = normalize_text(header_text)
|
|
428
|
+
similarity = fuzz.ratio(self.top_heading, header_text) / 100.0
|
|
429
|
+
if similarity > best_similarity:
|
|
430
|
+
best_similarity = similarity
|
|
431
|
+
best_match = header_text
|
|
432
|
+
top_threshold = max(
|
|
433
|
+
0.5,
|
|
434
|
+
1.0
|
|
435
|
+
- (self.max_diffs / (len(self.top_heading) if len(self.top_heading) > 0 else 1)),
|
|
436
|
+
)
|
|
437
|
+
if best_similarity >= top_threshold:
|
|
438
|
+
top_heading_found = True
|
|
439
|
+
break
|
|
440
|
+
|
|
441
|
+
# If no match found in col_headers, fall back to checking header rows
|
|
442
|
+
if not top_heading_found and header_rows:
|
|
443
|
+
for i in sorted(header_rows):
|
|
444
|
+
if i < row_idx and str(table_array[i, col_idx]).strip():
|
|
445
|
+
header_text = normalize_text(str(table_array[i, col_idx]))
|
|
446
|
+
similarity = fuzz.ratio(self.top_heading, header_text) / 100.0
|
|
447
|
+
if similarity > best_similarity:
|
|
448
|
+
best_similarity = similarity
|
|
449
|
+
best_match = header_text
|
|
450
|
+
top_threshold = max(
|
|
451
|
+
0.5,
|
|
452
|
+
1.0
|
|
453
|
+
- (
|
|
454
|
+
self.max_diffs / (len(self.top_heading) if len(self.top_heading) > 0 else 1)
|
|
455
|
+
),
|
|
456
|
+
)
|
|
457
|
+
if best_similarity >= top_threshold:
|
|
458
|
+
top_heading_found = True
|
|
459
|
+
break
|
|
460
|
+
|
|
461
|
+
# If still no match, use any non-empty cell above as a last resort
|
|
462
|
+
if not top_heading_found and not best_match and row_idx > 0:
|
|
463
|
+
for i in range(row_idx):
|
|
464
|
+
if str(table_array[i, col_idx]).strip():
|
|
465
|
+
header_text = normalize_text(str(table_array[i, col_idx]))
|
|
466
|
+
similarity = fuzz.ratio(self.top_heading, header_text) / 100.0
|
|
467
|
+
if similarity > best_similarity:
|
|
468
|
+
best_similarity = similarity
|
|
469
|
+
best_match = header_text
|
|
470
|
+
|
|
471
|
+
if not best_match:
|
|
472
|
+
all_relationships_satisfied = False
|
|
473
|
+
current_failed_reasons.append(f"No top heading found for cell at ({row_idx}, {col_idx})")
|
|
474
|
+
else:
|
|
475
|
+
top_threshold = max(
|
|
476
|
+
0.5,
|
|
477
|
+
1.0 - (self.max_diffs / (len(self.top_heading) if len(self.top_heading) > 0 else 1)),
|
|
478
|
+
)
|
|
479
|
+
if best_similarity < top_threshold:
|
|
480
|
+
all_relationships_satisfied = False
|
|
481
|
+
current_failed_reasons.append(
|
|
482
|
+
f"Top heading '{best_match}' doesn't "
|
|
483
|
+
f"match expected '{self.top_heading}' "
|
|
484
|
+
f"(similarity: {best_similarity:.2f})"
|
|
485
|
+
)
|
|
486
|
+
|
|
487
|
+
# Check left heading relationship
|
|
488
|
+
if self.left_heading:
|
|
489
|
+
left_heading_found = False
|
|
490
|
+
best_match = ""
|
|
491
|
+
best_similarity = 0.0
|
|
492
|
+
|
|
493
|
+
# Check the row_headers dictionary first
|
|
494
|
+
if row_idx in table_data.row_headers:
|
|
495
|
+
for _, header_text in table_data.row_headers[row_idx]:
|
|
496
|
+
header_text = normalize_text(header_text)
|
|
497
|
+
similarity = fuzz.ratio(self.left_heading, header_text) / 100.0
|
|
498
|
+
if similarity > best_similarity:
|
|
499
|
+
best_similarity = similarity
|
|
500
|
+
best_match = header_text
|
|
501
|
+
left_threshold = max(
|
|
502
|
+
0.5,
|
|
503
|
+
1.0
|
|
504
|
+
- (self.max_diffs / (len(self.left_heading) if len(self.left_heading) > 0 else 1)),
|
|
505
|
+
)
|
|
506
|
+
if best_similarity >= left_threshold:
|
|
507
|
+
left_heading_found = True
|
|
508
|
+
break
|
|
509
|
+
|
|
510
|
+
# If no match found in row_headers, fall back to checking header columns
|
|
511
|
+
if not left_heading_found and header_cols:
|
|
512
|
+
for j in sorted(header_cols):
|
|
513
|
+
if j < col_idx and str(table_array[row_idx, j]).strip():
|
|
514
|
+
header_text = normalize_text(str(table_array[row_idx, j]))
|
|
515
|
+
similarity = fuzz.ratio(self.left_heading, header_text) / 100.0
|
|
516
|
+
if similarity > best_similarity:
|
|
517
|
+
best_similarity = similarity
|
|
518
|
+
best_match = header_text
|
|
519
|
+
left_threshold = max(
|
|
520
|
+
0.5,
|
|
521
|
+
1.0
|
|
522
|
+
- (
|
|
523
|
+
self.max_diffs
|
|
524
|
+
/ (len(self.left_heading) if len(self.left_heading) > 0 else 1)
|
|
525
|
+
),
|
|
526
|
+
)
|
|
527
|
+
if best_similarity >= left_threshold:
|
|
528
|
+
left_heading_found = True
|
|
529
|
+
break
|
|
530
|
+
|
|
531
|
+
# If still no match, use any non-empty cell to the left as a last resort
|
|
532
|
+
if not left_heading_found and not best_match and col_idx > 0:
|
|
533
|
+
for j in range(col_idx):
|
|
534
|
+
if str(table_array[row_idx, j]).strip():
|
|
535
|
+
header_text = normalize_text(str(table_array[row_idx, j]))
|
|
536
|
+
similarity = fuzz.ratio(self.left_heading, header_text) / 100.0
|
|
537
|
+
if similarity > best_similarity:
|
|
538
|
+
best_similarity = similarity
|
|
539
|
+
best_match = header_text
|
|
540
|
+
|
|
541
|
+
if not best_match:
|
|
542
|
+
all_relationships_satisfied = False
|
|
543
|
+
current_failed_reasons.append(f"No left heading found for cell at ({row_idx}, {col_idx})")
|
|
544
|
+
else:
|
|
545
|
+
left_threshold = max(
|
|
546
|
+
0.5,
|
|
547
|
+
1.0 - (self.max_diffs / (len(self.left_heading) if len(self.left_heading) > 0 else 1)),
|
|
548
|
+
)
|
|
549
|
+
if best_similarity < left_threshold:
|
|
550
|
+
all_relationships_satisfied = False
|
|
551
|
+
current_failed_reasons.append(
|
|
552
|
+
f"Left heading '{best_match}' doesn't "
|
|
553
|
+
f"match expected '{self.left_heading}' "
|
|
554
|
+
f"(similarity: {best_similarity:.2f})"
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
# If all relationships are satisfied for this cell, the test passes
|
|
558
|
+
if all_relationships_satisfied:
|
|
559
|
+
return True, ""
|
|
560
|
+
else:
|
|
561
|
+
failed_reasons.extend(current_failed_reasons)
|
|
562
|
+
|
|
563
|
+
if not failed_reasons:
|
|
564
|
+
return (
|
|
565
|
+
False,
|
|
566
|
+
f"No cell matching '{self.cell}' found in any table with threshold {threshold}",
|
|
567
|
+
)
|
|
568
|
+
else:
|
|
569
|
+
return (
|
|
570
|
+
False,
|
|
571
|
+
f"Found cells matching '{self.cell}' but relationships were not satisfied: {'; '.join(failed_reasons)}",
|
|
572
|
+
)
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
class TablesValuesRule(ParseTestRule):
|
|
576
|
+
"""Test rule to verify that tables match ground truth tables."""
|
|
577
|
+
|
|
578
|
+
def __init__(self, rule_data: ParseTablesValuesRule | dict):
|
|
579
|
+
super().__init__(rule_data)
|
|
580
|
+
rule_data = cast(ParseTablesValuesRule, self._rule_data)
|
|
581
|
+
|
|
582
|
+
if self.type != TestType.TABLES_VALUES.value:
|
|
583
|
+
raise ValueError(f"Invalid type for TablesValuesRule: {self.type}")
|
|
584
|
+
|
|
585
|
+
self.table_variations = rule_data.table_variations
|
|
586
|
+
self.json_path = rule_data.json_path
|
|
587
|
+
self.table_match_threshold = rule_data.table_match_threshold
|
|
588
|
+
self.table_values_match_threshold = rule_data.table_values_match_threshold
|
|
589
|
+
self.add_check_num_rows_test = rule_data.add_check_num_rows_test
|
|
590
|
+
self.add_check_num_cols_test = rule_data.add_check_num_cols_test
|
|
591
|
+
|
|
592
|
+
# Must have either table_variations or json_path
|
|
593
|
+
if not self.table_variations and not self.json_path:
|
|
594
|
+
raise ValueError("Either table_variations or json_path must be provided")
|
|
595
|
+
|
|
596
|
+
self.relevant_gt: pd.DataFrame | None = None
|
|
597
|
+
self.relevant_pred: pd.DataFrame | None = None
|
|
598
|
+
|
|
599
|
+
def _load_json_table(self, json_file_path: str) -> dict[str, Any]:
|
|
600
|
+
"""Load the ground truth table JSON file."""
|
|
601
|
+
with open(json_file_path, encoding="utf-8") as f:
|
|
602
|
+
data = json.load(f)
|
|
603
|
+
|
|
604
|
+
# Validate the structure
|
|
605
|
+
required_keys = ["pdf", "page", "table_variations", "id", "table_match_threshold"]
|
|
606
|
+
for key in required_keys:
|
|
607
|
+
if key not in data:
|
|
608
|
+
raise ValueError(f"JSON file missing required key: {key}")
|
|
609
|
+
|
|
610
|
+
return data # type: ignore[no-any-return]
|
|
611
|
+
|
|
612
|
+
def _tabledata_to_dataframe(self, table_data: TableData) -> pd.DataFrame:
|
|
613
|
+
"""Convert a TableData object to a pandas DataFrame."""
|
|
614
|
+
return pd.DataFrame(table_data.data)
|
|
615
|
+
|
|
616
|
+
def _table_schema_to_dataframe(self, table_schema: dict[str, Any]) -> pd.DataFrame:
|
|
617
|
+
"""Convert a table schema (with rowspan/colspan) to a pandas DataFrame."""
|
|
618
|
+
if "rows" not in table_schema:
|
|
619
|
+
raise ValueError("table_schema missing 'rows' key")
|
|
620
|
+
|
|
621
|
+
rows_data = table_schema["rows"]
|
|
622
|
+
|
|
623
|
+
# First pass: determine the grid size and build a cell position map
|
|
624
|
+
grid = {} # (row_idx, col_idx) -> cell_text
|
|
625
|
+
max_cols = 0
|
|
626
|
+
|
|
627
|
+
for row_idx, row_dict in enumerate(rows_data):
|
|
628
|
+
if "cells" not in row_dict:
|
|
629
|
+
raise ValueError(f"Row {row_idx} missing 'cells' key")
|
|
630
|
+
|
|
631
|
+
cells = row_dict["cells"]
|
|
632
|
+
col_idx = 0
|
|
633
|
+
|
|
634
|
+
for cell_dict in cells:
|
|
635
|
+
# Skip columns that are already filled by previous rowspan/colspan
|
|
636
|
+
while (row_idx, col_idx) in grid:
|
|
637
|
+
col_idx += 1
|
|
638
|
+
|
|
639
|
+
# Extract cell properties
|
|
640
|
+
text = cell_dict.get("text", "")
|
|
641
|
+
colspan = cell_dict.get("colspan", 1)
|
|
642
|
+
rowspan = cell_dict.get("rowspan", 1)
|
|
643
|
+
|
|
644
|
+
# Fill the grid for this cell and all its spans
|
|
645
|
+
for r_offset in range(rowspan):
|
|
646
|
+
for c_offset in range(colspan):
|
|
647
|
+
grid[(row_idx + r_offset, col_idx + c_offset)] = text
|
|
648
|
+
|
|
649
|
+
col_idx += colspan
|
|
650
|
+
max_cols = max(max_cols, col_idx)
|
|
651
|
+
|
|
652
|
+
# Second pass: build the DataFrame from the grid
|
|
653
|
+
num_rows = max(r for r, c in grid.keys()) + 1 if grid else 0
|
|
654
|
+
num_cols = max_cols
|
|
655
|
+
|
|
656
|
+
# Create a 2D list for the DataFrame
|
|
657
|
+
data_array = []
|
|
658
|
+
for r in range(num_rows):
|
|
659
|
+
row = []
|
|
660
|
+
for c in range(num_cols):
|
|
661
|
+
cell_value = grid.get((r, c), "")
|
|
662
|
+
row.append(cell_value)
|
|
663
|
+
data_array.append(row)
|
|
664
|
+
|
|
665
|
+
# Convert to DataFrame
|
|
666
|
+
df = pd.DataFrame(data_array)
|
|
667
|
+
|
|
668
|
+
return df
|
|
669
|
+
|
|
670
|
+
def _normalize_cell(self, cell: str) -> str:
|
|
671
|
+
"""Normalize a cell value for comparison."""
|
|
672
|
+
text = unidecode(str(cell)).lower()
|
|
673
|
+
# Remove all whitespace
|
|
674
|
+
text = re.sub(r"\s+", "", text)
|
|
675
|
+
# Remove zero-width characters
|
|
676
|
+
text = re.sub(r"[\u200B-\u200D\uFEFF\u00AD]", "", text)
|
|
677
|
+
|
|
678
|
+
# Handle numbers with commas
|
|
679
|
+
number_pattern = r"(-?\d{1,3}(?:,\d{3})*\.?\d*)"
|
|
680
|
+
match = re.search(number_pattern, text)
|
|
681
|
+
|
|
682
|
+
if match:
|
|
683
|
+
number_str = match.group(1)
|
|
684
|
+
try:
|
|
685
|
+
clean_number = number_str.replace(",", "")
|
|
686
|
+
float_val = float(clean_number)
|
|
687
|
+
|
|
688
|
+
if "," in number_str:
|
|
689
|
+
if float_val >= 1000:
|
|
690
|
+
normalized_number = f"{float_val:,.10g}".rstrip("0").rstrip(".")
|
|
691
|
+
else:
|
|
692
|
+
normalized_number = f"{float_val:g}"
|
|
693
|
+
else:
|
|
694
|
+
normalized_number = f"{float_val:g}"
|
|
695
|
+
|
|
696
|
+
return text.replace(number_str, normalized_number)
|
|
697
|
+
except ValueError:
|
|
698
|
+
pass
|
|
699
|
+
|
|
700
|
+
return text
|
|
701
|
+
|
|
702
|
+
def _compute_single_table_similarity(self, gt_df: pd.DataFrame, pred_df: pd.DataFrame) -> float:
|
|
703
|
+
# Extract and normalize all words from ground truth table
|
|
704
|
+
gt_words = []
|
|
705
|
+
for row_idx in range(gt_df.shape[0]):
|
|
706
|
+
for col_idx in range(gt_df.shape[1]):
|
|
707
|
+
cell_value = str(gt_df.iloc[row_idx, col_idx])
|
|
708
|
+
normalized = self._normalize_cell(cell_value)
|
|
709
|
+
if normalized:
|
|
710
|
+
gt_words.append(normalized)
|
|
711
|
+
|
|
712
|
+
gt_counter = Counter(gt_words)
|
|
713
|
+
|
|
714
|
+
# If ground truth is empty, return 0
|
|
715
|
+
if not gt_words:
|
|
716
|
+
return 0.0
|
|
717
|
+
|
|
718
|
+
# Extract and normalize all words from predicted table
|
|
719
|
+
pred_words = []
|
|
720
|
+
for row_idx in range(pred_df.shape[0]):
|
|
721
|
+
for col_idx in range(pred_df.shape[1]):
|
|
722
|
+
cell_value = str(pred_df.iloc[row_idx, col_idx])
|
|
723
|
+
normalized = self._normalize_cell(cell_value)
|
|
724
|
+
if normalized:
|
|
725
|
+
pred_words.append(normalized)
|
|
726
|
+
|
|
727
|
+
pred_counter = Counter(pred_words)
|
|
728
|
+
|
|
729
|
+
# If predicted table is empty, return 0
|
|
730
|
+
if not pred_words:
|
|
731
|
+
return 0.0
|
|
732
|
+
|
|
733
|
+
# Compute intersection (minimum counts for each word)
|
|
734
|
+
intersection = sum((gt_counter & pred_counter).values())
|
|
735
|
+
|
|
736
|
+
# Compute union (maximum counts for each word)
|
|
737
|
+
union = sum((gt_counter | pred_counter).values())
|
|
738
|
+
|
|
739
|
+
# Compute Jaccard similarity with counts
|
|
740
|
+
if union > 0:
|
|
741
|
+
return intersection / union
|
|
742
|
+
|
|
743
|
+
return 0.0
|
|
744
|
+
|
|
745
|
+
def _compare_cells_exactly(self, gt_df: pd.DataFrame, pred_df: pd.DataFrame) -> tuple[int, int, float]:
|
|
746
|
+
"""Compare cells one-by-one between ground truth and predicted DataFrames."""
|
|
747
|
+
# Compare only the overlapping region
|
|
748
|
+
min_rows = min(len(gt_df), len(pred_df))
|
|
749
|
+
min_cols = min(len(gt_df.columns), len(pred_df.columns))
|
|
750
|
+
|
|
751
|
+
matching_cells = 0
|
|
752
|
+
total_cells = min_rows * min_cols
|
|
753
|
+
|
|
754
|
+
if total_cells == 0:
|
|
755
|
+
return 0, 0, 0.0
|
|
756
|
+
|
|
757
|
+
for row_idx in range(min_rows):
|
|
758
|
+
for col_idx in range(min_cols):
|
|
759
|
+
gt_cell = str(gt_df.iloc[row_idx, col_idx])
|
|
760
|
+
pred_cell = str(pred_df.iloc[row_idx, col_idx])
|
|
761
|
+
|
|
762
|
+
# Normalize both cells
|
|
763
|
+
gt_normalized = self._normalize_cell(gt_cell)
|
|
764
|
+
pred_normalized = self._normalize_cell(pred_cell)
|
|
765
|
+
|
|
766
|
+
# Exact match after normalization
|
|
767
|
+
if gt_normalized == pred_normalized:
|
|
768
|
+
matching_cells += 1
|
|
769
|
+
|
|
770
|
+
match_ratio = matching_cells / total_cells if total_cells > 0 else 0.0
|
|
771
|
+
return matching_cells, total_cells, match_ratio
|
|
772
|
+
|
|
773
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
774
|
+
"""Run the table values test on provided content."""
|
|
775
|
+
# Extract tables from content
|
|
776
|
+
pred_tables = []
|
|
777
|
+
|
|
778
|
+
# Parse markdown tables
|
|
779
|
+
md_tables = parse_markdown_tables(content)
|
|
780
|
+
pred_tables.extend(md_tables)
|
|
781
|
+
|
|
782
|
+
# Parse HTML tables
|
|
783
|
+
html_tables = parse_html_tables(content)
|
|
784
|
+
pred_tables.extend(html_tables)
|
|
785
|
+
|
|
786
|
+
if not pred_tables:
|
|
787
|
+
return False, "No tables found in the content"
|
|
788
|
+
|
|
789
|
+
pred_tables = [self._tabledata_to_dataframe(table) for table in pred_tables]
|
|
790
|
+
|
|
791
|
+
# Load table variations either from embedded data or external JSON
|
|
792
|
+
try:
|
|
793
|
+
if self.table_variations:
|
|
794
|
+
table_variations = self.table_variations
|
|
795
|
+
elif self.json_path:
|
|
796
|
+
gt_data = self._load_json_table(self.json_path)
|
|
797
|
+
table_variations = gt_data["table_variations"]
|
|
798
|
+
else:
|
|
799
|
+
return False, "No table variations available (neither embedded nor in json_path)"
|
|
800
|
+
|
|
801
|
+
if not table_variations:
|
|
802
|
+
return False, "No table variations found"
|
|
803
|
+
|
|
804
|
+
except Exception as e:
|
|
805
|
+
return False, f"Error loading ground truth data: {e}"
|
|
806
|
+
|
|
807
|
+
# Track the best variation and its score
|
|
808
|
+
best_variation_idx = -1
|
|
809
|
+
best_score = 0.0
|
|
810
|
+
best_pred_idx = -1
|
|
811
|
+
|
|
812
|
+
for var_idx, table_schema in enumerate(table_variations):
|
|
813
|
+
try:
|
|
814
|
+
# Convert the table schema to a DataFrame
|
|
815
|
+
gt_df = self._table_schema_to_dataframe(table_schema)
|
|
816
|
+
|
|
817
|
+
# Compare this GT variation with each predicted table
|
|
818
|
+
for pred_idx, pred_table in enumerate(pred_tables):
|
|
819
|
+
# Compute similarity between this specific GT and this specific pred table
|
|
820
|
+
similarity = self._compute_single_table_similarity(gt_df, pred_table)
|
|
821
|
+
|
|
822
|
+
# Track the overall best score across all GT-pred pairs
|
|
823
|
+
if similarity > best_score:
|
|
824
|
+
best_score = similarity
|
|
825
|
+
best_variation_idx = var_idx
|
|
826
|
+
best_pred_idx = pred_idx
|
|
827
|
+
self.relevant_pred = pred_table
|
|
828
|
+
self.relevant_gt = gt_df
|
|
829
|
+
|
|
830
|
+
except Exception:
|
|
831
|
+
# If conversion fails, continue to next variation
|
|
832
|
+
continue
|
|
833
|
+
|
|
834
|
+
# Check if the best variation passes the threshold
|
|
835
|
+
threshold = self.table_match_threshold
|
|
836
|
+
|
|
837
|
+
if best_score < threshold:
|
|
838
|
+
return (
|
|
839
|
+
False,
|
|
840
|
+
f"Best match: GT variation {best_variation_idx} with pred table {best_pred_idx} "
|
|
841
|
+
f"scored {best_score:.3f}, below threshold {threshold:.3f}",
|
|
842
|
+
)
|
|
843
|
+
|
|
844
|
+
# Perform exact cell-by-cell comparison on the best match
|
|
845
|
+
if self.relevant_gt is None or self.relevant_pred is None:
|
|
846
|
+
return False, "No relevant GT or pred table found for cell comparison"
|
|
847
|
+
|
|
848
|
+
matching_cells, total_cells, match_ratio = self._compare_cells_exactly(self.relevant_gt, self.relevant_pred)
|
|
849
|
+
cell_threshold = self.table_values_match_threshold
|
|
850
|
+
|
|
851
|
+
if match_ratio < cell_threshold:
|
|
852
|
+
return ( # type: ignore[return-value]
|
|
853
|
+
False,
|
|
854
|
+
f"Best match: GT variation {best_variation_idx} with pred table {best_pred_idx} "
|
|
855
|
+
f"scored {best_score:.3f} (>= {threshold:.3f}), "
|
|
856
|
+
f"but cell exact match "
|
|
857
|
+
f"{matching_cells}/{total_cells} "
|
|
858
|
+
f"({match_ratio:.3f}) below threshold "
|
|
859
|
+
f"{cell_threshold:.3f}",
|
|
860
|
+
f"({match_ratio:.3f}) below threshold {cell_threshold:.3f}",
|
|
861
|
+
)
|
|
862
|
+
|
|
863
|
+
return (
|
|
864
|
+
True,
|
|
865
|
+
f"Best match: GT variation {best_variation_idx} with pred table {best_pred_idx} "
|
|
866
|
+
f"scored {best_score:.3f} (>= {threshold:.3f}), "
|
|
867
|
+
f"cell exact match {matching_cells}/{total_cells} "
|
|
868
|
+
f"({match_ratio:.3f})",
|
|
869
|
+
)
|
|
870
|
+
|
|
871
|
+
|
|
872
|
+
class TablesNumRowsRule(ParseTestRule):
|
|
873
|
+
"""Test rule to verify that predicted table has the correct number of rows."""
|
|
874
|
+
|
|
875
|
+
def __init__(self, rule_data: ParseTablesNumRowsRule | dict):
|
|
876
|
+
super().__init__(rule_data)
|
|
877
|
+
rule_data = cast(ParseTablesNumRowsRule, self._rule_data)
|
|
878
|
+
|
|
879
|
+
if self.type != TestType.TABLES_NUM_ROWS.value:
|
|
880
|
+
raise ValueError(f"Invalid type for TablesNumRowsRule: {self.type}")
|
|
881
|
+
|
|
882
|
+
self.expected_num_rows = rule_data.expected_num_rows
|
|
883
|
+
self.actual_num_rows = rule_data.actual_num_rows
|
|
884
|
+
|
|
885
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
886
|
+
"""Check if row count matches."""
|
|
887
|
+
if self.actual_num_rows is None:
|
|
888
|
+
return False, "Row count not populated"
|
|
889
|
+
|
|
890
|
+
if self.actual_num_rows == self.expected_num_rows:
|
|
891
|
+
return True, f"Row count matches: {self.actual_num_rows}"
|
|
892
|
+
else:
|
|
893
|
+
return (
|
|
894
|
+
False,
|
|
895
|
+
f"Row count mismatch: expected {self.expected_num_rows}, got {self.actual_num_rows}",
|
|
896
|
+
)
|
|
897
|
+
|
|
898
|
+
|
|
899
|
+
class TablesNumColsRule(ParseTestRule):
|
|
900
|
+
"""Test rule to verify that predicted table has the correct number of columns."""
|
|
901
|
+
|
|
902
|
+
def __init__(self, rule_data: ParseTablesNumColsRule | dict):
|
|
903
|
+
super().__init__(rule_data)
|
|
904
|
+
rule_data = cast(ParseTablesNumColsRule, self._rule_data)
|
|
905
|
+
|
|
906
|
+
if self.type != TestType.TABLES_NUM_COLS.value:
|
|
907
|
+
raise ValueError(f"Invalid type for TablesNumColsRule: {self.type}")
|
|
908
|
+
|
|
909
|
+
self.expected_num_cols = rule_data.expected_num_cols
|
|
910
|
+
self.actual_num_cols = rule_data.actual_num_cols
|
|
911
|
+
|
|
912
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
913
|
+
"""Check if column count matches."""
|
|
914
|
+
if self.actual_num_cols is None:
|
|
915
|
+
return False, "Column count not populated"
|
|
916
|
+
|
|
917
|
+
if self.actual_num_cols == self.expected_num_cols:
|
|
918
|
+
return True, f"Column count matches: {self.actual_num_cols}"
|
|
919
|
+
else:
|
|
920
|
+
return (
|
|
921
|
+
False,
|
|
922
|
+
f"Column count mismatch: expected {self.expected_num_cols}, got {self.actual_num_cols}",
|
|
923
|
+
)
|
|
924
|
+
|
|
925
|
+
|
|
926
|
+
# =============================================================================
|
|
927
|
+
# Table Hierarchy Rules
|
|
928
|
+
# =============================================================================
|
|
929
|
+
|
|
930
|
+
|
|
931
|
+
class TableColspanRule(ParseTestRule):
|
|
932
|
+
"""Test rule to verify a cell has the expected colspan attribute."""
|
|
933
|
+
|
|
934
|
+
def __init__(self, rule_data: ParseTableColspanRule | dict):
|
|
935
|
+
super().__init__(rule_data)
|
|
936
|
+
rule_data = cast(ParseTableColspanRule, self._rule_data)
|
|
937
|
+
|
|
938
|
+
if self.type != TestType.TABLE_COLSPAN.value:
|
|
939
|
+
raise ValueError(f"Invalid type for TableColspanRule: {self.type}")
|
|
940
|
+
|
|
941
|
+
self.cell = normalize_text(rule_data.cell)
|
|
942
|
+
self.expected_colspan = rule_data.expected_colspan
|
|
943
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
944
|
+
|
|
945
|
+
if not self.cell:
|
|
946
|
+
raise ValueError("cell must be provided")
|
|
947
|
+
if self.expected_colspan < 1:
|
|
948
|
+
raise ValueError("expected_colspan must be >= 1")
|
|
949
|
+
|
|
950
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
951
|
+
"""Check if cell has expected colspan attribute."""
|
|
952
|
+
grids = find_all_html_tables(content)
|
|
953
|
+
if not grids:
|
|
954
|
+
return False, "No HTML tables found in content"
|
|
955
|
+
|
|
956
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
957
|
+
if self.table_anchor_cells:
|
|
958
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
959
|
+
if anchor_result.grid is not None:
|
|
960
|
+
grids = [anchor_result.grid]
|
|
961
|
+
elif anchor_result.is_ambiguous:
|
|
962
|
+
return False, (
|
|
963
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
964
|
+
f"tables - could not uniquely identify target table"
|
|
965
|
+
)
|
|
966
|
+
else:
|
|
967
|
+
return False, "Table anchor cells not found in any table"
|
|
968
|
+
|
|
969
|
+
# Step 2: Find cell and check colspan
|
|
970
|
+
match = find_cell_in_grids(grids, self.cell)
|
|
971
|
+
if not match:
|
|
972
|
+
return False, f"Cell '{self.cell}' not found in target table"
|
|
973
|
+
|
|
974
|
+
grid, cell, row_idx, col_idx = match
|
|
975
|
+
|
|
976
|
+
if cell.colspan == self.expected_colspan:
|
|
977
|
+
return True, f"Cell '{self.cell}' has correct colspan={cell.colspan}"
|
|
978
|
+
else:
|
|
979
|
+
return (
|
|
980
|
+
False,
|
|
981
|
+
f"Cell '{self.cell}' has colspan={cell.colspan}, expected {self.expected_colspan}",
|
|
982
|
+
)
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
class TableRowspanRule(ParseTestRule):
|
|
986
|
+
"""Test rule to verify a cell has the expected rowspan attribute."""
|
|
987
|
+
|
|
988
|
+
def __init__(self, rule_data: ParseTableRowspanRule | dict):
|
|
989
|
+
super().__init__(rule_data)
|
|
990
|
+
rule_data = cast(ParseTableRowspanRule, self._rule_data)
|
|
991
|
+
|
|
992
|
+
if self.type != TestType.TABLE_ROWSPAN.value:
|
|
993
|
+
raise ValueError(f"Invalid type for TableRowspanRule: {self.type}")
|
|
994
|
+
|
|
995
|
+
self.cell = normalize_text(rule_data.cell)
|
|
996
|
+
self.expected_rowspan = rule_data.expected_rowspan
|
|
997
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
998
|
+
|
|
999
|
+
if not self.cell:
|
|
1000
|
+
raise ValueError("cell must be provided")
|
|
1001
|
+
if self.expected_rowspan < 1:
|
|
1002
|
+
raise ValueError("expected_rowspan must be >= 1")
|
|
1003
|
+
|
|
1004
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1005
|
+
"""Check if cell has expected rowspan attribute."""
|
|
1006
|
+
grids = find_all_html_tables(content)
|
|
1007
|
+
if not grids:
|
|
1008
|
+
return False, "No HTML tables found in content"
|
|
1009
|
+
|
|
1010
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
1011
|
+
if self.table_anchor_cells:
|
|
1012
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1013
|
+
if anchor_result.grid is not None:
|
|
1014
|
+
grids = [anchor_result.grid]
|
|
1015
|
+
elif anchor_result.is_ambiguous:
|
|
1016
|
+
return False, (
|
|
1017
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1018
|
+
f"tables - could not uniquely identify target table"
|
|
1019
|
+
)
|
|
1020
|
+
else:
|
|
1021
|
+
return False, "Table anchor cells not found in any table"
|
|
1022
|
+
|
|
1023
|
+
# Step 2: Find cell and check rowspan
|
|
1024
|
+
match = find_cell_in_grids(grids, self.cell)
|
|
1025
|
+
if not match:
|
|
1026
|
+
return False, f"Cell '{self.cell}' not found in target table"
|
|
1027
|
+
|
|
1028
|
+
grid, cell, row_idx, col_idx = match
|
|
1029
|
+
|
|
1030
|
+
if cell.rowspan == self.expected_rowspan:
|
|
1031
|
+
return True, f"Cell '{self.cell}' has correct rowspan={cell.rowspan}"
|
|
1032
|
+
else:
|
|
1033
|
+
return (
|
|
1034
|
+
False,
|
|
1035
|
+
f"Cell '{self.cell}' has rowspan={cell.rowspan}, expected {self.expected_rowspan}",
|
|
1036
|
+
)
|
|
1037
|
+
|
|
1038
|
+
|
|
1039
|
+
class TableSameRowRule(ParseTestRule):
|
|
1040
|
+
"""Test rule to verify two cells share a logical row (considering rowspan)."""
|
|
1041
|
+
|
|
1042
|
+
def __init__(self, rule_data: ParseTableSameRowRule | dict):
|
|
1043
|
+
super().__init__(rule_data)
|
|
1044
|
+
rule_data = cast(ParseTableSameRowRule, self._rule_data)
|
|
1045
|
+
|
|
1046
|
+
if self.type != TestType.TABLE_SAME_ROW.value:
|
|
1047
|
+
raise ValueError(f"Invalid type for TableSameRowRule: {self.type}")
|
|
1048
|
+
|
|
1049
|
+
self.cell_a = normalize_text(rule_data.cell_a)
|
|
1050
|
+
self.cell_b = normalize_text(rule_data.cell_b)
|
|
1051
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
1052
|
+
|
|
1053
|
+
if not self.cell_a or not self.cell_b:
|
|
1054
|
+
raise ValueError("Both cell_a and cell_b must be provided")
|
|
1055
|
+
|
|
1056
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1057
|
+
"""Check if two cells share a logical row."""
|
|
1058
|
+
grids = find_all_html_tables(content)
|
|
1059
|
+
if not grids:
|
|
1060
|
+
return False, "No HTML tables found in content"
|
|
1061
|
+
|
|
1062
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
1063
|
+
if self.table_anchor_cells:
|
|
1064
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1065
|
+
if anchor_result.grid is not None:
|
|
1066
|
+
grids = [anchor_result.grid]
|
|
1067
|
+
elif anchor_result.is_ambiguous:
|
|
1068
|
+
return False, (
|
|
1069
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1070
|
+
f"tables - could not uniquely identify target table"
|
|
1071
|
+
)
|
|
1072
|
+
else:
|
|
1073
|
+
return False, "Table anchor cells not found in any table"
|
|
1074
|
+
|
|
1075
|
+
match_a = find_cell_in_grids(grids, self.cell_a)
|
|
1076
|
+
if not match_a:
|
|
1077
|
+
return False, f"Cell '{self.cell_a}' not found in target table"
|
|
1078
|
+
|
|
1079
|
+
match_b = find_cell_in_grids(grids, self.cell_b)
|
|
1080
|
+
if not match_b:
|
|
1081
|
+
return False, f"Cell '{self.cell_b}' not found in target table"
|
|
1082
|
+
|
|
1083
|
+
grid_a, cell_a, row_a, col_a = match_a
|
|
1084
|
+
grid_b, cell_b, row_b, col_b = match_b
|
|
1085
|
+
|
|
1086
|
+
# Must be in the same table
|
|
1087
|
+
if grid_a is not grid_b:
|
|
1088
|
+
return False, "Cells are in different tables"
|
|
1089
|
+
|
|
1090
|
+
# Calculate row ranges for each cell (considering rowspan)
|
|
1091
|
+
rows_a = set(range(cell_a.original_row, cell_a.original_row + cell_a.rowspan))
|
|
1092
|
+
rows_b = set(range(cell_b.original_row, cell_b.original_row + cell_b.rowspan))
|
|
1093
|
+
|
|
1094
|
+
if rows_a & rows_b: # Intersection
|
|
1095
|
+
return True, f"Cells share rows: {rows_a & rows_b}"
|
|
1096
|
+
else:
|
|
1097
|
+
return False, f"Cells do not share any row. A: rows {rows_a}, B: rows {rows_b}"
|
|
1098
|
+
|
|
1099
|
+
|
|
1100
|
+
class TableSameColumnRule(ParseTestRule):
|
|
1101
|
+
"""Test rule to verify two cells share a logical column (considering colspan)."""
|
|
1102
|
+
|
|
1103
|
+
def __init__(self, rule_data: ParseTableSameColumnRule | dict):
|
|
1104
|
+
super().__init__(rule_data)
|
|
1105
|
+
rule_data = cast(ParseTableSameColumnRule, self._rule_data)
|
|
1106
|
+
|
|
1107
|
+
if self.type != TestType.TABLE_SAME_COLUMN.value:
|
|
1108
|
+
raise ValueError(f"Invalid type for TableSameColumnRule: {self.type}")
|
|
1109
|
+
|
|
1110
|
+
self.cell_a = normalize_text(rule_data.cell_a)
|
|
1111
|
+
self.cell_b = normalize_text(rule_data.cell_b)
|
|
1112
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
1113
|
+
|
|
1114
|
+
if not self.cell_a or not self.cell_b:
|
|
1115
|
+
raise ValueError("Both cell_a and cell_b must be provided")
|
|
1116
|
+
|
|
1117
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1118
|
+
"""Check if two cells share a logical column."""
|
|
1119
|
+
grids = find_all_html_tables(content)
|
|
1120
|
+
if not grids:
|
|
1121
|
+
return False, "No HTML tables found in content"
|
|
1122
|
+
|
|
1123
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
1124
|
+
if self.table_anchor_cells:
|
|
1125
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1126
|
+
if anchor_result.grid is not None:
|
|
1127
|
+
grids = [anchor_result.grid]
|
|
1128
|
+
elif anchor_result.is_ambiguous:
|
|
1129
|
+
return False, (
|
|
1130
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1131
|
+
f"tables - could not uniquely identify target table"
|
|
1132
|
+
)
|
|
1133
|
+
else:
|
|
1134
|
+
return False, "Table anchor cells not found in any table"
|
|
1135
|
+
|
|
1136
|
+
match_a = find_cell_in_grids(grids, self.cell_a)
|
|
1137
|
+
if not match_a:
|
|
1138
|
+
return False, f"Cell '{self.cell_a}' not found in target table"
|
|
1139
|
+
|
|
1140
|
+
match_b = find_cell_in_grids(grids, self.cell_b)
|
|
1141
|
+
if not match_b:
|
|
1142
|
+
return False, f"Cell '{self.cell_b}' not found in target table"
|
|
1143
|
+
|
|
1144
|
+
grid_a, cell_a, row_a, col_a = match_a
|
|
1145
|
+
grid_b, cell_b, row_b, col_b = match_b
|
|
1146
|
+
|
|
1147
|
+
# Must be in the same table
|
|
1148
|
+
if grid_a is not grid_b:
|
|
1149
|
+
return False, "Cells are in different tables"
|
|
1150
|
+
|
|
1151
|
+
# Calculate column ranges for each cell (considering colspan)
|
|
1152
|
+
cols_a = set(range(cell_a.original_col, cell_a.original_col + cell_a.colspan))
|
|
1153
|
+
cols_b = set(range(cell_b.original_col, cell_b.original_col + cell_b.colspan))
|
|
1154
|
+
|
|
1155
|
+
if cols_a & cols_b: # Intersection
|
|
1156
|
+
return True, f"Cells share columns: {cols_a & cols_b}"
|
|
1157
|
+
else:
|
|
1158
|
+
return False, f"Cells do not share any column. A: cols {cols_a}, B: cols {cols_b}"
|
|
1159
|
+
|
|
1160
|
+
|
|
1161
|
+
class TableHeaderChainRule(ParseTestRule):
|
|
1162
|
+
"""Test rule to verify a data cell has the correct header chain."""
|
|
1163
|
+
|
|
1164
|
+
def __init__(self, rule_data: ParseTableHeaderChainRule | dict):
|
|
1165
|
+
super().__init__(rule_data)
|
|
1166
|
+
rule_data = cast(ParseTableHeaderChainRule, self._rule_data)
|
|
1167
|
+
|
|
1168
|
+
if self.type != TestType.TABLE_HEADER_CHAIN.value:
|
|
1169
|
+
raise ValueError(f"Invalid type for TableHeaderChainRule: {self.type}")
|
|
1170
|
+
|
|
1171
|
+
self.data_cell = normalize_text(rule_data.data_cell)
|
|
1172
|
+
self.column_headers = rule_data.column_headers
|
|
1173
|
+
self.row_headers = rule_data.row_headers
|
|
1174
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
1175
|
+
|
|
1176
|
+
if not self.data_cell:
|
|
1177
|
+
raise ValueError("data_cell must be provided")
|
|
1178
|
+
if not self.column_headers and not self.row_headers:
|
|
1179
|
+
raise ValueError("At least one of column_headers or row_headers must be provided")
|
|
1180
|
+
|
|
1181
|
+
def _get_column_headers(self, grid: ResolvedGrid, data_row: int, data_col: int) -> list[str]:
|
|
1182
|
+
"""Get all column headers above the data cell."""
|
|
1183
|
+
headers = []
|
|
1184
|
+
seen_cells: set[tuple[int, int]] = set()
|
|
1185
|
+
|
|
1186
|
+
for row_idx in range(data_row):
|
|
1187
|
+
cell = grid.cells[row_idx][data_col]
|
|
1188
|
+
if cell is None:
|
|
1189
|
+
continue
|
|
1190
|
+
# Use original position as key to avoid duplicates
|
|
1191
|
+
cell_key = (cell.original_row, cell.original_col)
|
|
1192
|
+
if cell_key in seen_cells:
|
|
1193
|
+
continue
|
|
1194
|
+
seen_cells.add(cell_key)
|
|
1195
|
+
|
|
1196
|
+
if cell.text:
|
|
1197
|
+
headers.append(cell.text)
|
|
1198
|
+
|
|
1199
|
+
return headers
|
|
1200
|
+
|
|
1201
|
+
def _get_row_headers(self, grid: ResolvedGrid, data_row: int, data_col: int) -> list[str]:
|
|
1202
|
+
"""Get all row headers to the left of the data cell."""
|
|
1203
|
+
headers = []
|
|
1204
|
+
seen_cells: set[tuple[int, int]] = set()
|
|
1205
|
+
|
|
1206
|
+
for col_idx in range(data_col):
|
|
1207
|
+
cell = grid.cells[data_row][col_idx]
|
|
1208
|
+
if cell is None:
|
|
1209
|
+
continue
|
|
1210
|
+
# Use original position as key to avoid duplicates
|
|
1211
|
+
cell_key = (cell.original_row, cell.original_col)
|
|
1212
|
+
if cell_key in seen_cells:
|
|
1213
|
+
continue
|
|
1214
|
+
seen_cells.add(cell_key)
|
|
1215
|
+
|
|
1216
|
+
if cell.text:
|
|
1217
|
+
headers.append(cell.text)
|
|
1218
|
+
|
|
1219
|
+
return headers
|
|
1220
|
+
|
|
1221
|
+
def _fuzzy_list_match(self, expected: list[str], actual: list[str], threshold: float = 0.8) -> tuple[bool, str]:
|
|
1222
|
+
"""Check if two lists match using fuzzy matching."""
|
|
1223
|
+
if len(expected) != len(actual):
|
|
1224
|
+
return (
|
|
1225
|
+
False,
|
|
1226
|
+
f"Length mismatch: expected {len(expected)} headers, got {len(actual)}",
|
|
1227
|
+
)
|
|
1228
|
+
|
|
1229
|
+
for i, (exp, act) in enumerate(zip(expected, actual, strict=False)):
|
|
1230
|
+
exp_norm = normalize_text(exp)
|
|
1231
|
+
act_norm = normalize_text(act)
|
|
1232
|
+
similarity = fuzz.ratio(exp_norm, act_norm) / 100.0
|
|
1233
|
+
if similarity < threshold:
|
|
1234
|
+
return (
|
|
1235
|
+
False,
|
|
1236
|
+
f"Header {i} mismatch: expected '{exp}', got '{act}' (similarity: {similarity:.2f})",
|
|
1237
|
+
)
|
|
1238
|
+
|
|
1239
|
+
return True, ""
|
|
1240
|
+
|
|
1241
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1242
|
+
"""Check if data cell has correct header chain."""
|
|
1243
|
+
grids = find_all_html_tables(content)
|
|
1244
|
+
if not grids:
|
|
1245
|
+
return False, "No HTML tables found in content"
|
|
1246
|
+
|
|
1247
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
1248
|
+
if self.table_anchor_cells:
|
|
1249
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1250
|
+
if anchor_result.grid is not None:
|
|
1251
|
+
grids = [anchor_result.grid]
|
|
1252
|
+
elif anchor_result.is_ambiguous:
|
|
1253
|
+
return False, (
|
|
1254
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1255
|
+
f"tables - could not uniquely identify target table"
|
|
1256
|
+
)
|
|
1257
|
+
else:
|
|
1258
|
+
return False, "Table anchor cells not found in any table"
|
|
1259
|
+
|
|
1260
|
+
match = find_cell_in_grids(grids, self.data_cell)
|
|
1261
|
+
if not match:
|
|
1262
|
+
return False, f"Data cell '{self.data_cell}' not found in target table"
|
|
1263
|
+
|
|
1264
|
+
grid, cell, row_idx, col_idx = match
|
|
1265
|
+
|
|
1266
|
+
errors = []
|
|
1267
|
+
|
|
1268
|
+
# Check column headers if expected
|
|
1269
|
+
if self.column_headers:
|
|
1270
|
+
actual_col_headers = self._get_column_headers(grid, row_idx, col_idx)
|
|
1271
|
+
passed, err = self._fuzzy_list_match(self.column_headers, actual_col_headers)
|
|
1272
|
+
if not passed:
|
|
1273
|
+
errors.append(f"Column headers: {err}. Expected: {self.column_headers}, Got: {actual_col_headers}")
|
|
1274
|
+
|
|
1275
|
+
# Check row headers if expected
|
|
1276
|
+
if self.row_headers:
|
|
1277
|
+
actual_row_headers = self._get_row_headers(grid, row_idx, col_idx)
|
|
1278
|
+
passed, err = self._fuzzy_list_match(self.row_headers, actual_row_headers)
|
|
1279
|
+
if not passed:
|
|
1280
|
+
errors.append(f"Row headers: {err}. Expected: {self.row_headers}, Got: {actual_row_headers}")
|
|
1281
|
+
|
|
1282
|
+
if errors:
|
|
1283
|
+
return False, "; ".join(errors)
|
|
1284
|
+
else:
|
|
1285
|
+
return True, f"Header chain verified for '{self.data_cell}'"
|
|
1286
|
+
|
|
1287
|
+
|
|
1288
|
+
# =============================================================================
|
|
1289
|
+
# Table Adjacency and Header Rules
|
|
1290
|
+
# =============================================================================
|
|
1291
|
+
|
|
1292
|
+
|
|
1293
|
+
class TableAdjacentRule(ParseTestRule):
|
|
1294
|
+
"""
|
|
1295
|
+
Base class for table adjacency rules.
|
|
1296
|
+
|
|
1297
|
+
Tests that anchor_cell has expected_neighbor in a specific direction.
|
|
1298
|
+
Handles duplicate anchor cells by checking ALL occurrences.
|
|
1299
|
+
"""
|
|
1300
|
+
|
|
1301
|
+
def __init__(self, rule_data: AdjacentTableRuleData | dict):
|
|
1302
|
+
super().__init__(rule_data)
|
|
1303
|
+
rule_data = cast(AdjacentTableRuleData, self._rule_data)
|
|
1304
|
+
|
|
1305
|
+
self.anchor_cell = normalize_text(rule_data.anchor_cell)
|
|
1306
|
+
self.expected_neighbor = normalize_text(rule_data.expected_neighbor)
|
|
1307
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
1308
|
+
self.direction = "" # Set by subclass
|
|
1309
|
+
|
|
1310
|
+
if not self.anchor_cell:
|
|
1311
|
+
raise ValueError("anchor_cell must be provided")
|
|
1312
|
+
if not self.expected_neighbor:
|
|
1313
|
+
raise ValueError("expected_neighbor must be provided")
|
|
1314
|
+
|
|
1315
|
+
def _get_neighbor_position(self, row: int, col: int, grid: ResolvedGrid) -> tuple[int, int] | None:
|
|
1316
|
+
"""Get neighbor position based on direction."""
|
|
1317
|
+
if self.direction == "up" and row > 0:
|
|
1318
|
+
return (row - 1, col)
|
|
1319
|
+
elif self.direction == "down" and row < grid.num_rows - 1:
|
|
1320
|
+
return (row + 1, col)
|
|
1321
|
+
elif self.direction == "left" and col > 0:
|
|
1322
|
+
return (row, col - 1)
|
|
1323
|
+
elif self.direction == "right" and col < grid.num_cols - 1:
|
|
1324
|
+
return (row, col + 1)
|
|
1325
|
+
return None
|
|
1326
|
+
|
|
1327
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1328
|
+
grids = find_all_html_tables(content)
|
|
1329
|
+
if not grids:
|
|
1330
|
+
return False, "No HTML tables found"
|
|
1331
|
+
|
|
1332
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
1333
|
+
if self.table_anchor_cells:
|
|
1334
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1335
|
+
if anchor_result.grid is not None:
|
|
1336
|
+
grids = [anchor_result.grid]
|
|
1337
|
+
elif anchor_result.is_ambiguous:
|
|
1338
|
+
return False, (
|
|
1339
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1340
|
+
f"tables - could not uniquely identify target table"
|
|
1341
|
+
)
|
|
1342
|
+
else:
|
|
1343
|
+
return False, "Table anchor cells not found in any table"
|
|
1344
|
+
|
|
1345
|
+
for grid in grids:
|
|
1346
|
+
for row_idx, row in enumerate(grid.cells):
|
|
1347
|
+
for col_idx, cell in enumerate(row):
|
|
1348
|
+
if cell is None:
|
|
1349
|
+
continue
|
|
1350
|
+
if cell.original_row != row_idx or cell.original_col != col_idx:
|
|
1351
|
+
continue
|
|
1352
|
+
|
|
1353
|
+
similarity = fuzz.ratio(self.anchor_cell, cell.text) / 100.0
|
|
1354
|
+
if similarity < CELL_FUZZY_MATCH_THRESHOLD:
|
|
1355
|
+
continue
|
|
1356
|
+
|
|
1357
|
+
neighbor_pos = self._get_neighbor_position(row_idx, col_idx, grid)
|
|
1358
|
+
if neighbor_pos is None:
|
|
1359
|
+
continue
|
|
1360
|
+
|
|
1361
|
+
neighbor = grid.cells[neighbor_pos[0]][neighbor_pos[1]]
|
|
1362
|
+
if neighbor is None:
|
|
1363
|
+
continue
|
|
1364
|
+
|
|
1365
|
+
neighbor_sim = fuzz.ratio(self.expected_neighbor, neighbor.text) / 100.0
|
|
1366
|
+
if neighbor_sim >= CELL_FUZZY_MATCH_THRESHOLD:
|
|
1367
|
+
return True, ""
|
|
1368
|
+
|
|
1369
|
+
return False, f"No '{self.anchor_cell}' has '{self.expected_neighbor}' {self.direction}"
|
|
1370
|
+
|
|
1371
|
+
|
|
1372
|
+
class TableAdjacentUpRule(TableAdjacentRule):
|
|
1373
|
+
def __init__(self, rule_data: ParseTableAdjacentUpRule | dict):
|
|
1374
|
+
super().__init__(rule_data)
|
|
1375
|
+
rule_data = cast(ParseTableAdjacentUpRule, self._rule_data)
|
|
1376
|
+
if self.type != TestType.TABLE_ADJACENT_UP.value:
|
|
1377
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1378
|
+
self.direction = "up"
|
|
1379
|
+
|
|
1380
|
+
|
|
1381
|
+
class TableAdjacentDownRule(TableAdjacentRule):
|
|
1382
|
+
def __init__(self, rule_data: ParseTableAdjacentDownRule | dict):
|
|
1383
|
+
super().__init__(rule_data)
|
|
1384
|
+
rule_data = cast(ParseTableAdjacentDownRule, self._rule_data)
|
|
1385
|
+
if self.type != TestType.TABLE_ADJACENT_DOWN.value:
|
|
1386
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1387
|
+
self.direction = "down"
|
|
1388
|
+
|
|
1389
|
+
|
|
1390
|
+
class TableAdjacentLeftRule(TableAdjacentRule):
|
|
1391
|
+
def __init__(self, rule_data: ParseTableAdjacentLeftRule | dict):
|
|
1392
|
+
super().__init__(rule_data)
|
|
1393
|
+
rule_data = cast(ParseTableAdjacentLeftRule, self._rule_data)
|
|
1394
|
+
if self.type != TestType.TABLE_ADJACENT_LEFT.value:
|
|
1395
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1396
|
+
self.direction = "left"
|
|
1397
|
+
|
|
1398
|
+
|
|
1399
|
+
class TableAdjacentRightRule(TableAdjacentRule):
|
|
1400
|
+
def __init__(self, rule_data: ParseTableAdjacentRightRule | dict):
|
|
1401
|
+
super().__init__(rule_data)
|
|
1402
|
+
rule_data = cast(ParseTableAdjacentRightRule, self._rule_data)
|
|
1403
|
+
if self.type != TestType.TABLE_ADJACENT_RIGHT.value:
|
|
1404
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1405
|
+
self.direction = "right"
|
|
1406
|
+
|
|
1407
|
+
|
|
1408
|
+
class TableTopHeaderRule(ParseTestRule):
|
|
1409
|
+
"""
|
|
1410
|
+
Test that a data cell has a specific column header above it.
|
|
1411
|
+
|
|
1412
|
+
Handles duplicate data cells by checking ALL occurrences.
|
|
1413
|
+
"""
|
|
1414
|
+
|
|
1415
|
+
def __init__(self, rule_data: ParseTableTopHeaderRule | dict):
|
|
1416
|
+
super().__init__(rule_data)
|
|
1417
|
+
rule_data = cast(ParseTableTopHeaderRule, self._rule_data)
|
|
1418
|
+
|
|
1419
|
+
if self.type != TestType.TABLE_TOP_HEADER.value:
|
|
1420
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1421
|
+
self.data_cell = normalize_text(rule_data.data_cell)
|
|
1422
|
+
self.expected_header = normalize_text(rule_data.expected_header)
|
|
1423
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
1424
|
+
|
|
1425
|
+
if not self.data_cell:
|
|
1426
|
+
raise ValueError("data_cell must be provided")
|
|
1427
|
+
if not self.expected_header:
|
|
1428
|
+
raise ValueError("expected_header must be provided")
|
|
1429
|
+
|
|
1430
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1431
|
+
grids = find_all_html_tables(content)
|
|
1432
|
+
if not grids:
|
|
1433
|
+
return False, "No HTML tables found"
|
|
1434
|
+
|
|
1435
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
1436
|
+
if self.table_anchor_cells:
|
|
1437
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1438
|
+
if anchor_result.grid is not None:
|
|
1439
|
+
grids = [anchor_result.grid]
|
|
1440
|
+
elif anchor_result.is_ambiguous:
|
|
1441
|
+
return False, (
|
|
1442
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1443
|
+
f"tables - could not uniquely identify target table"
|
|
1444
|
+
)
|
|
1445
|
+
else:
|
|
1446
|
+
return False, "Table anchor cells not found in any table"
|
|
1447
|
+
|
|
1448
|
+
for grid in grids:
|
|
1449
|
+
for row_idx, row in enumerate(grid.cells):
|
|
1450
|
+
for col_idx, cell in enumerate(row):
|
|
1451
|
+
if cell is None:
|
|
1452
|
+
continue
|
|
1453
|
+
if cell.original_row != row_idx or cell.original_col != col_idx:
|
|
1454
|
+
continue
|
|
1455
|
+
|
|
1456
|
+
similarity = fuzz.ratio(self.data_cell, cell.text) / 100.0
|
|
1457
|
+
if similarity < CELL_FUZZY_MATCH_THRESHOLD:
|
|
1458
|
+
continue
|
|
1459
|
+
|
|
1460
|
+
# Look above for header
|
|
1461
|
+
for header_row in range(row_idx):
|
|
1462
|
+
header_cell = grid.cells[header_row][col_idx]
|
|
1463
|
+
if header_cell is None:
|
|
1464
|
+
continue
|
|
1465
|
+
|
|
1466
|
+
header_sim = fuzz.ratio(self.expected_header, header_cell.text) / 100.0
|
|
1467
|
+
if header_sim >= CELL_FUZZY_MATCH_THRESHOLD:
|
|
1468
|
+
return True, ""
|
|
1469
|
+
|
|
1470
|
+
return False, f"No '{self.data_cell}' has header '{self.expected_header}' above"
|
|
1471
|
+
|
|
1472
|
+
|
|
1473
|
+
class TableLeftHeaderRule(ParseTestRule):
|
|
1474
|
+
"""
|
|
1475
|
+
Test that a data cell has a specific row header to its left.
|
|
1476
|
+
|
|
1477
|
+
Handles duplicate data cells by checking ALL occurrences.
|
|
1478
|
+
"""
|
|
1479
|
+
|
|
1480
|
+
def __init__(self, rule_data: ParseTableLeftHeaderRule | dict):
|
|
1481
|
+
super().__init__(rule_data)
|
|
1482
|
+
rule_data = cast(ParseTableLeftHeaderRule, self._rule_data)
|
|
1483
|
+
|
|
1484
|
+
if self.type != TestType.TABLE_LEFT_HEADER.value:
|
|
1485
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1486
|
+
self.data_cell = normalize_text(rule_data.data_cell)
|
|
1487
|
+
self.expected_header = normalize_text(rule_data.expected_header)
|
|
1488
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
1489
|
+
|
|
1490
|
+
if not self.data_cell:
|
|
1491
|
+
raise ValueError("data_cell must be provided")
|
|
1492
|
+
if not self.expected_header:
|
|
1493
|
+
raise ValueError("expected_header must be provided")
|
|
1494
|
+
|
|
1495
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1496
|
+
grids = find_all_html_tables(content)
|
|
1497
|
+
if not grids:
|
|
1498
|
+
return False, "No HTML tables found"
|
|
1499
|
+
|
|
1500
|
+
# Step 1: Find the correct table using anchor cells if provided
|
|
1501
|
+
if self.table_anchor_cells:
|
|
1502
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1503
|
+
if anchor_result.grid is not None:
|
|
1504
|
+
grids = [anchor_result.grid]
|
|
1505
|
+
elif anchor_result.is_ambiguous:
|
|
1506
|
+
return False, (
|
|
1507
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1508
|
+
f"tables - could not uniquely identify target table"
|
|
1509
|
+
)
|
|
1510
|
+
else:
|
|
1511
|
+
return False, "Table anchor cells not found in any table"
|
|
1512
|
+
|
|
1513
|
+
for grid in grids:
|
|
1514
|
+
for row_idx, row in enumerate(grid.cells):
|
|
1515
|
+
for col_idx, cell in enumerate(row):
|
|
1516
|
+
if cell is None:
|
|
1517
|
+
continue
|
|
1518
|
+
if cell.original_row != row_idx or cell.original_col != col_idx:
|
|
1519
|
+
continue
|
|
1520
|
+
|
|
1521
|
+
similarity = fuzz.ratio(self.data_cell, cell.text) / 100.0
|
|
1522
|
+
if similarity < CELL_FUZZY_MATCH_THRESHOLD:
|
|
1523
|
+
continue
|
|
1524
|
+
|
|
1525
|
+
# Look left for header
|
|
1526
|
+
for header_col in range(col_idx):
|
|
1527
|
+
header_cell = grid.cells[row_idx][header_col]
|
|
1528
|
+
if header_cell is None:
|
|
1529
|
+
continue
|
|
1530
|
+
|
|
1531
|
+
header_sim = fuzz.ratio(self.expected_header, header_cell.text) / 100.0
|
|
1532
|
+
if header_sim >= CELL_FUZZY_MATCH_THRESHOLD:
|
|
1533
|
+
return True, ""
|
|
1534
|
+
|
|
1535
|
+
return False, f"No '{self.data_cell}' has header '{self.expected_header}' to left"
|
|
1536
|
+
|
|
1537
|
+
|
|
1538
|
+
# =============================================================================
|
|
1539
|
+
# Table Border Rules (Negative Tests)
|
|
1540
|
+
# =============================================================================
|
|
1541
|
+
|
|
1542
|
+
|
|
1543
|
+
class TableNoBorderRule(ParseTestRule):
|
|
1544
|
+
"""
|
|
1545
|
+
Base class for table border rules that verify absence of cells.
|
|
1546
|
+
|
|
1547
|
+
These are "negative tests" that ensure predicted tables don't have
|
|
1548
|
+
extra rows/columns beyond the ground truth boundaries.
|
|
1549
|
+
"""
|
|
1550
|
+
|
|
1551
|
+
def __init__(self, rule_data: NoBorderTableRuleData | dict):
|
|
1552
|
+
super().__init__(rule_data)
|
|
1553
|
+
rule_data = cast(NoBorderTableRuleData, self._rule_data)
|
|
1554
|
+
|
|
1555
|
+
self.cell = normalize_text(rule_data.cell)
|
|
1556
|
+
self.table_anchor_cells = rule_data.table_anchor_cells
|
|
1557
|
+
self.direction = "" # Set by subclass: "left", "right", "up", "down"
|
|
1558
|
+
|
|
1559
|
+
if not self.cell:
|
|
1560
|
+
raise ValueError("cell must be provided")
|
|
1561
|
+
|
|
1562
|
+
def _get_neighbor_position(self, row: int, col: int, grid: ResolvedGrid) -> tuple[int, int] | None:
|
|
1563
|
+
"""Get neighbor position based on direction. Returns None if out of bounds."""
|
|
1564
|
+
if self.direction == "up":
|
|
1565
|
+
return (row - 1, col) if row > 0 else None
|
|
1566
|
+
elif self.direction == "down":
|
|
1567
|
+
return (row + 1, col) if row < grid.num_rows - 1 else None
|
|
1568
|
+
elif self.direction == "left":
|
|
1569
|
+
return (row, col - 1) if col > 0 else None
|
|
1570
|
+
elif self.direction == "right":
|
|
1571
|
+
return (row, col + 1) if col < grid.num_cols - 1 else None
|
|
1572
|
+
return None
|
|
1573
|
+
|
|
1574
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
1575
|
+
grids = find_all_html_tables(content)
|
|
1576
|
+
if not grids:
|
|
1577
|
+
return False, "No HTML tables found"
|
|
1578
|
+
|
|
1579
|
+
# Find the correct table using anchor cells if provided
|
|
1580
|
+
if self.table_anchor_cells:
|
|
1581
|
+
anchor_result = find_table_by_anchors(grids, self.table_anchor_cells)
|
|
1582
|
+
if anchor_result.grid is not None:
|
|
1583
|
+
grids = [anchor_result.grid]
|
|
1584
|
+
elif anchor_result.is_ambiguous:
|
|
1585
|
+
return False, (
|
|
1586
|
+
f"[AMBIGUOUS ANCHORS] Anchors matched {anchor_result.num_candidates} "
|
|
1587
|
+
f"tables - could not uniquely identify target table"
|
|
1588
|
+
)
|
|
1589
|
+
else:
|
|
1590
|
+
return False, "Table anchor cells not found in any table"
|
|
1591
|
+
|
|
1592
|
+
for grid in grids:
|
|
1593
|
+
for row_idx, row in enumerate(grid.cells):
|
|
1594
|
+
for col_idx, cell in enumerate(row):
|
|
1595
|
+
if cell is None:
|
|
1596
|
+
continue
|
|
1597
|
+
if cell.original_row != row_idx or cell.original_col != col_idx:
|
|
1598
|
+
continue
|
|
1599
|
+
|
|
1600
|
+
similarity = fuzz.ratio(self.cell, cell.text) / 100.0
|
|
1601
|
+
if similarity < CELL_FUZZY_MATCH_THRESHOLD:
|
|
1602
|
+
continue
|
|
1603
|
+
|
|
1604
|
+
# Found the cell - now check if there's NO neighbor in the direction
|
|
1605
|
+
neighbor_pos = self._get_neighbor_position(row_idx, col_idx, grid)
|
|
1606
|
+
|
|
1607
|
+
if neighbor_pos is None:
|
|
1608
|
+
# No neighbor position possible (at grid boundary) - PASS
|
|
1609
|
+
return True, ""
|
|
1610
|
+
|
|
1611
|
+
neighbor = grid.cells[neighbor_pos[0]][neighbor_pos[1]]
|
|
1612
|
+
if neighbor is None or not neighbor.text.strip():
|
|
1613
|
+
# No neighbor cell or empty cell - PASS
|
|
1614
|
+
return True, ""
|
|
1615
|
+
|
|
1616
|
+
# There IS a neighbor - this is a FAILURE for border tests
|
|
1617
|
+
return (
|
|
1618
|
+
False,
|
|
1619
|
+
f"Cell '{self.cell}' has unexpected neighbor '{neighbor.text}' to {self.direction}",
|
|
1620
|
+
)
|
|
1621
|
+
|
|
1622
|
+
return False, f"Could not find cell '{self.cell}' in any table"
|
|
1623
|
+
|
|
1624
|
+
|
|
1625
|
+
class TableNoLeftRule(TableNoBorderRule):
|
|
1626
|
+
"""Test that a cell has no cell to its left (leftmost column boundary)."""
|
|
1627
|
+
|
|
1628
|
+
def __init__(self, rule_data: ParseTableNoLeftRule | dict):
|
|
1629
|
+
super().__init__(rule_data)
|
|
1630
|
+
rule_data = cast(ParseTableNoLeftRule, self._rule_data)
|
|
1631
|
+
if self.type != TestType.TABLE_NO_LEFT.value:
|
|
1632
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1633
|
+
self.direction = "left"
|
|
1634
|
+
|
|
1635
|
+
|
|
1636
|
+
class TableNoRightRule(TableNoBorderRule):
|
|
1637
|
+
"""Test that a cell has no cell to its right (rightmost column boundary)."""
|
|
1638
|
+
|
|
1639
|
+
def __init__(self, rule_data: ParseTableNoRightRule | dict):
|
|
1640
|
+
super().__init__(rule_data)
|
|
1641
|
+
rule_data = cast(ParseTableNoRightRule, self._rule_data)
|
|
1642
|
+
if self.type != TestType.TABLE_NO_RIGHT.value:
|
|
1643
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1644
|
+
self.direction = "right"
|
|
1645
|
+
|
|
1646
|
+
|
|
1647
|
+
class TableNoAboveRule(TableNoBorderRule):
|
|
1648
|
+
"""Test that a cell has no cell above it (top row boundary)."""
|
|
1649
|
+
|
|
1650
|
+
def __init__(self, rule_data: ParseTableNoAboveRule | dict):
|
|
1651
|
+
super().__init__(rule_data)
|
|
1652
|
+
rule_data = cast(ParseTableNoAboveRule, self._rule_data)
|
|
1653
|
+
if self.type != TestType.TABLE_NO_ABOVE.value:
|
|
1654
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1655
|
+
self.direction = "up"
|
|
1656
|
+
|
|
1657
|
+
|
|
1658
|
+
class TableNoBelowRule(TableNoBorderRule):
|
|
1659
|
+
"""Test that a cell has no cell below it (bottom row boundary)."""
|
|
1660
|
+
|
|
1661
|
+
def __init__(self, rule_data: ParseTableNoBelowRule | dict):
|
|
1662
|
+
super().__init__(rule_data)
|
|
1663
|
+
rule_data = cast(ParseTableNoBelowRule, self._rule_data)
|
|
1664
|
+
if self.type != TestType.TABLE_NO_BELOW.value:
|
|
1665
|
+
raise ValueError(f"Invalid type: {self.type}")
|
|
1666
|
+
self.direction = "down"
|