parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,2274 @@
1
+ """Form field test rule.
2
+
3
+ A `form_field` rule locates a labeled field in the parsed markdown/HTML and
4
+ checks its value. Three value types are supported in v0.1: ``text``,
5
+ ``checkbox``, and ``signature``. The matcher tries a small set of
6
+ high-confidence patterns:
7
+
8
+ - Bold-colon (``**Label:** value`` and ``**Label**: value``) — supports
9
+ multiple bold-colon pairs on the same line.
10
+ - Plain colon on its own line (``Label: value``).
11
+ - 2-column markdown tables AND 2-column HTML tables (label in first cell,
12
+ value in second).
13
+ - Per-line checkbox tokenization for inline groups, handling both
14
+ glyph-first (``☐ Single ☑ Married``) and label-first
15
+ (``Single ☐ Married ☑``) orderings.
16
+ - Markdown task-list checkboxes (``- [x] Label`` / ``- [ ] Label``).
17
+ - Multi-label yes/no checkbox groups, e.g.
18
+ ``["Multistage cement?", "No"]`` matches
19
+ ``Multistage cement? Yes [ ] No [x]``.
20
+ - Multi-label table cells where one value is identified by a row/column
21
+ label set, e.g. ``["PLUG #1", "Cementing Date"]``. Label-list order is
22
+ ignored; all labels must match anchors around the same value cell.
23
+
24
+ When the rule has a ``page`` and the metric injects a ``parse_output``,
25
+ matching is scoped to that page's markdown only.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import re
31
+ from typing import cast
32
+
33
+ from bs4 import BeautifulSoup
34
+ from rapidfuzz import fuzz
35
+
36
+ from parse_bench.evaluation.metrics.parse.rules_base import (
37
+ CELL_FUZZY_MATCH_THRESHOLD,
38
+ ParseTestRule,
39
+ )
40
+ from parse_bench.evaluation.metrics.parse.rules_chart import normalize_number_string
41
+ from parse_bench.evaluation.metrics.parse.table_parsing import (
42
+ TableData,
43
+ parse_html_tables,
44
+ parse_markdown_tables,
45
+ )
46
+ from parse_bench.evaluation.metrics.parse.test_types import TestType
47
+ from parse_bench.evaluation.metrics.parse.utils import normalize_text
48
+ from parse_bench.test_cases.parse_rule_schemas import ParseFormFieldRule
49
+
50
+ # Glyphs that represent a checked / unchecked state. Sourced from the most
51
+ # common Unicode shapes parsers emit when surfacing form widgets. The extra
52
+ # circle/dot glyphs (◉●⦿/○◯⊙) appear in Gemini and OpenAI outputs which mirror
53
+ # radio-button widgets; the extra X glyphs (⊠⊗) appear in IRS/USCIS forms.
54
+ _CHECKED_GLYPHS = "☑☒▣✓✔◉●⦿⊠⊗"
55
+ _UNCHECKED_GLYPHS = "☐□○◯⊙"
56
+ _CHECKBOX_GLYPHS = _CHECKED_GLYPHS + _UNCHECKED_GLYPHS
57
+ _GLYPH_RE = re.compile(f"[{_CHECKBOX_GLYPHS}]")
58
+ # Combined marker regex: either a single Unicode glyph OR an ASCII bracket
59
+ # pair ``[x]`` / ``[ ]`` (with optional ``\`` escapes around the brackets).
60
+ # Used by the per-line tokenizer so inline ASCII checkbox groups
61
+ # (``\[x] Single \[ ] Married``) are parsed the same as Unicode-glyph groups.
62
+ _MARKER_RE = re.compile(rf"[{_CHECKBOX_GLYPHS}]|\\?\[[ xX]\\?\]")
63
+
64
+
65
+ def _marker_is_checked(token: str) -> bool:
66
+ """Decide if a checkbox marker token represents the checked state."""
67
+
68
+ if len(token) == 1:
69
+ return token in _CHECKED_GLYPHS
70
+ return any(c in "xX" for c in token)
71
+
72
+
73
+ # Boolean coercion table for textual yes/no values.
74
+ _TRUTHY_TEXT = {"yes", "y", "true", "t", "1", "checked", "x", "selected", "on"}
75
+ _FALSY_TEXT = {"no", "n", "false", "f", "0", "unchecked", "unselected", "off", ""}
76
+
77
+
78
+ _PARTIAL_RATIO_THRESHOLD = 0.90
79
+ _PARTIAL_RATIO_MIN_LEN = 6
80
+ # Penalty applied to the partial-ratio score so a partial hit cannot beat
81
+ # an equally strong strict-ratio hit. Empirically 0.05 keeps partial 1.0
82
+ # above strict 0.86 (so e.g. ``API NO. (if available)`` still resolves
83
+ # against GT ``API NO.`` when the only candidate is the full noisy label)
84
+ # while preventing partial 0.95 (``County`` ⊂ ``Country``) from beating a
85
+ # strict 1.0 ``Country`` match on the *correct* row.
86
+ _PARTIAL_RATIO_PENALTY = 0.05
87
+
88
+
89
+ def _strip_label_punct(s: str) -> str:
90
+ """Remove punctuation that varies between abbreviation styles.
91
+
92
+ ``K.B.`` vs ``KB``, ``D.F.`` vs ``DF``, ``API NO:`` vs ``API NO``,
93
+ ``Tel.`` vs ``Tel`` are the same label semantically. This strips
94
+ dots, colons, and commas — separators that the parser may add or drop
95
+ while preserving the underlying tokens. Operates after
96
+ ``normalize_text`` so it sees a case-folded, whitespace-collapsed
97
+ string.
98
+ """
99
+
100
+ s = s.replace(".", "")
101
+ s = s.replace(":", "")
102
+ s = s.replace(",", "")
103
+ s = re.sub(r"\s+", " ", s).strip()
104
+ return s
105
+
106
+
107
+ def _compact_label_text_for_distance(s: str) -> str:
108
+ return re.sub(r"[^0-9a-z]+", "", normalize_text(s))
109
+
110
+
111
+ def _label_match_score(candidate: str, label: str, max_diffs: int | float = 0) -> float:
112
+ """Score how well *candidate* matches *label* on the ``_label_matches`` axes.
113
+
114
+ Returns ``0.0`` when the candidate fails every path (same as
115
+ ``_label_matches`` returning False); otherwise returns a value in
116
+ ``(0.0, 1.0]`` where higher means a better label match.
117
+
118
+ Why scoring instead of a bool: when the document contains two visually
119
+ similar labels (the classic ``Country`` / ``County`` collision), the
120
+ old first-match-wins iteration would latch onto whichever fuzzy hit
121
+ came first and return that row's value. With a scoring function, the
122
+ caller can collect every candidate and pick the *best* match — an
123
+ exact ``Country`` (score ``1.0``) beats a fuzzy ``County`` (score
124
+ ``~0.92``) even when ``County`` appears earlier in the markdown.
125
+
126
+ Scoring:
127
+
128
+ - Strict-ratio path (``fuzz.ratio >= CELL_FUZZY_MATCH_THRESHOLD``):
129
+ score = the ratio itself.
130
+ - Partial-ratio fallback (``fuzz.partial_ratio >= _PARTIAL_RATIO_THRESHOLD``
131
+ with ``shorter >= _PARTIAL_RATIO_MIN_LEN``): score = the partial
132
+ ratio minus ``_PARTIAL_RATIO_PENALTY`` (currently 0.05). The penalty
133
+ keeps partial hits strictly below same-strength strict hits — a
134
+ partial 1.0 (``County`` substring inside ``County, TX``) scores
135
+ ``0.95``, which still loses to any strict-ratio match >= 0.95 but
136
+ wins over a strict-ratio 0.86 fuzzy match.
137
+ - Punctuation-stripped exact path
138
+ (``_strip_label_punct(cand) == _strip_label_punct(lbl)``): score
139
+ ``1.0``. Exact-equality (not fuzz) keeps this path narrow — short
140
+ labels like ``KB`` won't collide with ``KBC`` (``fuzz.ratio`` happens
141
+ to hit exactly 0.80 between those two strings, which would leak
142
+ through if we ran fuzz on the stripped variants) while still
143
+ matching dotted abbreviation variants like ``K.B.`` ≡ ``KB``.
144
+ - Best path wins when several fire.
145
+ """
146
+
147
+ cand = normalize_text(candidate)
148
+ lbl = normalize_text(label)
149
+ if not cand or not lbl:
150
+ return 0.0
151
+
152
+ best = 0.0
153
+ ratio = fuzz.ratio(cand, lbl) / 100.0
154
+ if ratio >= CELL_FUZZY_MATCH_THRESHOLD:
155
+ best = ratio
156
+ shorter = min(len(cand), len(lbl))
157
+ if shorter >= _PARTIAL_RATIO_MIN_LEN:
158
+ partial = fuzz.partial_ratio(cand, lbl) / 100.0
159
+ if partial >= _PARTIAL_RATIO_THRESHOLD:
160
+ penalized = max(partial - _PARTIAL_RATIO_PENALTY, 0.0)
161
+ if penalized > best:
162
+ best = penalized
163
+
164
+ # Punctuation-stripped exact equality — narrowest of the three paths,
165
+ # only fires when the strip actually collapses two different surface
166
+ # forms onto the same string. Scored at 1.0 so legitimate abbreviation
167
+ # variants beat a coincidental ratio-0.80 collision (the ``K.B.``/
168
+ # ``KBC`` boundary case) when both candidates appear in the document.
169
+ cand_stripped = _strip_label_punct(cand)
170
+ lbl_stripped = _strip_label_punct(lbl)
171
+ if cand_stripped and lbl_stripped and cand_stripped == lbl_stripped:
172
+ if 1.0 > best:
173
+ best = 1.0
174
+
175
+ allowed = int(max_diffs) if max_diffs and max_diffs > 0 else 0
176
+ if allowed > 0:
177
+ cand_compact = _compact_label_text_for_distance(candidate)
178
+ lbl_compact = _compact_label_text_for_distance(label)
179
+ if cand_compact and lbl_compact:
180
+ dist = _levenshtein_distance_at_most(cand_compact, lbl_compact, allowed)
181
+ if dist <= allowed:
182
+ tolerant_score = max(0.01, 1.0 - (dist / max(len(cand_compact), len(lbl_compact), 1)))
183
+ if tolerant_score > best:
184
+ best = tolerant_score
185
+
186
+ return best
187
+
188
+
189
+ def _label_matches(candidate: str, label: str, max_diffs: int | float = 0) -> bool:
190
+ """Boolean predicate over :func:`_label_match_score` for legacy callers.
191
+
192
+ Used by callers that only need a boolean (adjacent-line fallback,
193
+ underscore-blank label-seen detection, checkbox-state matching). The
194
+ main text-value lookup uses :func:`_label_match_score` directly so it
195
+ can score-and-pick-best across multiple candidate KV pairs.
196
+ """
197
+
198
+ return _label_match_score(candidate, label, max_diffs) > 0.0
199
+
200
+
201
+ def _coerce_bool(value: str | bool | list[str]) -> bool | None:
202
+ """Coerce a value to True/False, or None if ambiguous.
203
+
204
+ List inputs are not supported by checkbox semantics and return None;
205
+ the caller surfaces a "must be coercible to bool" error in that case.
206
+ """
207
+
208
+ if isinstance(value, bool):
209
+ return value
210
+ if isinstance(value, list):
211
+ return None
212
+ text = str(value).strip().lower()
213
+ if len(text) == 1 and text in _CHECKED_GLYPHS:
214
+ return True
215
+ if len(text) == 1 and text in _UNCHECKED_GLYPHS:
216
+ return False
217
+ if text in _TRUTHY_TEXT:
218
+ return True
219
+ if text in _FALSY_TEXT:
220
+ return False
221
+ return None
222
+
223
+
224
+ def _value_alternatives(value: str | bool | list[str]) -> list[str]:
225
+ """Return the list of acceptable string values for a text-typed rule.
226
+
227
+ Supports both single-string and list-of-strings GTs. A list lets a rule
228
+ declare multiple acceptable readings for genuinely ambiguous fields
229
+ (e.g. illegible handwriting). The single-string form is the default and
230
+ keeps the GT clean for the common case.
231
+ """
232
+
233
+ if isinstance(value, list):
234
+ return [str(v) for v in value]
235
+ return [str(value)]
236
+
237
+
238
+ def _label_parts_with_indexes(label: str | list[str]) -> list[tuple[str, int]]:
239
+ """Normalize form-field labels and keep original indexes for aligned tolerances."""
240
+ if isinstance(label, list):
241
+ return [(text, idx) for idx, part in enumerate(label) if (text := str(part).strip())]
242
+ text = str(label).strip()
243
+ return [(text, 0)] if text else []
244
+
245
+
246
+ def _label_parts(label: str | list[str]) -> list[str]:
247
+ """Normalize a form-field label into one or more required visible keys."""
248
+
249
+ return [part for part, _ in _label_parts_with_indexes(label)]
250
+
251
+
252
+ def _format_label_for_message(label: str | list[str]) -> str:
253
+ parts = _label_parts(label)
254
+ if len(parts) <= 1:
255
+ return repr(parts[0] if parts else "")
256
+ return repr(parts)
257
+
258
+
259
+ def _label_max_diffs_parts(
260
+ value: int | float | list[int | float],
261
+ count: int,
262
+ source_indexes: list[int] | None = None,
263
+ ) -> list[int]:
264
+ if isinstance(value, list):
265
+ indexes = source_indexes or list(range(count))
266
+ return [
267
+ int(value[idx]) if idx < len(value) and isinstance(value[idx], (int, float)) and value[idx] > 0 else 0
268
+ for idx in indexes[:count]
269
+ ] + [0] * max(count - len(indexes), 0)
270
+ n = int(value) if isinstance(value, (int, float)) and value > 0 else 0
271
+ return [n] * count
272
+
273
+
274
+ def _dedupe_nonempty_text(parts: list[str]) -> list[str]:
275
+ out: list[str] = []
276
+ seen: set[str] = set()
277
+ for part in parts:
278
+ clean = str(part).strip()
279
+ if not clean:
280
+ continue
281
+ key = normalize_text(clean)
282
+ if key in seen:
283
+ continue
284
+ out.append(clean)
285
+ seen.add(key)
286
+ return out
287
+
288
+
289
+ def _multi_col_header_data_pairs(table: TableData) -> list[tuple[str, str]]:
290
+ """For a >2-col table with header rows, yield (col_header, data_value) for
291
+ every (column, data row) pair so a label that names a column matches the
292
+ value in that column's data row(s)."""
293
+
294
+ out: list[tuple[str, str]] = []
295
+ rows, cols = table.data.shape
296
+ if cols <= 2 or rows == 0:
297
+ return out
298
+ header_rows = getattr(table, "header_rows", set()) or set()
299
+ n_header = (max(header_rows) + 1) if header_rows else 1
300
+ for col_idx in range(cols):
301
+ header_text = _column_header_for_index(table, col_idx)
302
+ if not header_text:
303
+ continue
304
+ for row_idx in range(n_header, rows):
305
+ cell_value = str(table.data[row_idx, col_idx]).strip()
306
+ out.append((header_text, cell_value))
307
+ return out
308
+
309
+
310
+ def _iter_html_cell_kv_pairs(content: str) -> list[tuple[str, str]]:
311
+ """Yield (label, value) pairs extracted from HTML cells whose internal
312
+ layout stacks the label above the value via ``<br/>``.
313
+
314
+ Pattern: ``<td>Label<br/><strong>Value</strong></td>``. Common in parsers
315
+ that try to mirror the visual two-line widget within a single cell."""
316
+
317
+ out: list[tuple[str, str]] = []
318
+ if "<table" not in content.lower():
319
+ return out
320
+ soup = BeautifulSoup(content, "lxml")
321
+ for table in soup.find_all("table"):
322
+ for cell in table.find_all(["td", "th"]):
323
+ for br in cell.find_all("br"):
324
+ br.replace_with("\n")
325
+ cell_text = cell.get_text().strip()
326
+ if "\n" not in cell_text:
327
+ continue
328
+ parts = [p.strip() for p in cell_text.split("\n", 1)]
329
+ if len(parts) != 2:
330
+ continue
331
+ label_part, value_part = parts
332
+ label_part = label_part.strip("*_ \t")
333
+ value_part = value_part.strip("*_ \t")
334
+ if label_part:
335
+ out.append((label_part, value_part))
336
+ return out
337
+
338
+
339
+ def _iter_html_table_kv_rows(content: str) -> list[tuple[str, str]]:
340
+ """Yield (label, value) tuples from HTML tables.
341
+
342
+ - 2-col tables: yield each row as ``(col0, col1)`` (label-then-value layout).
343
+ - >2-col tables with header rows: yield ``(col_header, data_row_value)``
344
+ for every column × data row, so a label naming a column matches the
345
+ value in that column's data row.
346
+ - Any cell that contains in-cell ``<br/>`` separators: yield
347
+ ``(top_half, bottom_half)`` so ``<td>Label<br/><strong>Value</strong></td>``
348
+ is captured.
349
+ """
350
+
351
+ out: list[tuple[str, str]] = []
352
+ if "<table" not in content.lower():
353
+ return out
354
+ # Cell-internal label/value (label<br/>value inside one cell) takes
355
+ # precedence over the row-wise 2-col interpretation; otherwise a cell
356
+ # like ``<td>Last Name<br/>Nguyen</td>`` would be mangled into a single
357
+ # blob ``Last Name Nguyen`` by the row-wise path before the cell-level
358
+ # pair is ever consulted.
359
+ out.extend(_iter_html_cell_kv_pairs(content))
360
+ for table in parse_html_tables(content):
361
+ rows, cols = table.data.shape
362
+ if cols == 2:
363
+ for row_idx in range(rows):
364
+ label_text = str(table.data[row_idx, 0]).strip()
365
+ value_text = str(table.data[row_idx, 1]).strip()
366
+ out.append((label_text, value_text))
367
+ elif cols > 2:
368
+ # Header-then-data-row binding (one record per data row).
369
+ # Interleaved label/value layouts inside wide HTML tables
370
+ # (well-log report headers, rotated form pages) are handled
371
+ # downstream by ``_iter_html_cell_neighbor_pairs`` — that
372
+ # iterator classifies each neighbor as label-shaped vs
373
+ # value-shaped before pairing, so the score path never sees
374
+ # spurious ``(LABEL, OTHER_LABEL)`` candidates.
375
+ out.extend(_multi_col_header_data_pairs(table))
376
+ return out
377
+
378
+
379
+ # Heuristic: signals that a cell *looks like* a form-label rather than a value.
380
+ # Used as a tie-breaker by ``_iter_html_cell_neighbor_pairs`` when picking
381
+ # between a right-neighbor and a below-neighbor in wide HTML form tables —
382
+ # we only want to return value-shaped neighbors, not adjacent label cells.
383
+ #
384
+ # A cell is considered label-like when any of these holds:
385
+ # 1. trailing colon (``FILE NO:``);
386
+ # 2. short ALL-CAPS with no digits / no value-style punctuation
387
+ # (``WELL``, ``COMPANY``, ``OTHER SERVICES``);
388
+ # 3. structurally repeats elsewhere in the same table — handled by the
389
+ # caller, which threads the per-table text-count map in.
390
+ #
391
+ # Values like ``LEHMAN #1``, ``42-157-33282``, ``KEBO OIL & GAS, INC.``,
392
+ # ``15-MAY-2023`` keep digits / hashes / commas / parens so they fail
393
+ # heuristic (2) and are correctly classified as value-shaped.
394
+ #
395
+ # Note: ``&`` is intentionally absent from the disqualifier so common
396
+ # value strings like ``"KEBO OIL & GAS, INC."`` (rescued by the comma)
397
+ # stay value-shaped without forcing every label with ``&`` (e.g. an
398
+ # ``"OIL & GAS"`` column header) to be misread as a value. A naked
399
+ # ``"X & Y"`` value with no other punctuation would be misclassified as a
400
+ # label, but that pattern hasn't surfaced in real benchmark data.
401
+ _LABEL_LIKE_DISQUALIFIER_RE = re.compile(r"[\d#@/_,()$%]")
402
+
403
+
404
+ def _cell_text_is_label_like(text: str) -> bool:
405
+ s = text.strip()
406
+ if not s:
407
+ return False
408
+ if s.endswith(":"):
409
+ return True
410
+ # Length cap: typical form labels are short (1-3 words). Long ALL-CAPS
411
+ # strings like ``"PERMITTED FOR RECOMPLETION TO PRODUCE FROM"`` skip
412
+ # heuristic (2) and stay value-shaped, which is the safer default — the
413
+ # cost of mis-flagging a long label is a missed neighbor, but the cost
414
+ # of flagging a long value is returning the wrong neighbor.
415
+ if len(s) > 30:
416
+ return False
417
+ if _LABEL_LIKE_DISQUALIFIER_RE.search(s):
418
+ return False
419
+ if s != s.upper():
420
+ return False
421
+ if not re.search(r"[A-Z]", s):
422
+ return False
423
+ return True
424
+
425
+
426
+ def _iter_md_table_kv_rows(content: str) -> list[tuple[str, str]]:
427
+ """Yield (label, value) tuples from markdown tables.
428
+
429
+ - 2-col tables: yield each row as ``(col0, col1)`` (existing behavior).
430
+ - >2-col tables: yield ``(col_header, data_row_value)`` for every
431
+ column × data row.
432
+ """
433
+
434
+ out: list[tuple[str, str]] = []
435
+ for table in parse_markdown_tables(content):
436
+ rows, cols = table.data.shape
437
+ if cols == 2:
438
+ for row_idx in range(rows):
439
+ label_text = str(table.data[row_idx, 0]).strip()
440
+ value_text = str(table.data[row_idx, 1]).strip()
441
+ out.append((label_text, value_text))
442
+ elif cols > 2:
443
+ out.extend(_multi_col_header_data_pairs(table))
444
+ return out
445
+
446
+
447
+ # Generic HTML tag stripper for label/value normalization. Form-field values
448
+ # never legitimately contain ``<tag>...</tag>`` markup — names, addresses, IDs,
449
+ # and currency don't — but parsers leak HTML wrappers into extracted spans
450
+ # (haiku preserves ``<strong>``/``<td>``, gemini emits ``<u>`` underline-fill,
451
+ # OpenAI sometimes leaves ``</p>``). The pattern is restricted to well-formed
452
+ # HTML element opens/closes: a tag name must start with an ASCII letter and
453
+ # contain only alphanumerics afterwards, optionally followed by a
454
+ # whitespace-introduced attribute run. This deliberately excludes markdown
455
+ # email/URL autolinks like ``<wei.lin@host.com>`` and ``<https://...>``,
456
+ # whose first character after ``<`` is a letter but whose body contains
457
+ # ``.``/``@``/``:`` that disqualify them from the tag-name shape.
458
+ _HTML_TAG_RE = re.compile(r"<\s*/?\s*[a-zA-Z][a-zA-Z0-9]*(?:\s[^<>]*)?\s*/?\s*>")
459
+
460
+
461
+ def _strip_html_tags(s: str) -> str:
462
+ return _HTML_TAG_RE.sub("", s)
463
+
464
+
465
+ # Tagged-line prefix used by some parsers to mark a field-extraction event,
466
+ # e.g. ``[FORM FIELD] Label: value``. The bracketed prefix is parser noise,
467
+ # not part of the label. We only strip it from the start of a candidate
468
+ # label, never mid-string, so legitimate labels containing brackets like
469
+ # ``[Effective Date]`` (uncommon but possible) are preserved unless the
470
+ # bracket is the leading token.
471
+ _LABEL_TAG_PREFIX_RE = re.compile(r"^\s*\[[^\]\n]+\]\s+")
472
+
473
+
474
+ def _trim_value_at_next_field(value: str) -> str:
475
+ """Trim a captured value at a ``| Next Label: ...`` boundary.
476
+
477
+ Some parsers concatenate multiple labelled fields onto one line with
478
+ ``|`` separators (e.g. ``Date: 2026-04-27 | Borrower's Name: Maya | ...``).
479
+ Without this trim, the plain-colon regex captures the entire tail as the
480
+ value of the first field. We only split when the part after the ``|``
481
+ looks like another labelled field (contains ``:``), so legitimate values
482
+ with embedded ``|`` (rare in form data) are preserved.
483
+ """
484
+
485
+ parts = re.split(r"\s+\|\s+", value, maxsplit=1)
486
+ if len(parts) == 2 and ":" in parts[1]:
487
+ return parts[0].strip()
488
+ return value
489
+
490
+
491
+ def _split_pipe_concatenated_pairs(value: str) -> list[tuple[str, str]]:
492
+ """Split a run-on ``Label1: v1 | Label2: v2 | ...`` value tail into pairs.
493
+
494
+ Companion to :func:`_trim_value_at_next_field`. The first call trims the
495
+ value of the *initial* labelled field; this function recovers any
496
+ *subsequent* ``Label: value`` pairs that were riding along on the same
497
+ line so a single-line run-on yields one pair per labelled field.
498
+ """
499
+
500
+ out: list[tuple[str, str]] = []
501
+ if " | " not in value:
502
+ return out
503
+ for segment in re.split(r"\s+\|\s+", value):
504
+ if ":" not in segment:
505
+ continue
506
+ # Same horizontal-only colon split as _PLAIN_COLON_RE so we don't
507
+ # accidentally bleed time-of-day strings ("11:30 AM") into pairs.
508
+ m = re.match(r"^[ \t]*([^:\n*][^:\n]{0,200}?)[ \t]*:[ \t]*(.+?)[ \t]*$", segment)
509
+ if not m:
510
+ continue
511
+ seg_label = _strip_html_tags(m.group(1).strip()).strip()
512
+ seg_label = _LABEL_TAG_PREFIX_RE.sub("", seg_label).strip()
513
+ seg_value = _strip_html_tags(m.group(2).strip()).strip()
514
+ if seg_label:
515
+ out.append((seg_label, seg_value))
516
+ return out
517
+
518
+
519
+ # Bullet-line shape for safe aggregation: ``- item`` / ``* item`` / ``+ item``
520
+ # (with optional leading ``\`` escape some renderers emit). The negative
521
+ # lookahead rejects checkbox-bearing bullets (``- [x] ...``) — those rows
522
+ # describe their own state, not a continuation of the preceding label.
523
+ _AGGREGATE_BULLET_RE = re.compile(r"^\\?[-*+]\s+(?!\\?\[)")
524
+
525
+
526
+ def _aggregate_following_lines(content: str, after_offset: int, max_lines: int = 8) -> str:
527
+ """Collect bullet-list lines after *after_offset* into a single value
528
+ string, joined with ``, ``.
529
+
530
+ This is a narrow fallback for the audit-A3 pattern: a bold-colon header
531
+ with an empty inline value followed by a multi-line address laid out as
532
+ bullets (HUD voucher ``Mail Payments To`` blocks, etc.). Strict gating
533
+ keeps it from pulling unrelated form structure into the value:
534
+
535
+ 1. Every line must be a clean bullet (``-``/``*``/``+`` with no
536
+ ``[x]``/``[ ]`` checkbox marker — those rows belong to a different
537
+ field).
538
+ 2. No line may carry any checkbox glyph or ASCII bracket marker.
539
+ 3. At least 2 collected bullets are required. A single bullet is too
540
+ ambiguous to attribute as the value — leaving the value empty is
541
+ safer than risking a wrong attribution.
542
+ 4. Stops at blank line, ATX heading, HTML boundary, or another bold-
543
+ colon header. Returns ``""`` if any constraint fails so the caller
544
+ falls back to the normal empty-value path.
545
+ """
546
+
547
+ tail = content[after_offset:]
548
+ lines = tail.splitlines()
549
+ # Skip the line containing the header itself (we matched into it).
550
+ start_idx = 1 if lines else 0
551
+ collected: list[str] = []
552
+ for raw in lines[start_idx : start_idx + max_lines]:
553
+ stripped = raw.strip()
554
+ if not stripped:
555
+ break
556
+ if stripped.startswith(("#", ">", "|", "<")):
557
+ break
558
+ # Stop at the start of a new bold-colon header.
559
+ if "**" in stripped and ":" in stripped:
560
+ break
561
+ if not _AGGREGATE_BULLET_RE.match(stripped):
562
+ return ""
563
+ if _MARKER_RE.search(stripped):
564
+ return ""
565
+ cleaned = re.sub(r"^\\?[-*+]\s+", "", stripped).strip()
566
+ cleaned = _strip_html_tags(cleaned).strip()
567
+ if cleaned:
568
+ collected.append(cleaned)
569
+ if len(collected) < 2:
570
+ return ""
571
+ return ", ".join(collected)
572
+
573
+
574
+ # Bold-colon pattern. Matches **Label:** value and **Label**: value, allowing
575
+ # multiple pairs on a single line. All inter-token whitespace is restricted
576
+ # to horizontal whitespace ([ \t]) so a match cannot span blank lines or
577
+ # headings — without this, an empty "**Label**:\n\n# Heading\n\n**Other**:"
578
+ # would attribute the heading text to Label as the value.
579
+ _BOLD_COLON_RE = re.compile(
580
+ r"\*\*[ \t]*([^*\n]+?)[ \t]*\*\*[ \t]*:?[ \t]*([^\n*]*?)(?=[ \t]*\*\*|$)",
581
+ re.MULTILINE,
582
+ )
583
+ _EXPLICIT_BOLD_COLON_RE = re.compile(
584
+ r"\*\*[ \t]*([^*\n]+?)[ \t]*(?:[ \t]*:[ \t]*\*\*[ \t]*|\*\*[ \t]*:[ \t]*)([^\n*]*?)(?=[ \t]*\*\*|$)",
585
+ re.MULTILINE,
586
+ )
587
+ _ANY_BOLD_COLON_ON_LINE_RE = re.compile(r"\*\*[ \t]*[^*\n]+?[ \t]*\*\*[ \t]*:")
588
+ _ANY_BOLD_INTERNAL_COLON_ON_LINE_RE = re.compile(r"\*\*[ \t]*[^*\n]+?:[ \t]*\*\*")
589
+ _BOLD_VALUE_RE = re.compile(r"\*\*[ \t]*([^*\n]+?)[ \t]*\*\*")
590
+ _LIST_LINE_RE = re.compile(r"^\s*\\?[-*+]\s+")
591
+
592
+
593
+ def _has_bold_colon_label(content: str) -> bool:
594
+ """Return true if any line contains an explicit bold label marker."""
595
+
596
+ return bool(_ANY_BOLD_COLON_ON_LINE_RE.search(content) or _ANY_BOLD_INTERNAL_COLON_ON_LINE_RE.search(content))
597
+
598
+
599
+ # Connector words that the parser sometimes wraps in bold inside a numeric
600
+ # range, e.g. ``**Depth Drilled**: 105 **to**: 15437`` or
601
+ # ``Temperature: 32 **to** 100 F``. Without special-casing, the bold-colon
602
+ # value regex stops at the connector's leading ``**`` and only captures the
603
+ # left half. We re-join the trailing value when the bold span between two
604
+ # value chunks is one of these connectors. The connectors are matched whole-
605
+ # word, case-insensitively. Allows leading horizontal whitespace so the
606
+ # splice cursor doesn't have to land exactly on the ``**``.
607
+ _BOLD_CONNECTOR_RE = re.compile(
608
+ r"[ \t]*\*\*[ \t]*(to|and|or|&|thru|through|until)[ \t]*\*\*[ \t]*:?[ \t]*([^\n*]*?)"
609
+ r"(?=[ \t]*\*\*|$)",
610
+ re.IGNORECASE | re.MULTILINE,
611
+ )
612
+
613
+
614
+ def _extend_value_across_bold_connectors(content: str, value_end_offset: int, base_value: str) -> str:
615
+ """Re-join a bold-colon value that was clipped at a bold connector token.
616
+
617
+ The bold-colon regex terminates the value at the next ``**``. When the
618
+ next bold span is a connector word (``to``, ``and``, ...), the value
619
+ actually continues across it. This helper looks at the content
620
+ immediately following the captured value and, while it sees a bold
621
+ connector followed by more inline content, splices everything into a
622
+ single value string.
623
+
624
+ Stops as soon as the next bold span is anything other than a recognized
625
+ connector — that's a real label boundary, not a continuation.
626
+ """
627
+
628
+ if not base_value:
629
+ return base_value
630
+ cursor = value_end_offset
631
+ joined = base_value
632
+ while True:
633
+ match = _BOLD_CONNECTOR_RE.match(content, cursor)
634
+ if not match:
635
+ break
636
+ connector = match.group(1)
637
+ extra = match.group(2).strip()
638
+ joined = f"{joined} {connector} {extra}".strip()
639
+ cursor = match.end()
640
+ return joined
641
+
642
+
643
+ def _iter_bold_colon_pairs_from_regex(content: str, pattern: re.Pattern[str]) -> list[tuple[str, str]]:
644
+ """Yield every (label, value) pair surfaced via bold-colon syntax.
645
+
646
+ Generic post-processing applied to every yielded pair: HTML tags
647
+ stripped from both label and value, leading ``[tag]`` prefix removed
648
+ from the label, ``| Next Label:`` boundary trimmed from the value, and
649
+ when the inline value is empty, the next few non-blank list/text lines
650
+ are aggregated into the value (multi-line address pattern).
651
+ """
652
+
653
+ out: list[tuple[str, str]] = []
654
+ for match in pattern.finditer(content):
655
+ cand_label = match.group(1).strip(": ").strip()
656
+ # Strip trailing markdown line-continuation backslash before whitespace.
657
+ # Some parsers emit ``**Label**: \`` for empty fields; without this
658
+ # strip the value would be ``"\\"``, never matching empty expected.
659
+ raw_value = match.group(2).strip().rstrip("\\").strip()
660
+ # Splice bold connectors (``**to**``, ``**and**``) back into the value
661
+ # so numeric ranges like ``**Depth Drilled**: 105 **to** 15437`` aren't
662
+ # truncated at the connector.
663
+ raw_value = _extend_value_across_bold_connectors(content, match.end(), raw_value)
664
+ cand_label = _strip_html_tags(cand_label).strip()
665
+ cand_label = _LABEL_TAG_PREFIX_RE.sub("", cand_label).strip()
666
+ cand_value = _strip_html_tags(raw_value).strip()
667
+ cand_value = _trim_value_at_next_field(cand_value)
668
+ if not cand_value:
669
+ cand_value = _aggregate_following_lines(content, match.end())
670
+ if cand_label:
671
+ out.append((cand_label, cand_value))
672
+ # Recover any sibling pipe-concatenated pairs riding the same line.
673
+ out.extend(_split_pipe_concatenated_pairs(raw_value))
674
+ return out
675
+
676
+
677
+ def _iter_bold_colon_pairs(content: str) -> list[tuple[str, str]]:
678
+ """Yield every (label, value) pair surfaced via bold-colon syntax.
679
+
680
+ Kept intentionally backward-compatible with older generated rules: this
681
+ accepts both ``**Label:** value`` / ``**Label**: value`` and the historical
682
+ no-colon form. New rule generation should use
683
+ :func:`_iter_explicit_bold_colon_pairs` so bold emphasis on values is not
684
+ mistaken for a label.
685
+ """
686
+
687
+ return _iter_bold_colon_pairs_from_regex(content, _BOLD_COLON_RE)
688
+
689
+
690
+ def _iter_explicit_bold_colon_pairs(content: str) -> list[tuple[str, str]]:
691
+ """Yield only explicit ``**Label:** value`` / ``**Label**: value`` pairs."""
692
+
693
+ return _iter_bold_colon_pairs_from_regex(content, _EXPLICIT_BOLD_COLON_RE)
694
+
695
+
696
+ # Plain-colon pattern. Inter-token whitespace is restricted to horizontal
697
+ # whitespace ([ \t]) so a colon at end-of-line cannot consume the next line as
698
+ # the value (parallel to the bold-colon regex; same blank-line crossing bug).
699
+ _PLAIN_COLON_RE = re.compile(r"^[ \t]*([^:\n*][^:\n]{0,200}?)[ \t]*:[ \t]*(.+?)[ \t]*$", re.MULTILINE)
700
+ _LIST_MARKER_RE = re.compile(r"^\\?[-*+]\s+")
701
+
702
+ # Underscore blank field: ``Processor's Name _________________``. The label
703
+ # sits before a run of three or more underscores acting as a fill-in line for
704
+ # an empty field. No colon, no bold, just a label-then-underscore-blank.
705
+ _UNDERSCORE_BLANK_RE = re.compile(r"^\s*([^_\n]+?)\s+_{3,}\s*$", re.MULTILINE)
706
+
707
+
708
+ def _iter_plain_colon_pairs(content: str) -> list[tuple[str, str]]:
709
+ """Yield (label, value) pairs from `Label: value` lines (plain text).
710
+
711
+ Plain bullet items with the ``Label: value`` shape (``- Defendant: Devon``)
712
+ are stripped of their leading marker and yielded — markdown task lists
713
+ (``- [x] Foo``) are still skipped because they are handled by the checkbox
714
+ scanners. Headings, fenced code, blockquotes, and bold-formatted lines
715
+ are skipped here too.
716
+
717
+ Generic post-processing on every yielded pair: HTML tags stripped from
718
+ both label and value, leading ``[tag]`` prefix removed from the label,
719
+ and the value trimmed at any ``| Next Label:`` boundary so a single line
720
+ like ``A: x | B: y`` yields two pairs instead of one with a run-on
721
+ value.
722
+ """
723
+
724
+ out: list[tuple[str, str]] = []
725
+ for match in _PLAIN_COLON_RE.finditer(content):
726
+ cand_label = match.group(1).strip()
727
+ raw_value = match.group(2).strip()
728
+ # Skip multi-cell HTML table rows: a single line that opens more than
729
+ # one ``<th>`` / ``<td>`` is a wide table row, not a single
730
+ # ``label: value`` line. Without this guard
731
+ # ``<tr><th>API NO:</th><th>WELL</th><th>LEHMAN #1</th></tr>`` matches
732
+ # the plain-colon regex and yields ``("API NO", "WELLLEHMAN #1")``
733
+ # because HTML-tag stripping collapses adjacent cells into a single
734
+ # value run. Single-cell rows (``<th>Company: CIMARRON ...</th>``)
735
+ # carry exactly one inline KV pair and stay on this path — wide HTML
736
+ # form tables are handled by ``_iter_html_table_kv_rows`` and
737
+ # ``_iter_html_cell_neighbor_pairs``.
738
+ raw_line = match.group(0)
739
+ if len(re.findall(r"<t[hd]\b", raw_line)) > 1:
740
+ continue
741
+ if cand_label.startswith(("#", "`", ">")):
742
+ continue
743
+ if "**" in cand_label:
744
+ continue
745
+ if cand_label.startswith(("\\-", "-", "*", "+")):
746
+ stripped = _LIST_MARKER_RE.sub("", cand_label).strip()
747
+ # Tasklist-shaped bullets (``[x] ...``) belong to the checkbox path.
748
+ if stripped.startswith(("\\[", "[")):
749
+ continue
750
+ if not stripped:
751
+ continue
752
+ cand_label = stripped
753
+ cand_label = _strip_html_tags(cand_label).strip()
754
+ cand_label = _LABEL_TAG_PREFIX_RE.sub("", cand_label).strip()
755
+ cand_value = _strip_html_tags(raw_value).strip()
756
+ cand_value = _trim_value_at_next_field(cand_value)
757
+ if cand_label:
758
+ out.append((cand_label, cand_value))
759
+ # Recover any sibling pipe-concatenated pairs riding the same line.
760
+ out.extend(_split_pipe_concatenated_pairs(raw_value))
761
+ return out
762
+
763
+
764
+ _LABEL_BEFORE_BOLD_KNOWN_SUFFIXES = (
765
+ "Name of Field in which well is located",
766
+ "Date well was plugged",
767
+ "Name of Company or Operator",
768
+ "Name of Farm or Lease",
769
+ "Name of Party Plugging Well",
770
+ "Name of Lease",
771
+ "No. of Acres",
772
+ "Well No.",
773
+ "Sec. No.",
774
+ "Blk No.",
775
+ "Company",
776
+ "Address",
777
+ "Survey",
778
+ "County",
779
+ "Has this well ever produced oil or gas?",
780
+ "Located",
781
+ "Oil",
782
+ "Gas",
783
+ "Dry",
784
+ "Total Depth",
785
+ "Top of each producing sand",
786
+ "Name",
787
+ "Title",
788
+ )
789
+ _LABEL_BEFORE_BOLD_REJECT_PREFIXES = (
790
+ "i,",
791
+ "i ",
792
+ "if you answered",
793
+ "subscribed ",
794
+ "being ",
795
+ )
796
+ _LABEL_BEFORE_BOLD_REJECT_SUFFIXES = (
797
+ " and",
798
+ " at",
799
+ " by",
800
+ " for",
801
+ " from",
802
+ " in",
803
+ " of",
804
+ " on",
805
+ " or",
806
+ " to",
807
+ )
808
+
809
+
810
+ def _clean_label_before_bold(raw: str) -> str:
811
+ """Normalize the label chunk immediately before a bold value span."""
812
+
813
+ label = raw
814
+ # Empty fill lines often sit between two fields:
815
+ # ``Blk No. ______ Survey **T.E.&L.**``. The label for the bold value is the
816
+ # suffix after the blank, not the earlier empty field.
817
+ label = re.split(r"(?:\\?_){3,}", label)[-1]
818
+ label = re.sub(r"<br\s*/?>", " ", label, flags=re.IGNORECASE)
819
+ label = re.sub(r"^[\s\\*_#>\-+|]+", "", label)
820
+ label = _LABEL_TAG_PREFIX_RE.sub("", label)
821
+ label = _strip_html_tags(label).strip()
822
+ label = label.strip("*_ \t:-,;")
823
+ if ":" in label:
824
+ label = label.rsplit(":", 1)[-1].strip()
825
+
826
+ lowered = label.lower()
827
+ best: str | None = None
828
+ for suffix in _LABEL_BEFORE_BOLD_KNOWN_SUFFIXES:
829
+ idx = lowered.rfind(suffix.lower())
830
+ if idx < 0:
831
+ continue
832
+ if lowered[idx:].strip() != suffix.lower():
833
+ continue
834
+ if best is None or len(suffix) > len(best):
835
+ best = suffix
836
+ if best is not None:
837
+ label = label[-len(best) :].strip()
838
+
839
+ return label.strip("*_ \t:-,;")
840
+
841
+
842
+ def _looks_like_label_before_bold(label: str) -> bool:
843
+ if not label or len(label) > 100:
844
+ return False
845
+ lowered = label.lower()
846
+ if lowered.startswith(_LABEL_BEFORE_BOLD_REJECT_PREFIXES):
847
+ return False
848
+ if not any(ch.isalpha() for ch in label):
849
+ return False
850
+ if ")" in label and "(" not in label:
851
+ return False
852
+ if len(label) <= 2:
853
+ return False
854
+ if lowered.endswith(_LABEL_BEFORE_BOLD_REJECT_SUFFIXES):
855
+ return False
856
+ if re.fullmatch(r"(?:19|20)?\d{1,2}", label):
857
+ return False
858
+ first_alpha = next((ch for ch in label if ch.isalpha()), "")
859
+ if first_alpha and first_alpha.islower():
860
+ return False
861
+ return True
862
+
863
+
864
+ def _extend_inline_year_suffix(
865
+ value: str,
866
+ between_this_and_next: str,
867
+ next_match: re.Match[str] | None,
868
+ ) -> str:
869
+ """Join ``**June 17,** 194 **3**`` into ``June 17, 1943``."""
870
+
871
+ if next_match is None:
872
+ return value
873
+ year_prefix = re.sub(r"\s+", "", between_this_and_next)
874
+ if not re.fullmatch(r"(?:19|20)?\d{0,2}", year_prefix):
875
+ return value
876
+ suffix = next_match.group(1).strip()
877
+ if not re.fullmatch(r"\d{1,2}", suffix):
878
+ return value
879
+ year = f"{year_prefix}{suffix}"
880
+ if not re.fullmatch(r"(?:19|20)\d{2}", year):
881
+ return value
882
+ return re.sub(r"\s+", " ", f"{value} {year}").strip()
883
+
884
+
885
+ def _iter_label_before_bold_value_pairs(content: str) -> list[tuple[str, str]]:
886
+ """Yield ``(label, value)`` for visual rows shaped as ``Label **Value**``.
887
+
888
+ Gemini-style parse output often preserves the printed form row as
889
+ ``Well No. **Z-5** Name of Lease **W. H. Portwood**`` rather than normalizing
890
+ it to ``**Well No.**: Z-5``. This matcher treats the text immediately before
891
+ each bold span as the label and the bold span as the value. It is deliberately
892
+ lower priority than explicit colon/table sources.
893
+ """
894
+
895
+ out: list[tuple[str, str]] = []
896
+ for line in content.splitlines():
897
+ if "**" not in line:
898
+ continue
899
+ stripped = line.strip()
900
+ if not stripped or stripped.startswith(("#", ">", "|", "[Figure")):
901
+ continue
902
+ if re.search(r"\binstructions?\b", stripped, flags=re.IGNORECASE):
903
+ continue
904
+ if _LIST_LINE_RE.match(line):
905
+ continue
906
+ if _has_bold_colon_label(line):
907
+ continue
908
+ matches = list(_BOLD_VALUE_RE.finditer(line))
909
+ if not matches:
910
+ continue
911
+ prev_end = 0
912
+ for idx, match in enumerate(matches):
913
+ label = _clean_label_before_bold(line[prev_end : match.start()])
914
+ value = _strip_html_tags(match.group(1)).strip()
915
+ next_match = matches[idx + 1] if idx + 1 < len(matches) else None
916
+ next_start = next_match.start() if next_match is not None else len(line)
917
+ value = _extend_inline_year_suffix(value, line[match.end() : next_start], next_match)
918
+ if _looks_like_label_before_bold(label) and value:
919
+ out.append((label, value))
920
+ prev_end = match.end()
921
+ return out
922
+
923
+
924
+ # Underline fill-in pattern: parsers that preserve the form's "fill in the
925
+ # blank" layout emit the filled value wrapped in ``<u>...</u>`` tags inline
926
+ # in the surrounding prose, e.g.
927
+ #
928
+ # **2. PROPERTY:** Lot <u>12</u>, Block <u>C</u>, City of <u>Austin</u>...
929
+ #
930
+ # The label sits immediately before the underline span, terminated by a
931
+ # punctuation/whitespace boundary on its left side. We yield (label, value)
932
+ # for each such span so the standard ``_label_matches`` fuzzy-matcher can
933
+ # bridge GT labels like "Block" or "City of (Street Address and City)".
934
+ _UNDERLINE_FILL_RE = re.compile(r"<u>([^<\n]+)</u>")
935
+ _LABEL_LEFT_TERMINATORS = ".,;:()\n>"
936
+
937
+
938
+ def _iter_underline_fill_pairs(content: str) -> list[tuple[str, str]]:
939
+ """Yield (preceding_label, underlined_value) pairs from ``<u>...</u>`` runs."""
940
+
941
+ out: list[tuple[str, str]] = []
942
+ for match in _UNDERLINE_FILL_RE.finditer(content):
943
+ value = match.group(1).strip()
944
+ if not value:
945
+ continue
946
+ before = content[max(0, match.start() - 100) : match.start()]
947
+ # Walk backward to the nearest sentence/clause terminator. Anything
948
+ # left of that terminator belongs to a different label (or to a
949
+ # heading/inline header), so we stop there.
950
+ cut = -1
951
+ for ch in _LABEL_LEFT_TERMINATORS:
952
+ cut = max(cut, before.rfind(ch))
953
+ label_chunk = before[cut + 1 :]
954
+ # Strip markdown noise: leading bullet, bold/italic markers, stray
955
+ # backslashes, and trailing whitespace. The bracketed-tag prefix
956
+ # (``[FORM FIELD] ``) is dropped here too so it never bleeds into
957
+ # candidate labels.
958
+ label_chunk = re.sub(r"^[\s\\*_#>\-]+", "", label_chunk)
959
+ label_chunk = _LABEL_TAG_PREFIX_RE.sub("", label_chunk)
960
+ label_chunk = _strip_html_tags(label_chunk).strip()
961
+ label_chunk = label_chunk.strip("*_ \t").strip()
962
+ if not label_chunk:
963
+ continue
964
+ # Only the trailing 1-6 words can plausibly be the label — the rest
965
+ # is sentence context.
966
+ words = label_chunk.split()
967
+ if not words:
968
+ continue
969
+ label = " ".join(words[-6:])
970
+ if label:
971
+ out.append((label, value))
972
+ return out
973
+
974
+
975
+ def _iter_underscore_blank_pairs(content: str) -> list[tuple[str, str]]:
976
+ """Yield (label, "") pairs for ``Label ____`` underscore-blank fields."""
977
+
978
+ out: list[tuple[str, str]] = []
979
+ for match in _UNDERSCORE_BLANK_RE.finditer(content):
980
+ cand_label = match.group(1).strip()
981
+ if not cand_label:
982
+ continue
983
+ if cand_label.startswith(("#", "-", "*", "`", ">", "|")):
984
+ continue
985
+ if "**" in cand_label or ":" in cand_label:
986
+ continue
987
+ out.append((cand_label, ""))
988
+ return out
989
+
990
+
991
+ # Italic line shape: ``*Plaintiff*`` or ``_Address_`` (optionally with a
992
+ # trailing space + ``)`` from court-form layouts like ``*Plaintiff* )``).
993
+ _ITALIC_LABEL_LINE_RE = re.compile(r"^\s*([*_])\s*(\S.*?\S)\s*\1[\s)\\]*$")
994
+
995
+
996
+ def _find_text_value_adjacent_line(
997
+ content: str,
998
+ label: str,
999
+ label_max_diffs: int | float = 0,
1000
+ ) -> tuple[bool, str | None]:
1001
+ """Fallback for label-on-its-own-line layouts adjacent to an unlabelled value.
1002
+
1003
+ Two layouts share this scanner:
1004
+
1005
+ - **Italic caption below value** (federal court forms — AO398):
1006
+ ``Anthony Cole Jackson )\\n*Plaintiff* )``. The label sits italicized on
1007
+ the line below the value.
1008
+ - **Numbered/heading-style label above value** (UCC5, gemini sub-sections):
1009
+ ``1a. INITIAL FINANCING STATEMENT FILE NUMBER\\nOR-UCC-2025-00532600``.
1010
+ The label is its own line above the value.
1011
+
1012
+ Conservative heuristic: only fires for short label-shaped lines (≤ 80
1013
+ chars after stripping markers, no ``:`` and no ``**``) and only on a
1014
+ *strict* ratio match (≥ ``CELL_FUZZY_MATCH_THRESHOLD``). This means a
1015
+ long paragraph that *contains* the label as a substring is **not**
1016
+ treated as the label line — partial-ratio matching is reserved for the
1017
+ other (label-then-value) scanners.
1018
+
1019
+ Direction: italic line → look ABOVE first (caption convention); plain
1020
+ line → look BELOW first (label-then-value convention). Whichever
1021
+ direction lands a non-blank line wins.
1022
+ """
1023
+
1024
+ lines = content.splitlines()
1025
+ lbl_norm = normalize_text(label)
1026
+ if not lbl_norm:
1027
+ return False, None
1028
+ for i, raw in enumerate(lines):
1029
+ stripped = raw.strip()
1030
+ if not stripped or len(stripped) > 100:
1031
+ continue
1032
+ if stripped.startswith(("#", ">", "|", "<", "`")):
1033
+ continue
1034
+ if "**" in stripped or ":" in stripped:
1035
+ continue
1036
+ italic_match = _ITALIC_LABEL_LINE_RE.match(raw)
1037
+ if italic_match:
1038
+ cleaned = italic_match.group(2).strip()
1039
+ else:
1040
+ cleaned = re.sub(r"\s*[)\\]+\s*$", "", stripped)
1041
+ cleaned = re.sub(r"^\\?[-*+]\s+", "", cleaned).strip()
1042
+ cleaned = cleaned.strip("*_ \t").strip()
1043
+ if not cleaned or len(cleaned) > 80:
1044
+ continue
1045
+ cand_norm = normalize_text(cleaned)
1046
+ if not cand_norm:
1047
+ continue
1048
+ strict_match = fuzz.ratio(cand_norm, lbl_norm) / 100.0 >= CELL_FUZZY_MATCH_THRESHOLD
1049
+ tolerant_match = label_max_diffs > 0 and _label_match_score(cleaned, label, label_max_diffs) > 0.0
1050
+ if not strict_match and not tolerant_match:
1051
+ continue
1052
+ if italic_match:
1053
+ search_orders = [
1054
+ range(i - 1, max(i - 4, -1), -1),
1055
+ range(i + 1, min(i + 4, len(lines))),
1056
+ ]
1057
+ else:
1058
+ search_orders = [
1059
+ range(i + 1, min(i + 4, len(lines))),
1060
+ range(i - 1, max(i - 4, -1), -1),
1061
+ ]
1062
+ for order in search_orders:
1063
+ for j in order:
1064
+ cand_line = lines[j].strip()
1065
+ if not cand_line:
1066
+ continue
1067
+ if cand_line.startswith(("#", "|", ">")):
1068
+ break
1069
+ if "**" in cand_line and ":" in cand_line:
1070
+ break
1071
+ value = re.sub(r"^\\?[-*+]\s+", "", cand_line).strip()
1072
+ value = re.sub(r"\s*[)\\]+\s*$", "", value).strip()
1073
+ value = value.strip("*_ \t").strip()
1074
+ if value:
1075
+ return True, value
1076
+ return True, ""
1077
+ return False, None
1078
+
1079
+
1080
+ def _build_cell_text_counts(data, rows: int, cols: int) -> dict[str, int]: # type: ignore[no-untyped-def]
1081
+ """Per-table map of text → number of distinct *origin* cells.
1082
+
1083
+ ``parse_html_tables`` expands ``colspan``/``rowspan`` by duplicating cell
1084
+ text across every covered grid position, so a single ``<th
1085
+ colspan="4">KEBO</th>`` looks like four ``"KEBO"`` cells in the expanded
1086
+ grid. Counting raw grid cells would mis-classify any spanned value as a
1087
+ repeated label. Dedupe by skipping cells whose text equals the left or
1088
+ above neighbor — those are colspan / rowspan runs of the same origin.
1089
+ """
1090
+
1091
+ counts: dict[str, int] = {}
1092
+ for r in range(rows):
1093
+ for c in range(cols):
1094
+ t = str(data[r, c]).strip()
1095
+ if not t:
1096
+ continue
1097
+ if c > 0 and str(data[r, c - 1]).strip() == t:
1098
+ continue
1099
+ if r > 0 and str(data[r - 1, c]).strip() == t:
1100
+ continue
1101
+ counts[t] = counts.get(t, 0) + 1
1102
+ return counts
1103
+
1104
+
1105
+ def _neighbor_is_label_like(neighbor: str, text_counts: dict[str, int]) -> bool:
1106
+ if _cell_text_is_label_like(neighbor):
1107
+ return True
1108
+ # Short text that repeats elsewhere in the same table → structural label.
1109
+ if len(neighbor) <= 30 and text_counts.get(neighbor, 0) >= 2:
1110
+ return True
1111
+ return False
1112
+
1113
+
1114
+ def _is_value_shaped_cell(neighbor: str | None) -> bool:
1115
+ if not neighbor:
1116
+ return False
1117
+ s = neighbor.strip()
1118
+ if len(s) < 2:
1119
+ return False
1120
+ # Lone checkbox glyphs aren't useful values for text rules.
1121
+ if _GLYPH_RE.search(s) and len(s) <= 2:
1122
+ return False
1123
+ return True
1124
+
1125
+
1126
+ def _iter_html_cell_neighbor_pairs(content: str) -> list[tuple[str, str]]:
1127
+ """Yield ``(cell_text, neighbor_value)`` pairs for wide (>2 col) HTML
1128
+ tables, intended as a low-priority fallback source for
1129
+ ``_find_text_value_for_label``.
1130
+
1131
+ Targets form-style layouts where labels and values are spatially
1132
+ interleaved inside a single wide ``<table>`` rather than separated into
1133
+ a clean header row + data rows, e.g. well-log report headers::
1134
+
1135
+ <tr><th colspan="2">FILE NO:</th>
1136
+ <th colspan="2">COMPANY</th>
1137
+ <th colspan="4">KEBO OIL &amp; GAS, INC.</th></tr>
1138
+ <tr><th colspan="2">API NO:</th>
1139
+ <th colspan="2">WELL</th>
1140
+ <th colspan="4">LEHMAN #1</th></tr>
1141
+ <tr><th colspan="2">42-157-33282</th>
1142
+ <th colspan="2">FIELD</th>
1143
+ <th colspan="4">NEEDVILLE</th></tr>
1144
+
1145
+ For each non-empty cell in the expanded grid the iterator looks at two
1146
+ candidate neighbors:
1147
+
1148
+ * the first non-empty cell to the right in the same row, skipping
1149
+ colspan duplicates (cell text equal to the cell itself);
1150
+ * the first non-empty cell below in the same column, similarly skipping
1151
+ rowspan duplicates.
1152
+
1153
+ The chosen neighbor is the first one that is *value-shaped* (length ≥ 2,
1154
+ not a lone checkbox glyph) and *not label-shaped* per
1155
+ ``_cell_text_is_label_like`` or structural repetition in the same table.
1156
+ Right is preferred over below (matches left-to-right reading).
1157
+
1158
+ Cells with no value-shaped neighbor still emit ``(cell_text, "")`` so the
1159
+ caller's ``_collect`` records ``label_seen=True`` for empty-expected
1160
+ rules — same contract as the other pair sources.
1161
+
1162
+ The caller scores ``cell_text`` against the rule label via
1163
+ ``_label_match_score`` and picks the best candidate. We don't filter by
1164
+ label here so the caller can resolve adjacent-label collisions (the
1165
+ same way #978 made other sources do).
1166
+ """
1167
+
1168
+ if "<table" not in content.lower():
1169
+ return []
1170
+
1171
+ out: list[tuple[str, str]] = []
1172
+
1173
+ for table in parse_html_tables(content):
1174
+ rows, cols = table.data.shape
1175
+ if cols <= 2 or rows == 0:
1176
+ continue
1177
+
1178
+ # Per-table text-count map — a short text that exactly repeats in
1179
+ # ≥2 *distinct origin* cells (after collapsing colspan/rowspan runs
1180
+ # via ``_build_cell_text_counts``) is structurally likely to be a
1181
+ # column label / section header (e.g. ``KB`` / ``DF`` / ``GL`` rows
1182
+ # in well-log elevation blocks). Used as a tie-breaker for which
1183
+ # neighbor cell is value-shaped.
1184
+ text_counts = _build_cell_text_counts(table.data, rows, cols)
1185
+
1186
+ for r in range(rows):
1187
+ for c in range(cols):
1188
+ cell = str(table.data[r, c]).strip()
1189
+ if not cell:
1190
+ continue
1191
+
1192
+ # Right scan: first non-empty cell to the right that is not
1193
+ # a colspan duplicate (text != label cell text).
1194
+ right_val: str | None = None
1195
+ for cc in range(c + 1, cols):
1196
+ nxt = str(table.data[r, cc]).strip()
1197
+ if nxt and nxt != cell:
1198
+ right_val = nxt
1199
+ break
1200
+
1201
+ # Below scan: first non-empty cell directly below that is
1202
+ # not a rowspan duplicate.
1203
+ below_val: str | None = None
1204
+ for rr in range(r + 1, rows):
1205
+ nxt = str(table.data[rr, c]).strip()
1206
+ if nxt and nxt != cell:
1207
+ below_val = nxt
1208
+ break
1209
+
1210
+ # Score each candidate. Want a value-shaped neighbor that
1211
+ # does *not* itself look label-like. Right is preferred over
1212
+ # below when both qualify (matches left-to-right reading).
1213
+ right_ok = _is_value_shaped_cell(right_val) and not _neighbor_is_label_like(
1214
+ right_val or "", text_counts
1215
+ )
1216
+ below_ok = _is_value_shaped_cell(below_val) and not _neighbor_is_label_like(
1217
+ below_val or "", text_counts
1218
+ )
1219
+
1220
+ if right_ok:
1221
+ out.append((cell, right_val or ""))
1222
+ elif below_ok:
1223
+ out.append((cell, below_val or ""))
1224
+ else:
1225
+ # No value-shaped neighbor at this position. Still emit
1226
+ # an empty-value pair so a matching label sets the
1227
+ # caller's ``label_seen`` flag (mirrors the other pair
1228
+ # iterators that surface ``""`` for label-only hits).
1229
+ out.append((cell, ""))
1230
+
1231
+ return out
1232
+
1233
+
1234
+ def _find_text_value_for_label(
1235
+ content: str,
1236
+ label: str,
1237
+ expected_values: list[str] | None = None,
1238
+ max_diffs: int | float = 0,
1239
+ label_max_diffs: int | float = 0,
1240
+ ) -> tuple[bool, str | None]:
1241
+ """Look up the value for *label*. Returns (label_found, value).
1242
+
1243
+ The boolean tracks whether the label was located **at all** — useful for
1244
+ distinguishing "label missing from content" from "label present but value
1245
+ blank" (signature evaluation depends on this distinction). When the label
1246
+ is found only with empty values, returns ``(True, "")`` so callers can
1247
+ decide what to do (text rules with empty expected values pass; signature
1248
+ rules treat it as unsigned).
1249
+
1250
+ Matching strategy is **best-score across all sources**: every candidate
1251
+ KV pair from every source iterator is scored against the target label
1252
+ via :func:`_label_match_score`, and the highest-scoring non-empty value
1253
+ wins. Tie-breaks fall back to source priority (bold-colon > plain-colon
1254
+ > md-table > html-table > underline-fill) and then document order. This
1255
+ eliminates the classic ``Country`` / ``County`` adjacent-label
1256
+ collision: an exact ``Country`` hit (score 1.0) always wins over a
1257
+ fuzzy ``County`` hit (score ~0.86–0.95) no matter which comes first.
1258
+
1259
+ Multi-occurrence disambiguation via ``expected_values``
1260
+ -------------------------------------------------------
1261
+ A label text can legitimately appear multiple times at the **same**
1262
+ best score: ``KB`` / ``DF`` / ``GL`` are exact-match labels in well-
1263
+ log elevation blocks while also appearing as values of
1264
+ ``LOG MEASURED FROM`` / ``DRILL. MEAS. FROM`` (where the cell-
1265
+ neighbor matcher surfaces them with score 1.0). Without
1266
+ disambiguation, source-priority + doc-order tie-breaks would lock
1267
+ onto an arbitrary occurrence — which one happens to come first
1268
+ has no relation to which page occurrence the GT refers to.
1269
+
1270
+ When ``expected_values`` is supplied, the matcher applies the rule's
1271
+ expected value(s) as an oracle **among candidates at the top score
1272
+ level only**. That is: it picks the highest score; collects every
1273
+ candidate at that score; and returns the first one whose value
1274
+ matches any expected via :func:`_values_match_text`. If no
1275
+ top-score candidate matches, the legacy source-priority / doc-order
1276
+ tie-break fires — same as without ``expected_values``.
1277
+
1278
+ The "top score only" gate is what keeps the GT oracle from leaking
1279
+ across adjacent labels: ``Country`` (score 1.0) vs ``County``
1280
+ (partial 0.95) live at *different* score levels, so even if a
1281
+ ``County`` row's value coincidentally equals the GT's expected
1282
+ ``Country`` value, ``County`` is not eligible. Only when two
1283
+ candidates are equally good *label matches* does the value oracle
1284
+ intervene.
1285
+ """
1286
+
1287
+ label_seen = False
1288
+ # (negated_score, negated_priority, doc_order, value, source_name)
1289
+ # — we'll sort ascending so the *best* candidate (highest score, then
1290
+ # highest priority, then earliest doc order) sits at the top.
1291
+ candidates: list[tuple[float, int, int, str, str]] = []
1292
+
1293
+ def _collect(
1294
+ pairs: list[tuple[str, str]],
1295
+ priority: int,
1296
+ source_name: str,
1297
+ ) -> None:
1298
+ nonlocal label_seen
1299
+ for idx, (cand_label, cand_value) in enumerate(pairs):
1300
+ score = _label_match_score(cand_label, label, label_max_diffs)
1301
+ if score <= 0.0:
1302
+ continue
1303
+ label_seen = True
1304
+ if cand_value:
1305
+ candidates.append((-score, -priority, idx, cand_value, source_name))
1306
+
1307
+ # Higher priority numbers = more confident sources. The ordering matches
1308
+ # the original first-match-wins precedence so tie-breaks preserve legacy
1309
+ # behavior on documents where multiple sources produce equally strong
1310
+ # label matches.
1311
+ _collect(_iter_bold_colon_pairs(content), priority=4, source_name="bold_colon")
1312
+ _collect(_iter_plain_colon_pairs(content), priority=3, source_name="plain_colon")
1313
+ _collect(_iter_md_table_kv_rows(content), priority=2, source_name="md_table")
1314
+ _collect(_iter_html_table_kv_rows(content), priority=1, source_name="html_table")
1315
+ _collect(_iter_label_before_bold_value_pairs(content), priority=0, source_name="label_before_bold")
1316
+
1317
+ # Underscore blank fields (``Label ____``) — label seen, value empty.
1318
+ # The yielded value is always ""; we just record label presence so an
1319
+ # empty-expected text rule can pass via the ``label_seen`` short-circuit
1320
+ # below.
1321
+ for cand_label, _ in _iter_underscore_blank_pairs(content):
1322
+ if _label_matches(cand_label, label, label_max_diffs):
1323
+ label_seen = True
1324
+
1325
+ # Last-resort sources — only fire when no higher-confidence source
1326
+ # surfaced a non-empty value, so they never overwrite a strong-source
1327
+ # extraction. Both are gated on ``not candidates`` and added with
1328
+ # priorities below the strong sources; if both fire and both produce
1329
+ # candidates, ``priority`` breaks the tie in favor of underline_fill.
1330
+ if not candidates:
1331
+ # Underline fill-in (``Label <u>value</u>`` inline in prose).
1332
+ if "<u>" in content:
1333
+ _collect(
1334
+ _iter_underline_fill_pairs(content),
1335
+ priority=0,
1336
+ source_name="underline_fill",
1337
+ )
1338
+ # Wide-form HTML table cell-neighbor fallback. Targets layouts
1339
+ # where labels and values are spatially interleaved inside a single
1340
+ # wide ``<table>`` (well-log report headers, rotated form pages),
1341
+ # which neither the 2-col nor the multi-col header×data pair
1342
+ # iterator covers. Priority -1 keeps it strictly below
1343
+ # underline_fill on tie-breaks.
1344
+ _collect(
1345
+ _iter_html_cell_neighbor_pairs(content),
1346
+ priority=-1,
1347
+ source_name="html_cell_neighbor",
1348
+ )
1349
+
1350
+ if candidates:
1351
+ candidates.sort()
1352
+ # ``candidates`` is sorted ascending by (-score, -priority, doc_order,
1353
+ # ...), so the head is the best (label-match-score, source-priority,
1354
+ # doc-order) tuple. We use the rule's expected value as a tie-breaker
1355
+ # **only among candidates at the head score**, which keeps the GT
1356
+ # oracle from leaking across adjacent labels (Country score 1.0 vs
1357
+ # County score ~0.95 live at different levels, so County is never
1358
+ # eligible when Country is present).
1359
+ best_score_key = candidates[0][0]
1360
+ if expected_values:
1361
+ for neg_score, _prio, _doc, value, _src in candidates:
1362
+ if neg_score != best_score_key:
1363
+ break
1364
+ for exp in expected_values:
1365
+ if _values_match_text(value, exp, max_diffs):
1366
+ return True, value
1367
+ return True, candidates[0][3]
1368
+
1369
+ # Adjacent-line fallback (italic caption below value, or numbered label
1370
+ # above value). Only fires when no other matcher located the label.
1371
+ if not label_seen:
1372
+ adj_seen, adj_value = _find_text_value_adjacent_line(content, label, label_max_diffs)
1373
+ if adj_seen:
1374
+ return True, adj_value
1375
+
1376
+ if label_seen:
1377
+ return True, ""
1378
+ return False, None
1379
+
1380
+
1381
+ def _tokenize_checkbox_line(line: str) -> list[tuple[str, bool]]:
1382
+ """Pair every checkbox marker on *line* with its associated label.
1383
+
1384
+ Markers may be Unicode glyphs (``☐``/``☑``/``◉``/``○``/...) OR ASCII
1385
+ bracket pairs (``[x]``, ``\\[x\\]``, ``[ ]``). Handles both orderings:
1386
+
1387
+ - marker-first: ``☐ Single ☑ Married`` or ``\\[x] A \\[ ] B`` — each
1388
+ label sits between a marker and the next marker (or end of line).
1389
+ - label-first: ``Single ☐ Married ☑`` or ``Checking \\[x] Savings \\[ ]``
1390
+ — each label sits between the previous marker (or start) and the
1391
+ next marker.
1392
+
1393
+ Direction is decided by what comes before the first marker: if the line
1394
+ starts with the marker (after optional whitespace), use marker-first;
1395
+ otherwise use label-first. This covers the inline mid-line bracket
1396
+ pattern ``**Inaccuracy in financing statement** \\[ ]`` since the
1397
+ closing bracket is treated as a marker and the bold-label segment to
1398
+ its left becomes the label.
1399
+ """
1400
+
1401
+ marker_matches = list(_MARKER_RE.finditer(line))
1402
+ if not marker_matches:
1403
+ return []
1404
+
1405
+ text_before_first = line[: marker_matches[0].start()].strip()
1406
+ pairs: list[tuple[str, bool]] = []
1407
+
1408
+ if not text_before_first:
1409
+ # marker-first: label runs from marker end to next marker start (or EOL).
1410
+ for i, m in enumerate(marker_matches):
1411
+ label_start = m.end()
1412
+ label_end = marker_matches[i + 1].start() if i + 1 < len(marker_matches) else len(line)
1413
+ label_text = line[label_start:label_end].strip()
1414
+ if label_text:
1415
+ pairs.append((label_text, _marker_is_checked(m.group())))
1416
+ else:
1417
+ # label-first: label runs from previous marker end (or 0) to current marker.
1418
+ prev_end = 0
1419
+ for m in marker_matches:
1420
+ label_text = line[prev_end : m.start()].strip()
1421
+ prev_end = m.end()
1422
+ if label_text:
1423
+ pairs.append((label_text, _marker_is_checked(m.group())))
1424
+ return pairs
1425
+
1426
+
1427
+ _TRAILING_BOOL_OPTION_RE = re.compile(
1428
+ r"^(?P<context>.*?)(?P<option>\b(?:yes|no|y|n)\b)\s*$",
1429
+ re.IGNORECASE,
1430
+ )
1431
+
1432
+
1433
+ def _tokenize_checkbox_line_with_context(line: str) -> list[tuple[list[str], bool]]:
1434
+ """Pair inline yes/no checkbox markers with their question context.
1435
+
1436
+ Typical form parsers render a yes/no row as one line:
1437
+
1438
+ ``Multistage cement? Yes [ ] No [x]``
1439
+
1440
+ The plain tokenizer sees ``"Multistage cement? Yes" -> False`` and
1441
+ ``"No" -> True``. For a multi-label checkbox rule we want the more precise
1442
+ candidates ``["Multistage cement?", "Yes"] -> False`` and
1443
+ ``["Multistage cement?", "No"] -> True`` so repeated Yes/No options can be
1444
+ disambiguated by the surrounding question without adding a new schema.
1445
+ """
1446
+
1447
+ marker_matches = list(_MARKER_RE.finditer(line))
1448
+ if not marker_matches:
1449
+ return []
1450
+
1451
+ text_before_first = line[: marker_matches[0].start()].strip()
1452
+ if not text_before_first:
1453
+ return []
1454
+
1455
+ pairs: list[tuple[list[str], bool]] = []
1456
+ current_context: str | None = None
1457
+ prev_end = 0
1458
+ for marker in marker_matches:
1459
+ label_text = line[prev_end : marker.start()].strip()
1460
+ prev_end = marker.end()
1461
+ if not label_text:
1462
+ continue
1463
+
1464
+ labels = [label_text]
1465
+ option_match = _TRAILING_BOOL_OPTION_RE.match(label_text)
1466
+ if option_match:
1467
+ context = option_match.group("context").strip(" :;-")
1468
+ option = option_match.group("option").strip()
1469
+ if context:
1470
+ current_context = context
1471
+ labels = [context, option]
1472
+ elif current_context:
1473
+ labels = [current_context, option]
1474
+
1475
+ parts = _dedupe_nonempty_text(labels)
1476
+ if parts:
1477
+ pairs.append((parts, _marker_is_checked(marker.group())))
1478
+
1479
+ return pairs
1480
+
1481
+
1482
+ def _checkbox_label_list_match_score(
1483
+ candidate_labels: list[str],
1484
+ required_labels: list[str],
1485
+ label_max_diffs: list[int],
1486
+ ) -> float:
1487
+ if len(candidate_labels) < len(required_labels):
1488
+ return 0.0
1489
+
1490
+ used: set[int] = set()
1491
+ total = 0.0
1492
+ for req_idx, required in enumerate(required_labels):
1493
+ allowed = label_max_diffs[req_idx] if req_idx < len(label_max_diffs) else 0
1494
+ best_idx = -1
1495
+ best_score = 0.0
1496
+ for cand_idx, candidate in enumerate(candidate_labels):
1497
+ if cand_idx in used:
1498
+ continue
1499
+ score = _label_match_score(candidate, required, allowed)
1500
+ if score > best_score:
1501
+ best_idx = cand_idx
1502
+ best_score = score
1503
+ if best_idx < 0 or best_score <= 0.0:
1504
+ return 0.0
1505
+ used.add(best_idx)
1506
+ total += best_score
1507
+
1508
+ return total / max(len(required_labels), 1)
1509
+
1510
+
1511
+ # Markdown task-list. Allows optional ``\`` escapes around the list marker
1512
+ # AND the brackets — some parsers emit ``\[x\]`` (or even ``\- \[x\]`` for a
1513
+ # nested escaped bullet) so the markdown source survives literal-character
1514
+ # rendering. The marker accepts ``-``/``*``/``+`` and numbered-list ``\d+.`` —
1515
+ # USCIS citizenship attestations are rendered as ``1. [x] A citizen ...``.
1516
+ _LIST_MARKER_RE_INLINE = r"(?:[-*+]|\d+\.)"
1517
+ _MD_TASKLIST_RE = re.compile(
1518
+ rf"^\s*\\?{_LIST_MARKER_RE_INLINE}\s*\\?\[([ xX])\\?\]\s*(.+?)\s*$",
1519
+ re.MULTILINE,
1520
+ )
1521
+
1522
+ # Label-first bullet checkbox: ``* Checking \[x]`` or ``\- Savings [ ]``. The
1523
+ # label sits between the list marker and the bracket. Common in forms where
1524
+ # the parser surfaces the option label as the bullet text and the state as a
1525
+ # trailing widget marker. Both the bullet marker and the brackets may be
1526
+ # preceded by a literal backslash escape.
1527
+ _MD_BULLET_LABEL_FIRST_RE = re.compile(
1528
+ rf"^\s*\\?{_LIST_MARKER_RE_INLINE}\s+([^\[\n]+?)\s+\\?\[([ xX])\\?\]\s*$",
1529
+ re.MULTILINE,
1530
+ )
1531
+
1532
+ # Bullet-less task-list: a line that starts with ``\[x]`` / ``[ ]`` directly
1533
+ # with no leading bullet marker. ours_cost_effective and gemini render IRS
1534
+ # W-9 / USCIS / UCC5 checkboxes this way (``\[x] Individual/sole proprietor``
1535
+ # on its own line). The label group disallows ``[`` so a line with multiple
1536
+ # inline bracket markers (``\[ ] A \[x] B``) does NOT match here — those go
1537
+ # through the per-line tokenizer below where each bracket is paired with its
1538
+ # own label.
1539
+ _MD_BARE_TASKLIST_RE = re.compile(
1540
+ r"^\s*\\?\[([ xX])\\?\]\s*([^\[\n]+?)\s*$",
1541
+ re.MULTILINE,
1542
+ )
1543
+
1544
+
1545
+ def _find_checkbox_state_for_label(
1546
+ content: str,
1547
+ label: str,
1548
+ label_max_diffs: int | float = 0,
1549
+ ) -> bool | None:
1550
+ """Return True/False if *label* has a checkbox-style state nearby, else None.
1551
+
1552
+ Like :func:`_find_text_value_for_label`, this collects every candidate
1553
+ ``(label, state)`` across every checkbox source, scores the label, and
1554
+ returns the state attached to the highest-scoring candidate. Avoids
1555
+ adjacent-label collisions where two visually similar labels share a
1556
+ line and the wrong one gets picked just because it came first.
1557
+ """
1558
+
1559
+ # (negated_score, negated_priority, doc_order, state)
1560
+ candidates: list[tuple[float, int, int, bool]] = []
1561
+
1562
+ def _try_add(cand_label: str, state: bool, priority: int, idx: int) -> None:
1563
+ score = _label_match_score(cand_label, label, label_max_diffs)
1564
+ if score > 0.0:
1565
+ candidates.append((-score, -priority, idx, state))
1566
+
1567
+ # Markdown task-list: - [x] Label or - [ ] Label or 1. [x] Label
1568
+ for idx, match in enumerate(_MD_TASKLIST_RE.finditer(content)):
1569
+ state_char, cand_label = match.group(1), match.group(2).strip()
1570
+ _try_add(cand_label, state_char.strip().lower() == "x", priority=4, idx=idx)
1571
+
1572
+ # Label-first bullet: - Label [x] or * Label \[x]
1573
+ for idx, match in enumerate(_MD_BULLET_LABEL_FIRST_RE.finditer(content)):
1574
+ cand_label, state_char = match.group(1).strip(), match.group(2)
1575
+ _try_add(cand_label, state_char.strip().lower() == "x", priority=3, idx=idx)
1576
+
1577
+ # Bullet-less task-list: \[x] Label (no leading -/*/+/digit.)
1578
+ for idx, match in enumerate(_MD_BARE_TASKLIST_RE.finditer(content)):
1579
+ state_char, cand_label = match.group(1), match.group(2).strip()
1580
+ _try_add(cand_label, state_char.strip().lower() == "x", priority=2, idx=idx)
1581
+
1582
+ # Per-line marker tokenization (handles inline groups in either direction
1583
+ # and mid-line ASCII bracket markers after a bold label).
1584
+ inline_idx = 0
1585
+ for line in content.splitlines():
1586
+ if not _MARKER_RE.search(line):
1587
+ continue
1588
+ for cand_label, state in _tokenize_checkbox_line(line):
1589
+ _try_add(cand_label, state, priority=1, idx=inline_idx)
1590
+ inline_idx += 1
1591
+
1592
+ if candidates:
1593
+ candidates.sort()
1594
+ return candidates[0][3]
1595
+ return None
1596
+
1597
+
1598
+ def _find_checkbox_state_for_label_list(
1599
+ content: str,
1600
+ labels: list[str],
1601
+ label_max_diffs: list[int],
1602
+ ) -> bool | None:
1603
+ """Return the checkbox state for a multi-label yes/no option."""
1604
+
1605
+ candidates: list[tuple[float, int, bool]] = []
1606
+ candidate_idx = 0
1607
+ for line in content.splitlines():
1608
+ if not _MARKER_RE.search(line):
1609
+ continue
1610
+ for candidate_labels, state in _tokenize_checkbox_line_with_context(line):
1611
+ score = _checkbox_label_list_match_score(candidate_labels, labels, label_max_diffs)
1612
+ if score > 0.0:
1613
+ candidates.append((-score, candidate_idx, state))
1614
+ candidate_idx += 1
1615
+
1616
+ if candidates:
1617
+ candidates.sort()
1618
+ return candidates[0][2]
1619
+ return None
1620
+
1621
+
1622
+ # Strikethrough span — match a ``~~...~~`` block AND its contents so an edit
1623
+ # history like ``~~old~~ new`` collapses to just ``new``. ``normalize_text``
1624
+ # only strips the ``~~`` markers (leaving the crossed-out text behind), which
1625
+ # is the wrong shape when the GT records the final clean value. The pattern
1626
+ # is non-greedy and bounded to a single line so it can't span paragraphs.
1627
+ _STRIKETHROUGH_SPAN_RE = re.compile(r"~~[^~\n]+~~")
1628
+
1629
+
1630
+ def _strip_strikethrough_spans(s: str) -> str:
1631
+ return _STRIKETHROUGH_SPAN_RE.sub("", s).strip()
1632
+
1633
+
1634
+ def _compact_value_text_for_distance(s: str) -> str:
1635
+ """Keep only alphanumeric content after the normal form-value cleanup."""
1636
+
1637
+ return re.sub(r"[^0-9a-z]+", "", normalize_text(s))
1638
+
1639
+
1640
+ def _levenshtein_distance_at_most(left: str, right: str, max_distance: int) -> int:
1641
+ """Compute edit distance, stopping once it is already above the limit."""
1642
+
1643
+ if left == right:
1644
+ return 0
1645
+ if abs(len(left) - len(right)) > max_distance:
1646
+ return max_distance + 1
1647
+ if len(left) < len(right):
1648
+ left, right = right, left
1649
+
1650
+ previous = list(range(len(right) + 1))
1651
+ for i, lch in enumerate(left, start=1):
1652
+ current = [i]
1653
+ row_min = current[0]
1654
+ for j, rch in enumerate(right, start=1):
1655
+ cost = 0 if lch == rch else 1
1656
+ current.append(
1657
+ min(
1658
+ previous[j] + 1,
1659
+ current[j - 1] + 1,
1660
+ previous[j - 1] + cost,
1661
+ )
1662
+ )
1663
+ row_min = min(row_min, current[-1])
1664
+ if row_min > max_distance:
1665
+ return max_distance + 1
1666
+ previous = current
1667
+ return previous[-1]
1668
+
1669
+
1670
+ def _values_match_text(found: str, expected: str, max_diffs: int | float = 0) -> bool:
1671
+ """Compare two form-field text values with optional character tolerance.
1672
+
1673
+ Form values default to strict comparison on the meaningful text:
1674
+
1675
+ 1. Exact match after normalization (``normalize_text`` already case-folds
1676
+ and collapses whitespace), which handles ``Madison`` vs ``madison``,
1677
+ trailing whitespace, and unicode quote variants.
1678
+ 2. Strict numeric equality via ``normalize_number_string``, which lets
1679
+ ``1,234`` match ``1234`` and ``$1,234.00`` match ``1234`` (the same
1680
+ value written differently) — but rejects ``53703`` vs ``53704``.
1681
+ 3. Exact match after dropping separators/punctuation from the normalized
1682
+ value, which handles handwritten/date separators such as ``9-29`` vs
1683
+ ``9 29`` without making any character substitutions.
1684
+
1685
+ When ``max_diffs`` is positive, the compact normalized values may differ
1686
+ by that many Levenshtein edits. This is intended for hard-to-read form
1687
+ values where one digit/letter may be ambiguous; the default remains 0.
1688
+
1689
+ Strikethrough spans (``~~old~~ new``) are stripped from the *found*
1690
+ value before comparison so the parser's edit-history rendering matches
1691
+ the GT's clean final value. The expected side is left untouched on the
1692
+ assumption GT never contains ``~~``.
1693
+ """
1694
+
1695
+ found_stripped = _strip_strikethrough_spans(found)
1696
+
1697
+ f_norm = normalize_text(found_stripped)
1698
+ e_norm = normalize_text(expected)
1699
+ if f_norm == e_norm:
1700
+ return True
1701
+ if not f_norm or not e_norm:
1702
+ return False
1703
+
1704
+ f_num = normalize_number_string(found_stripped)
1705
+ e_num = normalize_number_string(expected)
1706
+ if f_num is not None and e_num is not None and f_num == e_num:
1707
+ return True
1708
+
1709
+ f_compact = _compact_value_text_for_distance(found_stripped)
1710
+ e_compact = _compact_value_text_for_distance(expected)
1711
+ if f_compact and e_compact and f_compact == e_compact:
1712
+ return True
1713
+
1714
+ allowed = int(max_diffs) if max_diffs and max_diffs > 0 else 0
1715
+ if allowed > 0 and f_compact and e_compact:
1716
+ return _levenshtein_distance_at_most(f_compact, e_compact, allowed) <= allowed
1717
+
1718
+ return False
1719
+
1720
+
1721
+ # Trailing ``(row N)`` annotation used by the form-field test generator to
1722
+ # point a label at a specific data row of a multi-column table. The column is
1723
+ # named by the prefix; ``N`` is 1-indexed over data rows (header rows are
1724
+ # skipped). This also covers simple repeated inline rows shaped as
1725
+ # ``FROM value TO value`` because some form parsers flatten ruled tables that
1726
+ # way instead of emitting a markdown table.
1727
+ _ROW_LABEL_RE = re.compile(r"\s*\(row\s+(\d+)\)\s*$", re.IGNORECASE)
1728
+ _INLINE_FROM_TO_RE = re.compile(r"^FROM\s*(?P<from>.*?)\s+TO\s*(?P<to>.*?)\s*$", re.IGNORECASE)
1729
+
1730
+
1731
+ def _split_row_label(label: str) -> tuple[str, int] | None:
1732
+ """Return ``(column_label, row_index_1based)`` if *label* has a ``(row N)``
1733
+ suffix, else None."""
1734
+
1735
+ m = _ROW_LABEL_RE.search(label)
1736
+ if not m:
1737
+ return None
1738
+ col_label = label[: m.start()].strip()
1739
+ if not col_label:
1740
+ return None
1741
+ return col_label, int(m.group(1))
1742
+
1743
+
1744
+ def _column_header_for_index(table: TableData, col_idx: int) -> str:
1745
+ """Concatenate every header cell stacked above column *col_idx* into one
1746
+ label. If the table has no recorded column headers (e.g. a markdown table
1747
+ where row 0 is the de facto header), fall back to row 0 of that column."""
1748
+
1749
+ parts: list[str] = []
1750
+ seen: set[str] = set()
1751
+ headers = getattr(table, "col_headers", {}) or {}
1752
+ for _, text in headers.get(col_idx, []):
1753
+ clean = (text or "").strip()
1754
+ if clean and clean not in seen:
1755
+ parts.append(clean)
1756
+ seen.add(clean)
1757
+ if parts:
1758
+ return " ".join(parts)
1759
+ if table.data.size and col_idx < table.data.shape[1]:
1760
+ return str(table.data[0, col_idx]).strip()
1761
+ return ""
1762
+
1763
+
1764
+ def _clean_inline_from_to_line(line: str) -> str:
1765
+ clean = _strip_html_tags(line)
1766
+ clean = re.sub(r"[*_`]+", "", clean)
1767
+ clean = clean.replace("\\", "")
1768
+ clean = re.sub(r"^\s*(?:[-+]\s+|\d+\.\s+)", "", clean)
1769
+ clean = re.sub(r"\s+", " ", clean).strip()
1770
+ return clean
1771
+
1772
+
1773
+ def _clean_inline_from_to_value(value: str) -> str:
1774
+ value = re.sub(r"(?:_|\s){3,}", " ", value)
1775
+ return value.strip(" :;-_")
1776
+
1777
+
1778
+ def _find_inline_from_to_row_label(
1779
+ content: str,
1780
+ col_label: str,
1781
+ row_n: int,
1782
+ label_max_diffs: int | float = 0,
1783
+ ) -> tuple[bool, str | None]:
1784
+ """Look up ``FROM``/``TO`` values in flattened interval rows.
1785
+
1786
+ Example parser output:
1787
+
1788
+ ``FROM none reported TO RRC``
1789
+
1790
+ The rule keeps the same row-label syntax as markdown tables:
1791
+ ``FROM (row 1)`` -> ``none reported`` and ``TO (row 1)`` -> ``RRC``.
1792
+ """
1793
+
1794
+ wants_from = _label_matches("FROM", col_label, label_max_diffs)
1795
+ wants_to = _label_matches("TO", col_label, label_max_diffs)
1796
+ if not wants_from and not wants_to:
1797
+ return False, None
1798
+
1799
+ rows: list[tuple[str, str]] = []
1800
+ for raw_line in content.splitlines():
1801
+ if "|" in raw_line:
1802
+ continue
1803
+ line = _clean_inline_from_to_line(raw_line)
1804
+ if not line:
1805
+ continue
1806
+ match = _INLINE_FROM_TO_RE.match(line)
1807
+ if not match:
1808
+ continue
1809
+ rows.append(
1810
+ (
1811
+ _clean_inline_from_to_value(match.group("from")),
1812
+ _clean_inline_from_to_value(match.group("to")),
1813
+ )
1814
+ )
1815
+
1816
+ if not rows:
1817
+ return False, None
1818
+ if row_n < 1 or row_n > len(rows):
1819
+ return True, ""
1820
+ row_from, row_to = rows[row_n - 1]
1821
+ return True, row_from if wants_from else row_to
1822
+
1823
+
1824
+ def _find_table_cell_for_row_label(
1825
+ content: str,
1826
+ label: str,
1827
+ label_max_diffs: int | float = 0,
1828
+ ) -> tuple[bool, str | None]:
1829
+ """Look up ``"<col_label> (row N)"`` in any multi-column table.
1830
+
1831
+ Returns ``(label_seen, value_or_None)``. ``label_seen`` is True if a
1832
+ matching column was found in some table, even when the data row is out
1833
+ of range or the cell is empty — that distinction lets text rules with
1834
+ empty expected values pass on real empty cells without giving signature
1835
+ rules a free pass for missing labels.
1836
+ """
1837
+
1838
+ parsed = _split_row_label(label)
1839
+ if parsed is None:
1840
+ return False, None
1841
+ col_label, row_n = parsed
1842
+
1843
+ label_seen = False
1844
+ for table in parse_html_tables(content) + parse_markdown_tables(content):
1845
+ if table.data.size == 0:
1846
+ continue
1847
+ rows, cols = table.data.shape
1848
+ # Determine which rows are headers. For HTML tables, header_rows is
1849
+ # populated from <thead>/<th>. For markdown tables, parse_markdown_tables
1850
+ # records header_rows={0} when a separator row is present.
1851
+ header_rows = getattr(table, "header_rows", set()) or set()
1852
+ n_header = (max(header_rows) + 1) if header_rows else 0
1853
+ data_row_idx = n_header + (row_n - 1)
1854
+
1855
+ for col_idx in range(cols):
1856
+ header_text = _column_header_for_index(table, col_idx)
1857
+ if not header_text:
1858
+ continue
1859
+ if not _label_matches(header_text, col_label, label_max_diffs):
1860
+ continue
1861
+ label_seen = True
1862
+ if 0 <= data_row_idx < rows:
1863
+ cell_value = str(table.data[data_row_idx, col_idx]).strip()
1864
+ if cell_value:
1865
+ return True, cell_value
1866
+ # Column matched but cell out of range or empty — keep looking
1867
+ # in case a sibling table has the same header populated.
1868
+
1869
+ inline_seen, inline_value = _find_inline_from_to_row_label(content, col_label, row_n, label_max_diffs)
1870
+ if inline_seen:
1871
+ return True, inline_value
1872
+
1873
+ if label_seen:
1874
+ return True, ""
1875
+ return False, None
1876
+
1877
+
1878
+ def _table_anchor_labels_for_cell(table: TableData, row_idx: int, col_idx: int) -> list[str]:
1879
+ """Return visible row/column labels that identify a table value cell."""
1880
+
1881
+ labels: list[str] = []
1882
+ header_rows = getattr(table, "header_rows", set()) or set()
1883
+ row_headers = getattr(table, "row_headers", {}) or {}
1884
+
1885
+ labels.append(_column_header_for_index(table, col_idx))
1886
+
1887
+ for _, text in row_headers.get(row_idx, []):
1888
+ labels.append(str(text))
1889
+
1890
+ # Keep markdown and simple HTML tables robust when header metadata is
1891
+ # sparse: add the header cells above the value and the cells to its left.
1892
+ for rr in sorted(header_rows):
1893
+ if rr < row_idx and col_idx < table.data.shape[1]:
1894
+ labels.append(str(table.data[rr, col_idx]))
1895
+ for cc in range(col_idx):
1896
+ labels.append(str(table.data[row_idx, cc]))
1897
+
1898
+ return _dedupe_nonempty_text(labels)
1899
+
1900
+
1901
+ def _compact_anchor_label(s: str) -> str:
1902
+ return _compact_label_text_for_distance(s)
1903
+
1904
+
1905
+ def _compact_contains_with_diffs(haystack: str, needle: str, allowed: int) -> bool:
1906
+ """Return True when any compact substring matches within edit distance."""
1907
+
1908
+ if allowed <= 0 or not haystack or not needle:
1909
+ return False
1910
+ if len(needle) > len(haystack):
1911
+ return _levenshtein_distance_at_most(haystack, needle, allowed) <= allowed
1912
+
1913
+ min_len = max(1, len(needle) - allowed)
1914
+ max_len = min(len(haystack), len(needle) + allowed)
1915
+ for start in range(0, len(haystack) - min_len + 1):
1916
+ for size in range(min_len, max_len + 1):
1917
+ end = start + size
1918
+ if end > len(haystack):
1919
+ break
1920
+ if _levenshtein_distance_at_most(haystack[start:end], needle, allowed) <= allowed:
1921
+ return True
1922
+ return False
1923
+
1924
+
1925
+ def _table_anchor_label_matches(candidate: str, required: str, max_diffs: int | float = 0) -> bool:
1926
+ """Stricter label match for multi-label table anchors.
1927
+
1928
+ The legacy fuzzy label scorer is intentionally broad for single-key form
1929
+ labels, but it is too broad for sibling columns like PLUG #1 / PLUG #2.
1930
+ Multi-key table lookup needs exact or containment-style evidence instead.
1931
+ """
1932
+
1933
+ cand = normalize_text(candidate)
1934
+ req = normalize_text(required)
1935
+ if not cand or not req:
1936
+ return False
1937
+ if _strip_label_punct(cand) == _strip_label_punct(req):
1938
+ return True
1939
+ if req in cand or cand in req:
1940
+ return True
1941
+
1942
+ cand_compact = _compact_anchor_label(candidate)
1943
+ req_compact = _compact_anchor_label(required)
1944
+ if not cand_compact or not req_compact:
1945
+ return False
1946
+ if cand_compact == req_compact:
1947
+ return True
1948
+ min_containment_len = 4
1949
+ if (
1950
+ len(req_compact) >= min_containment_len
1951
+ and req_compact in cand_compact
1952
+ or len(cand_compact) >= min_containment_len
1953
+ and cand_compact in req_compact
1954
+ ):
1955
+ return True
1956
+
1957
+ allowed = int(max_diffs) if max_diffs and max_diffs > 0 else 0
1958
+ return allowed > 0 and (
1959
+ _levenshtein_distance_at_most(cand_compact, req_compact, allowed) <= allowed
1960
+ or _compact_contains_with_diffs(cand_compact, req_compact, allowed)
1961
+ )
1962
+
1963
+
1964
+ def _labels_match_all(
1965
+ candidate_labels: list[str],
1966
+ required_labels: list[str],
1967
+ label_max_diffs: list[int],
1968
+ ) -> bool:
1969
+ """Order-insensitive match: every required label must match one anchor."""
1970
+
1971
+ for idx, required in enumerate(required_labels):
1972
+ allowed = label_max_diffs[idx] if idx < len(label_max_diffs) else 0
1973
+ if not any(_table_anchor_label_matches(candidate, required, allowed) for candidate in candidate_labels):
1974
+ return False
1975
+ return True
1976
+
1977
+
1978
+ def _is_table_value_cell(table: TableData, row_idx: int, col_idx: int) -> bool:
1979
+ header_rows = getattr(table, "header_rows", set()) or set()
1980
+ header_cols = getattr(table, "header_cols", set()) or set()
1981
+ header_cells = getattr(table, "header_cells", set()) or set()
1982
+
1983
+ if row_idx in header_rows:
1984
+ return False
1985
+ if header_cells:
1986
+ if (row_idx, col_idx) in header_cells:
1987
+ return False
1988
+ elif col_idx in header_cols:
1989
+ return False
1990
+
1991
+ # In row/column form tables, the first data column is normally the row
1992
+ # label stub ("Cementing Date", "Depth...", etc.), not a value cell.
1993
+ if col_idx == 0 and table.data.shape[1] > 1:
1994
+ return False
1995
+ return True
1996
+
1997
+
1998
+ def _find_table_cell_for_label_list(
1999
+ content: str,
2000
+ labels: list[str],
2001
+ expected_values: list[str] | None = None,
2002
+ max_diffs: int | float = 0,
2003
+ label_max_diffs: list[int] | None = None,
2004
+ ) -> tuple[bool, str | None]:
2005
+ """Look up a value cell by multiple row/column-style form labels.
2006
+
2007
+ This is deliberately narrower than full table evaluation: it only asks
2008
+ whether the same table cell is anchored by all requested visible labels.
2009
+ The rule JSON does not need row/column roles; labels are matched as an
2010
+ unordered set against the nearby table anchors.
2011
+ """
2012
+
2013
+ label_seen = False
2014
+ fallback_value: str | None = None
2015
+ label_diffs = label_max_diffs or [0] * len(labels)
2016
+ for table in parse_html_tables(content) + parse_markdown_tables(content):
2017
+ if table.data.size == 0:
2018
+ continue
2019
+ rows, cols = table.data.shape
2020
+
2021
+ for row_idx in range(rows):
2022
+ for col_idx in range(cols):
2023
+ if not _is_table_value_cell(table, row_idx, col_idx):
2024
+ continue
2025
+ anchors = _table_anchor_labels_for_cell(table, row_idx, col_idx)
2026
+ if not _labels_match_all(anchors, labels, label_diffs):
2027
+ continue
2028
+
2029
+ label_seen = True
2030
+ value = str(table.data[row_idx, col_idx]).strip()
2031
+ if expected_values:
2032
+ for exp in expected_values:
2033
+ if _values_match_text(value, exp, max_diffs):
2034
+ return True, value
2035
+ if fallback_value is None or (not fallback_value and value):
2036
+ fallback_value = value
2037
+
2038
+ if label_seen:
2039
+ return True, fallback_value or ""
2040
+ return False, None
2041
+
2042
+
2043
+ def _scope_to_page(content: str, parse_output, page: int | None) -> str: # type: ignore[no-untyped-def]
2044
+ """Return per-page markdown when ``parse_output`` and ``page`` are both set.
2045
+
2046
+ Fail-closed: once per-page IR is present (``pages`` or ``layout_pages``),
2047
+ scoping is strict — if the requested page has no entry (or its markdown
2048
+ is empty), return ``""`` rather than the full document. The old lenient
2049
+ fallback let repeated header/footer fields satisfy page-N rules on the
2050
+ wrong page and silently masked page-level extraction failures (see
2051
+ PR #897 for the reducto/extend variant of the same bug).
2052
+
2053
+ Only when no per-page IR is available (both lists empty) do we fall
2054
+ back to the document-level ``content``. Providers that emit neither
2055
+ list never had fair per-page scoring; the fallback preserves prior
2056
+ behavior rather than introducing a silent regression.
2057
+
2058
+ When ``layout_pages`` carries the per-page split but ``md`` is empty,
2059
+ synthesize from ``items`` (priority ``md > html > value``). ``html``
2060
+ ranks above ``value`` so table items keep their structure for the
2061
+ HTML cell-neighbor matcher.
2062
+
2063
+ ``parse_output`` is typed as ``ParseOutput`` upstream but kept loose
2064
+ here to avoid an import cycle.
2065
+ """
2066
+
2067
+ if parse_output is None or page is None:
2068
+ return content
2069
+
2070
+ pages = getattr(parse_output, "pages", None) or []
2071
+ layout_pages = getattr(parse_output, "layout_pages", None) or []
2072
+
2073
+ if not pages and not layout_pages:
2074
+ # Provider produced no per-page IR at all — fall back to full doc.
2075
+ return content
2076
+
2077
+ if pages:
2078
+ for p in pages:
2079
+ # PageIR.page_index is 0-indexed; rule.page is 1-indexed.
2080
+ if getattr(p, "page_index", None) == page - 1:
2081
+ return getattr(p, "markdown", "") or ""
2082
+ # ``pages`` populated but no matching page — fail closed.
2083
+ if not layout_pages:
2084
+ return ""
2085
+ # Fall through to ``layout_pages`` lookup; some providers populate
2086
+ # only one of the two lists per page.
2087
+
2088
+ for lp in layout_pages:
2089
+ if getattr(lp, "page_number", None) != page:
2090
+ continue
2091
+ md = getattr(lp, "md", "") or ""
2092
+ if md:
2093
+ return md
2094
+ # Synthesize from items: md > html > value (html ranks above value
2095
+ # so table items keep their structure for the HTML cell matcher).
2096
+ parts: list[str] = []
2097
+ for it in getattr(lp, "items", None) or []:
2098
+ text = getattr(it, "md", "") or getattr(it, "html", "") or getattr(it, "value", "")
2099
+ if text:
2100
+ parts.append(text)
2101
+ return "\n\n".join(parts)
2102
+
2103
+ # Per-page IR present but page not found — fail closed.
2104
+ return ""
2105
+
2106
+
2107
+ class FormFieldRule(ParseTestRule):
2108
+ """Test rule for form-field key-value extraction.
2109
+
2110
+ Locates a labeled field by its visible label in the parsed markdown/HTML
2111
+ and checks the extracted value matches the expected one.
2112
+ """
2113
+
2114
+ def __init__(self, rule_data: ParseFormFieldRule | dict):
2115
+ super().__init__(rule_data)
2116
+ rule_data = cast(ParseFormFieldRule, self._rule_data)
2117
+
2118
+ if self.type != TestType.FORM_FIELD.value:
2119
+ raise ValueError(f"Invalid type for FormFieldRule: {self.type}")
2120
+
2121
+ self.label = rule_data.label
2122
+ label_parts = _label_parts_with_indexes(rule_data.label)
2123
+ self.labels = [part for part, _ in label_parts]
2124
+ label_indexes = [idx for _, idx in label_parts]
2125
+ self.label_max_diffs = _label_max_diffs_parts(
2126
+ rule_data.label_max_diffs,
2127
+ len(self.labels),
2128
+ label_indexes,
2129
+ )
2130
+ self.value_max_diffs = rule_data.value_max_diffs
2131
+ self.value = rule_data.value
2132
+ self.value_type = rule_data.value_type
2133
+
2134
+ if not self.labels:
2135
+ raise ValueError("label field cannot be empty")
2136
+
2137
+ def _content_for_match(self, md_content: str) -> str:
2138
+ return _scope_to_page(md_content, self.parse_output, self.page)
2139
+
2140
+ def run(
2141
+ self,
2142
+ md_content: str,
2143
+ normalized_content: str | None = None,
2144
+ ) -> tuple[bool, str, float]:
2145
+ scoped = self._content_for_match(md_content)
2146
+ if self.value_type == "text":
2147
+ return self._run_text(scoped)
2148
+ if self.value_type == "checkbox":
2149
+ return self._run_checkbox(scoped)
2150
+ if self.value_type == "signature":
2151
+ return self._run_signature(scoped)
2152
+ return False, f"unknown value_type: {self.value_type}", 0.0
2153
+
2154
+ def _run_text(self, content: str) -> tuple[bool, str, float]:
2155
+ # `self.value` can be a list of acceptable alternatives — pass if any matches.
2156
+ expected_alternatives = _value_alternatives(self.value)
2157
+ label = self.labels[0]
2158
+ # Multi-col table cell lookup ("Column Name (row N)") takes precedence
2159
+ # over the bold-colon / 2-col / glyph paths because the suffix
2160
+ # explicitly names a tabular position.
2161
+ if len(self.labels) > 1:
2162
+ label_found, value = _find_table_cell_for_label_list(
2163
+ content,
2164
+ self.labels,
2165
+ expected_alternatives,
2166
+ self.value_max_diffs,
2167
+ self.label_max_diffs,
2168
+ )
2169
+ elif _split_row_label(label) is not None:
2170
+ label_found, value = _find_table_cell_for_row_label(content, label, self.label_max_diffs[0])
2171
+ else:
2172
+ # Thread expected_alternatives so the matcher can disambiguate
2173
+ # among candidates at the same top label-match score. The rule's
2174
+ # position in the test list says nothing about which page
2175
+ # occurrence the GT refers to when a label legitimately repeats
2176
+ # (e.g. ``KB`` appearing as both an elevation label and as the
2177
+ # value of ``LOG MEASURED FROM`` in well-log headers). The GT
2178
+ # oracle is applied **only** to candidates tied at the head
2179
+ # label-match score, so it never leaks across adjacent labels
2180
+ # like Country vs County which live at different score levels.
2181
+ label_found, value = _find_text_value_for_label(
2182
+ content,
2183
+ label,
2184
+ expected_alternatives,
2185
+ self.value_max_diffs,
2186
+ self.label_max_diffs[0],
2187
+ )
2188
+ if not label_found:
2189
+ return False, f"label not found: {_format_label_for_message(self.label)}", 0.0
2190
+ # value may be "" (label found, cell/value empty); _values_match_text
2191
+ # handles empty == empty correctly so empty-expected rules can pass.
2192
+ for expected in expected_alternatives:
2193
+ if _values_match_text(value or "", expected, self.value_max_diffs):
2194
+ return True, "match", 1.0
2195
+ if len(expected_alternatives) == 1:
2196
+ return False, f"expected {expected_alternatives[0]!r}, got {(value or '')!r}", 0.0
2197
+ return False, f"expected any of {expected_alternatives!r}, got {(value or '')!r}", 0.0
2198
+
2199
+ def _run_checkbox(self, content: str) -> tuple[bool, str, float]:
2200
+ expected_bool = _coerce_bool(self.value)
2201
+ if expected_bool is None:
2202
+ return False, f"checkbox value must be coercible to bool, got {self.value!r}", 0.0
2203
+
2204
+ # Prefer a real checkbox-shaped match.
2205
+ state = (
2206
+ _find_checkbox_state_for_label(content, self.labels[0], self.label_max_diffs[0])
2207
+ if len(self.labels) == 1
2208
+ else _find_checkbox_state_for_label_list(content, self.labels, self.label_max_diffs)
2209
+ )
2210
+ if state is None:
2211
+ # Fall back to a text-shaped value (e.g. **Married:** Yes / No).
2212
+ if len(self.labels) > 1:
2213
+ label_found, text_value = _find_table_cell_for_label_list(
2214
+ content,
2215
+ self.labels,
2216
+ label_max_diffs=self.label_max_diffs,
2217
+ )
2218
+ elif _split_row_label(self.labels[0]) is not None:
2219
+ label_found, text_value = _find_table_cell_for_row_label(
2220
+ content,
2221
+ self.labels[0],
2222
+ self.label_max_diffs[0],
2223
+ )
2224
+ else:
2225
+ label_found, text_value = _find_text_value_for_label(
2226
+ content,
2227
+ self.labels[0],
2228
+ label_max_diffs=self.label_max_diffs[0],
2229
+ )
2230
+ if not label_found or not text_value:
2231
+ return False, f"label not found: {_format_label_for_message(self.label)}", 0.0
2232
+ state = _coerce_bool(text_value)
2233
+ if state is None:
2234
+ return False, f"could not interpret {text_value!r} as checkbox state", 0.0
2235
+
2236
+ if state == expected_bool:
2237
+ return True, "match", 1.0
2238
+ return False, f"expected {expected_bool}, got {state}", 0.0
2239
+
2240
+ def _run_signature(self, content: str) -> tuple[bool, str, float]:
2241
+ # Relaxed semantics: the rule's value is treated as a presence indicator,
2242
+ # not a strict bool. A non-empty string (e.g. the actual signed name) is
2243
+ # equivalent to True — the matcher only checks "is something signed here"
2244
+ # rather than the exact handwriting. An empty string / False / None means
2245
+ # "expected unsigned". A list value collapses the same way: any non-empty
2246
+ # alternative means "expected signed".
2247
+ if isinstance(self.value, bool):
2248
+ expected_signed = self.value
2249
+ elif isinstance(self.value, list):
2250
+ expected_signed = any(bool(str(v).strip()) for v in self.value)
2251
+ else:
2252
+ expected_signed = bool(str(self.value).strip())
2253
+
2254
+ # Track label presence separately from value presence — an absent label
2255
+ # must NOT pass an "expected unsigned" rule. A form tuple benchmark
2256
+ # requires the parser to surface the field at all.
2257
+ if len(self.labels) > 1:
2258
+ label_found, text_value = _find_table_cell_for_label_list(
2259
+ content,
2260
+ self.labels,
2261
+ label_max_diffs=self.label_max_diffs,
2262
+ )
2263
+ else:
2264
+ label_found, text_value = _find_text_value_for_label(
2265
+ content,
2266
+ self.labels[0],
2267
+ label_max_diffs=self.label_max_diffs[0],
2268
+ )
2269
+ if not label_found:
2270
+ return False, f"label not found: {_format_label_for_message(self.label)}", 0.0
2271
+ signed = bool(text_value and text_value.strip())
2272
+ if signed == expected_signed:
2273
+ return True, "match", 1.0
2274
+ return False, f"expected signed={expected_signed}, got signed={signed}", 0.0