parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,2274 @@
|
|
|
1
|
+
"""Form field test rule.
|
|
2
|
+
|
|
3
|
+
A `form_field` rule locates a labeled field in the parsed markdown/HTML and
|
|
4
|
+
checks its value. Three value types are supported in v0.1: ``text``,
|
|
5
|
+
``checkbox``, and ``signature``. The matcher tries a small set of
|
|
6
|
+
high-confidence patterns:
|
|
7
|
+
|
|
8
|
+
- Bold-colon (``**Label:** value`` and ``**Label**: value``) — supports
|
|
9
|
+
multiple bold-colon pairs on the same line.
|
|
10
|
+
- Plain colon on its own line (``Label: value``).
|
|
11
|
+
- 2-column markdown tables AND 2-column HTML tables (label in first cell,
|
|
12
|
+
value in second).
|
|
13
|
+
- Per-line checkbox tokenization for inline groups, handling both
|
|
14
|
+
glyph-first (``☐ Single ☑ Married``) and label-first
|
|
15
|
+
(``Single ☐ Married ☑``) orderings.
|
|
16
|
+
- Markdown task-list checkboxes (``- [x] Label`` / ``- [ ] Label``).
|
|
17
|
+
- Multi-label yes/no checkbox groups, e.g.
|
|
18
|
+
``["Multistage cement?", "No"]`` matches
|
|
19
|
+
``Multistage cement? Yes [ ] No [x]``.
|
|
20
|
+
- Multi-label table cells where one value is identified by a row/column
|
|
21
|
+
label set, e.g. ``["PLUG #1", "Cementing Date"]``. Label-list order is
|
|
22
|
+
ignored; all labels must match anchors around the same value cell.
|
|
23
|
+
|
|
24
|
+
When the rule has a ``page`` and the metric injects a ``parse_output``,
|
|
25
|
+
matching is scoped to that page's markdown only.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import re
|
|
31
|
+
from typing import cast
|
|
32
|
+
|
|
33
|
+
from bs4 import BeautifulSoup
|
|
34
|
+
from rapidfuzz import fuzz
|
|
35
|
+
|
|
36
|
+
from parse_bench.evaluation.metrics.parse.rules_base import (
|
|
37
|
+
CELL_FUZZY_MATCH_THRESHOLD,
|
|
38
|
+
ParseTestRule,
|
|
39
|
+
)
|
|
40
|
+
from parse_bench.evaluation.metrics.parse.rules_chart import normalize_number_string
|
|
41
|
+
from parse_bench.evaluation.metrics.parse.table_parsing import (
|
|
42
|
+
TableData,
|
|
43
|
+
parse_html_tables,
|
|
44
|
+
parse_markdown_tables,
|
|
45
|
+
)
|
|
46
|
+
from parse_bench.evaluation.metrics.parse.test_types import TestType
|
|
47
|
+
from parse_bench.evaluation.metrics.parse.utils import normalize_text
|
|
48
|
+
from parse_bench.test_cases.parse_rule_schemas import ParseFormFieldRule
|
|
49
|
+
|
|
50
|
+
# Glyphs that represent a checked / unchecked state. Sourced from the most
|
|
51
|
+
# common Unicode shapes parsers emit when surfacing form widgets. The extra
|
|
52
|
+
# circle/dot glyphs (◉●⦿/○◯⊙) appear in Gemini and OpenAI outputs which mirror
|
|
53
|
+
# radio-button widgets; the extra X glyphs (⊠⊗) appear in IRS/USCIS forms.
|
|
54
|
+
_CHECKED_GLYPHS = "☑☒▣✓✔◉●⦿⊠⊗"
|
|
55
|
+
_UNCHECKED_GLYPHS = "☐□○◯⊙"
|
|
56
|
+
_CHECKBOX_GLYPHS = _CHECKED_GLYPHS + _UNCHECKED_GLYPHS
|
|
57
|
+
_GLYPH_RE = re.compile(f"[{_CHECKBOX_GLYPHS}]")
|
|
58
|
+
# Combined marker regex: either a single Unicode glyph OR an ASCII bracket
|
|
59
|
+
# pair ``[x]`` / ``[ ]`` (with optional ``\`` escapes around the brackets).
|
|
60
|
+
# Used by the per-line tokenizer so inline ASCII checkbox groups
|
|
61
|
+
# (``\[x] Single \[ ] Married``) are parsed the same as Unicode-glyph groups.
|
|
62
|
+
_MARKER_RE = re.compile(rf"[{_CHECKBOX_GLYPHS}]|\\?\[[ xX]\\?\]")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _marker_is_checked(token: str) -> bool:
|
|
66
|
+
"""Decide if a checkbox marker token represents the checked state."""
|
|
67
|
+
|
|
68
|
+
if len(token) == 1:
|
|
69
|
+
return token in _CHECKED_GLYPHS
|
|
70
|
+
return any(c in "xX" for c in token)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# Boolean coercion table for textual yes/no values.
|
|
74
|
+
_TRUTHY_TEXT = {"yes", "y", "true", "t", "1", "checked", "x", "selected", "on"}
|
|
75
|
+
_FALSY_TEXT = {"no", "n", "false", "f", "0", "unchecked", "unselected", "off", ""}
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
_PARTIAL_RATIO_THRESHOLD = 0.90
|
|
79
|
+
_PARTIAL_RATIO_MIN_LEN = 6
|
|
80
|
+
# Penalty applied to the partial-ratio score so a partial hit cannot beat
|
|
81
|
+
# an equally strong strict-ratio hit. Empirically 0.05 keeps partial 1.0
|
|
82
|
+
# above strict 0.86 (so e.g. ``API NO. (if available)`` still resolves
|
|
83
|
+
# against GT ``API NO.`` when the only candidate is the full noisy label)
|
|
84
|
+
# while preventing partial 0.95 (``County`` ⊂ ``Country``) from beating a
|
|
85
|
+
# strict 1.0 ``Country`` match on the *correct* row.
|
|
86
|
+
_PARTIAL_RATIO_PENALTY = 0.05
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _strip_label_punct(s: str) -> str:
|
|
90
|
+
"""Remove punctuation that varies between abbreviation styles.
|
|
91
|
+
|
|
92
|
+
``K.B.`` vs ``KB``, ``D.F.`` vs ``DF``, ``API NO:`` vs ``API NO``,
|
|
93
|
+
``Tel.`` vs ``Tel`` are the same label semantically. This strips
|
|
94
|
+
dots, colons, and commas — separators that the parser may add or drop
|
|
95
|
+
while preserving the underlying tokens. Operates after
|
|
96
|
+
``normalize_text`` so it sees a case-folded, whitespace-collapsed
|
|
97
|
+
string.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
s = s.replace(".", "")
|
|
101
|
+
s = s.replace(":", "")
|
|
102
|
+
s = s.replace(",", "")
|
|
103
|
+
s = re.sub(r"\s+", " ", s).strip()
|
|
104
|
+
return s
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _compact_label_text_for_distance(s: str) -> str:
|
|
108
|
+
return re.sub(r"[^0-9a-z]+", "", normalize_text(s))
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _label_match_score(candidate: str, label: str, max_diffs: int | float = 0) -> float:
|
|
112
|
+
"""Score how well *candidate* matches *label* on the ``_label_matches`` axes.
|
|
113
|
+
|
|
114
|
+
Returns ``0.0`` when the candidate fails every path (same as
|
|
115
|
+
``_label_matches`` returning False); otherwise returns a value in
|
|
116
|
+
``(0.0, 1.0]`` where higher means a better label match.
|
|
117
|
+
|
|
118
|
+
Why scoring instead of a bool: when the document contains two visually
|
|
119
|
+
similar labels (the classic ``Country`` / ``County`` collision), the
|
|
120
|
+
old first-match-wins iteration would latch onto whichever fuzzy hit
|
|
121
|
+
came first and return that row's value. With a scoring function, the
|
|
122
|
+
caller can collect every candidate and pick the *best* match — an
|
|
123
|
+
exact ``Country`` (score ``1.0``) beats a fuzzy ``County`` (score
|
|
124
|
+
``~0.92``) even when ``County`` appears earlier in the markdown.
|
|
125
|
+
|
|
126
|
+
Scoring:
|
|
127
|
+
|
|
128
|
+
- Strict-ratio path (``fuzz.ratio >= CELL_FUZZY_MATCH_THRESHOLD``):
|
|
129
|
+
score = the ratio itself.
|
|
130
|
+
- Partial-ratio fallback (``fuzz.partial_ratio >= _PARTIAL_RATIO_THRESHOLD``
|
|
131
|
+
with ``shorter >= _PARTIAL_RATIO_MIN_LEN``): score = the partial
|
|
132
|
+
ratio minus ``_PARTIAL_RATIO_PENALTY`` (currently 0.05). The penalty
|
|
133
|
+
keeps partial hits strictly below same-strength strict hits — a
|
|
134
|
+
partial 1.0 (``County`` substring inside ``County, TX``) scores
|
|
135
|
+
``0.95``, which still loses to any strict-ratio match >= 0.95 but
|
|
136
|
+
wins over a strict-ratio 0.86 fuzzy match.
|
|
137
|
+
- Punctuation-stripped exact path
|
|
138
|
+
(``_strip_label_punct(cand) == _strip_label_punct(lbl)``): score
|
|
139
|
+
``1.0``. Exact-equality (not fuzz) keeps this path narrow — short
|
|
140
|
+
labels like ``KB`` won't collide with ``KBC`` (``fuzz.ratio`` happens
|
|
141
|
+
to hit exactly 0.80 between those two strings, which would leak
|
|
142
|
+
through if we ran fuzz on the stripped variants) while still
|
|
143
|
+
matching dotted abbreviation variants like ``K.B.`` ≡ ``KB``.
|
|
144
|
+
- Best path wins when several fire.
|
|
145
|
+
"""
|
|
146
|
+
|
|
147
|
+
cand = normalize_text(candidate)
|
|
148
|
+
lbl = normalize_text(label)
|
|
149
|
+
if not cand or not lbl:
|
|
150
|
+
return 0.0
|
|
151
|
+
|
|
152
|
+
best = 0.0
|
|
153
|
+
ratio = fuzz.ratio(cand, lbl) / 100.0
|
|
154
|
+
if ratio >= CELL_FUZZY_MATCH_THRESHOLD:
|
|
155
|
+
best = ratio
|
|
156
|
+
shorter = min(len(cand), len(lbl))
|
|
157
|
+
if shorter >= _PARTIAL_RATIO_MIN_LEN:
|
|
158
|
+
partial = fuzz.partial_ratio(cand, lbl) / 100.0
|
|
159
|
+
if partial >= _PARTIAL_RATIO_THRESHOLD:
|
|
160
|
+
penalized = max(partial - _PARTIAL_RATIO_PENALTY, 0.0)
|
|
161
|
+
if penalized > best:
|
|
162
|
+
best = penalized
|
|
163
|
+
|
|
164
|
+
# Punctuation-stripped exact equality — narrowest of the three paths,
|
|
165
|
+
# only fires when the strip actually collapses two different surface
|
|
166
|
+
# forms onto the same string. Scored at 1.0 so legitimate abbreviation
|
|
167
|
+
# variants beat a coincidental ratio-0.80 collision (the ``K.B.``/
|
|
168
|
+
# ``KBC`` boundary case) when both candidates appear in the document.
|
|
169
|
+
cand_stripped = _strip_label_punct(cand)
|
|
170
|
+
lbl_stripped = _strip_label_punct(lbl)
|
|
171
|
+
if cand_stripped and lbl_stripped and cand_stripped == lbl_stripped:
|
|
172
|
+
if 1.0 > best:
|
|
173
|
+
best = 1.0
|
|
174
|
+
|
|
175
|
+
allowed = int(max_diffs) if max_diffs and max_diffs > 0 else 0
|
|
176
|
+
if allowed > 0:
|
|
177
|
+
cand_compact = _compact_label_text_for_distance(candidate)
|
|
178
|
+
lbl_compact = _compact_label_text_for_distance(label)
|
|
179
|
+
if cand_compact and lbl_compact:
|
|
180
|
+
dist = _levenshtein_distance_at_most(cand_compact, lbl_compact, allowed)
|
|
181
|
+
if dist <= allowed:
|
|
182
|
+
tolerant_score = max(0.01, 1.0 - (dist / max(len(cand_compact), len(lbl_compact), 1)))
|
|
183
|
+
if tolerant_score > best:
|
|
184
|
+
best = tolerant_score
|
|
185
|
+
|
|
186
|
+
return best
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _label_matches(candidate: str, label: str, max_diffs: int | float = 0) -> bool:
|
|
190
|
+
"""Boolean predicate over :func:`_label_match_score` for legacy callers.
|
|
191
|
+
|
|
192
|
+
Used by callers that only need a boolean (adjacent-line fallback,
|
|
193
|
+
underscore-blank label-seen detection, checkbox-state matching). The
|
|
194
|
+
main text-value lookup uses :func:`_label_match_score` directly so it
|
|
195
|
+
can score-and-pick-best across multiple candidate KV pairs.
|
|
196
|
+
"""
|
|
197
|
+
|
|
198
|
+
return _label_match_score(candidate, label, max_diffs) > 0.0
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _coerce_bool(value: str | bool | list[str]) -> bool | None:
|
|
202
|
+
"""Coerce a value to True/False, or None if ambiguous.
|
|
203
|
+
|
|
204
|
+
List inputs are not supported by checkbox semantics and return None;
|
|
205
|
+
the caller surfaces a "must be coercible to bool" error in that case.
|
|
206
|
+
"""
|
|
207
|
+
|
|
208
|
+
if isinstance(value, bool):
|
|
209
|
+
return value
|
|
210
|
+
if isinstance(value, list):
|
|
211
|
+
return None
|
|
212
|
+
text = str(value).strip().lower()
|
|
213
|
+
if len(text) == 1 and text in _CHECKED_GLYPHS:
|
|
214
|
+
return True
|
|
215
|
+
if len(text) == 1 and text in _UNCHECKED_GLYPHS:
|
|
216
|
+
return False
|
|
217
|
+
if text in _TRUTHY_TEXT:
|
|
218
|
+
return True
|
|
219
|
+
if text in _FALSY_TEXT:
|
|
220
|
+
return False
|
|
221
|
+
return None
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _value_alternatives(value: str | bool | list[str]) -> list[str]:
|
|
225
|
+
"""Return the list of acceptable string values for a text-typed rule.
|
|
226
|
+
|
|
227
|
+
Supports both single-string and list-of-strings GTs. A list lets a rule
|
|
228
|
+
declare multiple acceptable readings for genuinely ambiguous fields
|
|
229
|
+
(e.g. illegible handwriting). The single-string form is the default and
|
|
230
|
+
keeps the GT clean for the common case.
|
|
231
|
+
"""
|
|
232
|
+
|
|
233
|
+
if isinstance(value, list):
|
|
234
|
+
return [str(v) for v in value]
|
|
235
|
+
return [str(value)]
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _label_parts_with_indexes(label: str | list[str]) -> list[tuple[str, int]]:
|
|
239
|
+
"""Normalize form-field labels and keep original indexes for aligned tolerances."""
|
|
240
|
+
if isinstance(label, list):
|
|
241
|
+
return [(text, idx) for idx, part in enumerate(label) if (text := str(part).strip())]
|
|
242
|
+
text = str(label).strip()
|
|
243
|
+
return [(text, 0)] if text else []
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _label_parts(label: str | list[str]) -> list[str]:
|
|
247
|
+
"""Normalize a form-field label into one or more required visible keys."""
|
|
248
|
+
|
|
249
|
+
return [part for part, _ in _label_parts_with_indexes(label)]
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _format_label_for_message(label: str | list[str]) -> str:
|
|
253
|
+
parts = _label_parts(label)
|
|
254
|
+
if len(parts) <= 1:
|
|
255
|
+
return repr(parts[0] if parts else "")
|
|
256
|
+
return repr(parts)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _label_max_diffs_parts(
|
|
260
|
+
value: int | float | list[int | float],
|
|
261
|
+
count: int,
|
|
262
|
+
source_indexes: list[int] | None = None,
|
|
263
|
+
) -> list[int]:
|
|
264
|
+
if isinstance(value, list):
|
|
265
|
+
indexes = source_indexes or list(range(count))
|
|
266
|
+
return [
|
|
267
|
+
int(value[idx]) if idx < len(value) and isinstance(value[idx], (int, float)) and value[idx] > 0 else 0
|
|
268
|
+
for idx in indexes[:count]
|
|
269
|
+
] + [0] * max(count - len(indexes), 0)
|
|
270
|
+
n = int(value) if isinstance(value, (int, float)) and value > 0 else 0
|
|
271
|
+
return [n] * count
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _dedupe_nonempty_text(parts: list[str]) -> list[str]:
|
|
275
|
+
out: list[str] = []
|
|
276
|
+
seen: set[str] = set()
|
|
277
|
+
for part in parts:
|
|
278
|
+
clean = str(part).strip()
|
|
279
|
+
if not clean:
|
|
280
|
+
continue
|
|
281
|
+
key = normalize_text(clean)
|
|
282
|
+
if key in seen:
|
|
283
|
+
continue
|
|
284
|
+
out.append(clean)
|
|
285
|
+
seen.add(key)
|
|
286
|
+
return out
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _multi_col_header_data_pairs(table: TableData) -> list[tuple[str, str]]:
|
|
290
|
+
"""For a >2-col table with header rows, yield (col_header, data_value) for
|
|
291
|
+
every (column, data row) pair so a label that names a column matches the
|
|
292
|
+
value in that column's data row(s)."""
|
|
293
|
+
|
|
294
|
+
out: list[tuple[str, str]] = []
|
|
295
|
+
rows, cols = table.data.shape
|
|
296
|
+
if cols <= 2 or rows == 0:
|
|
297
|
+
return out
|
|
298
|
+
header_rows = getattr(table, "header_rows", set()) or set()
|
|
299
|
+
n_header = (max(header_rows) + 1) if header_rows else 1
|
|
300
|
+
for col_idx in range(cols):
|
|
301
|
+
header_text = _column_header_for_index(table, col_idx)
|
|
302
|
+
if not header_text:
|
|
303
|
+
continue
|
|
304
|
+
for row_idx in range(n_header, rows):
|
|
305
|
+
cell_value = str(table.data[row_idx, col_idx]).strip()
|
|
306
|
+
out.append((header_text, cell_value))
|
|
307
|
+
return out
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _iter_html_cell_kv_pairs(content: str) -> list[tuple[str, str]]:
|
|
311
|
+
"""Yield (label, value) pairs extracted from HTML cells whose internal
|
|
312
|
+
layout stacks the label above the value via ``<br/>``.
|
|
313
|
+
|
|
314
|
+
Pattern: ``<td>Label<br/><strong>Value</strong></td>``. Common in parsers
|
|
315
|
+
that try to mirror the visual two-line widget within a single cell."""
|
|
316
|
+
|
|
317
|
+
out: list[tuple[str, str]] = []
|
|
318
|
+
if "<table" not in content.lower():
|
|
319
|
+
return out
|
|
320
|
+
soup = BeautifulSoup(content, "lxml")
|
|
321
|
+
for table in soup.find_all("table"):
|
|
322
|
+
for cell in table.find_all(["td", "th"]):
|
|
323
|
+
for br in cell.find_all("br"):
|
|
324
|
+
br.replace_with("\n")
|
|
325
|
+
cell_text = cell.get_text().strip()
|
|
326
|
+
if "\n" not in cell_text:
|
|
327
|
+
continue
|
|
328
|
+
parts = [p.strip() for p in cell_text.split("\n", 1)]
|
|
329
|
+
if len(parts) != 2:
|
|
330
|
+
continue
|
|
331
|
+
label_part, value_part = parts
|
|
332
|
+
label_part = label_part.strip("*_ \t")
|
|
333
|
+
value_part = value_part.strip("*_ \t")
|
|
334
|
+
if label_part:
|
|
335
|
+
out.append((label_part, value_part))
|
|
336
|
+
return out
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _iter_html_table_kv_rows(content: str) -> list[tuple[str, str]]:
|
|
340
|
+
"""Yield (label, value) tuples from HTML tables.
|
|
341
|
+
|
|
342
|
+
- 2-col tables: yield each row as ``(col0, col1)`` (label-then-value layout).
|
|
343
|
+
- >2-col tables with header rows: yield ``(col_header, data_row_value)``
|
|
344
|
+
for every column × data row, so a label naming a column matches the
|
|
345
|
+
value in that column's data row.
|
|
346
|
+
- Any cell that contains in-cell ``<br/>`` separators: yield
|
|
347
|
+
``(top_half, bottom_half)`` so ``<td>Label<br/><strong>Value</strong></td>``
|
|
348
|
+
is captured.
|
|
349
|
+
"""
|
|
350
|
+
|
|
351
|
+
out: list[tuple[str, str]] = []
|
|
352
|
+
if "<table" not in content.lower():
|
|
353
|
+
return out
|
|
354
|
+
# Cell-internal label/value (label<br/>value inside one cell) takes
|
|
355
|
+
# precedence over the row-wise 2-col interpretation; otherwise a cell
|
|
356
|
+
# like ``<td>Last Name<br/>Nguyen</td>`` would be mangled into a single
|
|
357
|
+
# blob ``Last Name Nguyen`` by the row-wise path before the cell-level
|
|
358
|
+
# pair is ever consulted.
|
|
359
|
+
out.extend(_iter_html_cell_kv_pairs(content))
|
|
360
|
+
for table in parse_html_tables(content):
|
|
361
|
+
rows, cols = table.data.shape
|
|
362
|
+
if cols == 2:
|
|
363
|
+
for row_idx in range(rows):
|
|
364
|
+
label_text = str(table.data[row_idx, 0]).strip()
|
|
365
|
+
value_text = str(table.data[row_idx, 1]).strip()
|
|
366
|
+
out.append((label_text, value_text))
|
|
367
|
+
elif cols > 2:
|
|
368
|
+
# Header-then-data-row binding (one record per data row).
|
|
369
|
+
# Interleaved label/value layouts inside wide HTML tables
|
|
370
|
+
# (well-log report headers, rotated form pages) are handled
|
|
371
|
+
# downstream by ``_iter_html_cell_neighbor_pairs`` — that
|
|
372
|
+
# iterator classifies each neighbor as label-shaped vs
|
|
373
|
+
# value-shaped before pairing, so the score path never sees
|
|
374
|
+
# spurious ``(LABEL, OTHER_LABEL)`` candidates.
|
|
375
|
+
out.extend(_multi_col_header_data_pairs(table))
|
|
376
|
+
return out
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
# Heuristic: signals that a cell *looks like* a form-label rather than a value.
|
|
380
|
+
# Used as a tie-breaker by ``_iter_html_cell_neighbor_pairs`` when picking
|
|
381
|
+
# between a right-neighbor and a below-neighbor in wide HTML form tables —
|
|
382
|
+
# we only want to return value-shaped neighbors, not adjacent label cells.
|
|
383
|
+
#
|
|
384
|
+
# A cell is considered label-like when any of these holds:
|
|
385
|
+
# 1. trailing colon (``FILE NO:``);
|
|
386
|
+
# 2. short ALL-CAPS with no digits / no value-style punctuation
|
|
387
|
+
# (``WELL``, ``COMPANY``, ``OTHER SERVICES``);
|
|
388
|
+
# 3. structurally repeats elsewhere in the same table — handled by the
|
|
389
|
+
# caller, which threads the per-table text-count map in.
|
|
390
|
+
#
|
|
391
|
+
# Values like ``LEHMAN #1``, ``42-157-33282``, ``KEBO OIL & GAS, INC.``,
|
|
392
|
+
# ``15-MAY-2023`` keep digits / hashes / commas / parens so they fail
|
|
393
|
+
# heuristic (2) and are correctly classified as value-shaped.
|
|
394
|
+
#
|
|
395
|
+
# Note: ``&`` is intentionally absent from the disqualifier so common
|
|
396
|
+
# value strings like ``"KEBO OIL & GAS, INC."`` (rescued by the comma)
|
|
397
|
+
# stay value-shaped without forcing every label with ``&`` (e.g. an
|
|
398
|
+
# ``"OIL & GAS"`` column header) to be misread as a value. A naked
|
|
399
|
+
# ``"X & Y"`` value with no other punctuation would be misclassified as a
|
|
400
|
+
# label, but that pattern hasn't surfaced in real benchmark data.
|
|
401
|
+
_LABEL_LIKE_DISQUALIFIER_RE = re.compile(r"[\d#@/_,()$%]")
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def _cell_text_is_label_like(text: str) -> bool:
|
|
405
|
+
s = text.strip()
|
|
406
|
+
if not s:
|
|
407
|
+
return False
|
|
408
|
+
if s.endswith(":"):
|
|
409
|
+
return True
|
|
410
|
+
# Length cap: typical form labels are short (1-3 words). Long ALL-CAPS
|
|
411
|
+
# strings like ``"PERMITTED FOR RECOMPLETION TO PRODUCE FROM"`` skip
|
|
412
|
+
# heuristic (2) and stay value-shaped, which is the safer default — the
|
|
413
|
+
# cost of mis-flagging a long label is a missed neighbor, but the cost
|
|
414
|
+
# of flagging a long value is returning the wrong neighbor.
|
|
415
|
+
if len(s) > 30:
|
|
416
|
+
return False
|
|
417
|
+
if _LABEL_LIKE_DISQUALIFIER_RE.search(s):
|
|
418
|
+
return False
|
|
419
|
+
if s != s.upper():
|
|
420
|
+
return False
|
|
421
|
+
if not re.search(r"[A-Z]", s):
|
|
422
|
+
return False
|
|
423
|
+
return True
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _iter_md_table_kv_rows(content: str) -> list[tuple[str, str]]:
|
|
427
|
+
"""Yield (label, value) tuples from markdown tables.
|
|
428
|
+
|
|
429
|
+
- 2-col tables: yield each row as ``(col0, col1)`` (existing behavior).
|
|
430
|
+
- >2-col tables: yield ``(col_header, data_row_value)`` for every
|
|
431
|
+
column × data row.
|
|
432
|
+
"""
|
|
433
|
+
|
|
434
|
+
out: list[tuple[str, str]] = []
|
|
435
|
+
for table in parse_markdown_tables(content):
|
|
436
|
+
rows, cols = table.data.shape
|
|
437
|
+
if cols == 2:
|
|
438
|
+
for row_idx in range(rows):
|
|
439
|
+
label_text = str(table.data[row_idx, 0]).strip()
|
|
440
|
+
value_text = str(table.data[row_idx, 1]).strip()
|
|
441
|
+
out.append((label_text, value_text))
|
|
442
|
+
elif cols > 2:
|
|
443
|
+
out.extend(_multi_col_header_data_pairs(table))
|
|
444
|
+
return out
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
# Generic HTML tag stripper for label/value normalization. Form-field values
|
|
448
|
+
# never legitimately contain ``<tag>...</tag>`` markup — names, addresses, IDs,
|
|
449
|
+
# and currency don't — but parsers leak HTML wrappers into extracted spans
|
|
450
|
+
# (haiku preserves ``<strong>``/``<td>``, gemini emits ``<u>`` underline-fill,
|
|
451
|
+
# OpenAI sometimes leaves ``</p>``). The pattern is restricted to well-formed
|
|
452
|
+
# HTML element opens/closes: a tag name must start with an ASCII letter and
|
|
453
|
+
# contain only alphanumerics afterwards, optionally followed by a
|
|
454
|
+
# whitespace-introduced attribute run. This deliberately excludes markdown
|
|
455
|
+
# email/URL autolinks like ``<wei.lin@host.com>`` and ``<https://...>``,
|
|
456
|
+
# whose first character after ``<`` is a letter but whose body contains
|
|
457
|
+
# ``.``/``@``/``:`` that disqualify them from the tag-name shape.
|
|
458
|
+
_HTML_TAG_RE = re.compile(r"<\s*/?\s*[a-zA-Z][a-zA-Z0-9]*(?:\s[^<>]*)?\s*/?\s*>")
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _strip_html_tags(s: str) -> str:
|
|
462
|
+
return _HTML_TAG_RE.sub("", s)
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
# Tagged-line prefix used by some parsers to mark a field-extraction event,
|
|
466
|
+
# e.g. ``[FORM FIELD] Label: value``. The bracketed prefix is parser noise,
|
|
467
|
+
# not part of the label. We only strip it from the start of a candidate
|
|
468
|
+
# label, never mid-string, so legitimate labels containing brackets like
|
|
469
|
+
# ``[Effective Date]`` (uncommon but possible) are preserved unless the
|
|
470
|
+
# bracket is the leading token.
|
|
471
|
+
_LABEL_TAG_PREFIX_RE = re.compile(r"^\s*\[[^\]\n]+\]\s+")
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def _trim_value_at_next_field(value: str) -> str:
|
|
475
|
+
"""Trim a captured value at a ``| Next Label: ...`` boundary.
|
|
476
|
+
|
|
477
|
+
Some parsers concatenate multiple labelled fields onto one line with
|
|
478
|
+
``|`` separators (e.g. ``Date: 2026-04-27 | Borrower's Name: Maya | ...``).
|
|
479
|
+
Without this trim, the plain-colon regex captures the entire tail as the
|
|
480
|
+
value of the first field. We only split when the part after the ``|``
|
|
481
|
+
looks like another labelled field (contains ``:``), so legitimate values
|
|
482
|
+
with embedded ``|`` (rare in form data) are preserved.
|
|
483
|
+
"""
|
|
484
|
+
|
|
485
|
+
parts = re.split(r"\s+\|\s+", value, maxsplit=1)
|
|
486
|
+
if len(parts) == 2 and ":" in parts[1]:
|
|
487
|
+
return parts[0].strip()
|
|
488
|
+
return value
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def _split_pipe_concatenated_pairs(value: str) -> list[tuple[str, str]]:
|
|
492
|
+
"""Split a run-on ``Label1: v1 | Label2: v2 | ...`` value tail into pairs.
|
|
493
|
+
|
|
494
|
+
Companion to :func:`_trim_value_at_next_field`. The first call trims the
|
|
495
|
+
value of the *initial* labelled field; this function recovers any
|
|
496
|
+
*subsequent* ``Label: value`` pairs that were riding along on the same
|
|
497
|
+
line so a single-line run-on yields one pair per labelled field.
|
|
498
|
+
"""
|
|
499
|
+
|
|
500
|
+
out: list[tuple[str, str]] = []
|
|
501
|
+
if " | " not in value:
|
|
502
|
+
return out
|
|
503
|
+
for segment in re.split(r"\s+\|\s+", value):
|
|
504
|
+
if ":" not in segment:
|
|
505
|
+
continue
|
|
506
|
+
# Same horizontal-only colon split as _PLAIN_COLON_RE so we don't
|
|
507
|
+
# accidentally bleed time-of-day strings ("11:30 AM") into pairs.
|
|
508
|
+
m = re.match(r"^[ \t]*([^:\n*][^:\n]{0,200}?)[ \t]*:[ \t]*(.+?)[ \t]*$", segment)
|
|
509
|
+
if not m:
|
|
510
|
+
continue
|
|
511
|
+
seg_label = _strip_html_tags(m.group(1).strip()).strip()
|
|
512
|
+
seg_label = _LABEL_TAG_PREFIX_RE.sub("", seg_label).strip()
|
|
513
|
+
seg_value = _strip_html_tags(m.group(2).strip()).strip()
|
|
514
|
+
if seg_label:
|
|
515
|
+
out.append((seg_label, seg_value))
|
|
516
|
+
return out
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
# Bullet-line shape for safe aggregation: ``- item`` / ``* item`` / ``+ item``
|
|
520
|
+
# (with optional leading ``\`` escape some renderers emit). The negative
|
|
521
|
+
# lookahead rejects checkbox-bearing bullets (``- [x] ...``) — those rows
|
|
522
|
+
# describe their own state, not a continuation of the preceding label.
|
|
523
|
+
_AGGREGATE_BULLET_RE = re.compile(r"^\\?[-*+]\s+(?!\\?\[)")
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
def _aggregate_following_lines(content: str, after_offset: int, max_lines: int = 8) -> str:
|
|
527
|
+
"""Collect bullet-list lines after *after_offset* into a single value
|
|
528
|
+
string, joined with ``, ``.
|
|
529
|
+
|
|
530
|
+
This is a narrow fallback for the audit-A3 pattern: a bold-colon header
|
|
531
|
+
with an empty inline value followed by a multi-line address laid out as
|
|
532
|
+
bullets (HUD voucher ``Mail Payments To`` blocks, etc.). Strict gating
|
|
533
|
+
keeps it from pulling unrelated form structure into the value:
|
|
534
|
+
|
|
535
|
+
1. Every line must be a clean bullet (``-``/``*``/``+`` with no
|
|
536
|
+
``[x]``/``[ ]`` checkbox marker — those rows belong to a different
|
|
537
|
+
field).
|
|
538
|
+
2. No line may carry any checkbox glyph or ASCII bracket marker.
|
|
539
|
+
3. At least 2 collected bullets are required. A single bullet is too
|
|
540
|
+
ambiguous to attribute as the value — leaving the value empty is
|
|
541
|
+
safer than risking a wrong attribution.
|
|
542
|
+
4. Stops at blank line, ATX heading, HTML boundary, or another bold-
|
|
543
|
+
colon header. Returns ``""`` if any constraint fails so the caller
|
|
544
|
+
falls back to the normal empty-value path.
|
|
545
|
+
"""
|
|
546
|
+
|
|
547
|
+
tail = content[after_offset:]
|
|
548
|
+
lines = tail.splitlines()
|
|
549
|
+
# Skip the line containing the header itself (we matched into it).
|
|
550
|
+
start_idx = 1 if lines else 0
|
|
551
|
+
collected: list[str] = []
|
|
552
|
+
for raw in lines[start_idx : start_idx + max_lines]:
|
|
553
|
+
stripped = raw.strip()
|
|
554
|
+
if not stripped:
|
|
555
|
+
break
|
|
556
|
+
if stripped.startswith(("#", ">", "|", "<")):
|
|
557
|
+
break
|
|
558
|
+
# Stop at the start of a new bold-colon header.
|
|
559
|
+
if "**" in stripped and ":" in stripped:
|
|
560
|
+
break
|
|
561
|
+
if not _AGGREGATE_BULLET_RE.match(stripped):
|
|
562
|
+
return ""
|
|
563
|
+
if _MARKER_RE.search(stripped):
|
|
564
|
+
return ""
|
|
565
|
+
cleaned = re.sub(r"^\\?[-*+]\s+", "", stripped).strip()
|
|
566
|
+
cleaned = _strip_html_tags(cleaned).strip()
|
|
567
|
+
if cleaned:
|
|
568
|
+
collected.append(cleaned)
|
|
569
|
+
if len(collected) < 2:
|
|
570
|
+
return ""
|
|
571
|
+
return ", ".join(collected)
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
# Bold-colon pattern. Matches **Label:** value and **Label**: value, allowing
|
|
575
|
+
# multiple pairs on a single line. All inter-token whitespace is restricted
|
|
576
|
+
# to horizontal whitespace ([ \t]) so a match cannot span blank lines or
|
|
577
|
+
# headings — without this, an empty "**Label**:\n\n# Heading\n\n**Other**:"
|
|
578
|
+
# would attribute the heading text to Label as the value.
|
|
579
|
+
_BOLD_COLON_RE = re.compile(
|
|
580
|
+
r"\*\*[ \t]*([^*\n]+?)[ \t]*\*\*[ \t]*:?[ \t]*([^\n*]*?)(?=[ \t]*\*\*|$)",
|
|
581
|
+
re.MULTILINE,
|
|
582
|
+
)
|
|
583
|
+
_EXPLICIT_BOLD_COLON_RE = re.compile(
|
|
584
|
+
r"\*\*[ \t]*([^*\n]+?)[ \t]*(?:[ \t]*:[ \t]*\*\*[ \t]*|\*\*[ \t]*:[ \t]*)([^\n*]*?)(?=[ \t]*\*\*|$)",
|
|
585
|
+
re.MULTILINE,
|
|
586
|
+
)
|
|
587
|
+
_ANY_BOLD_COLON_ON_LINE_RE = re.compile(r"\*\*[ \t]*[^*\n]+?[ \t]*\*\*[ \t]*:")
|
|
588
|
+
_ANY_BOLD_INTERNAL_COLON_ON_LINE_RE = re.compile(r"\*\*[ \t]*[^*\n]+?:[ \t]*\*\*")
|
|
589
|
+
_BOLD_VALUE_RE = re.compile(r"\*\*[ \t]*([^*\n]+?)[ \t]*\*\*")
|
|
590
|
+
_LIST_LINE_RE = re.compile(r"^\s*\\?[-*+]\s+")
|
|
591
|
+
|
|
592
|
+
|
|
593
|
+
def _has_bold_colon_label(content: str) -> bool:
|
|
594
|
+
"""Return true if any line contains an explicit bold label marker."""
|
|
595
|
+
|
|
596
|
+
return bool(_ANY_BOLD_COLON_ON_LINE_RE.search(content) or _ANY_BOLD_INTERNAL_COLON_ON_LINE_RE.search(content))
|
|
597
|
+
|
|
598
|
+
|
|
599
|
+
# Connector words that the parser sometimes wraps in bold inside a numeric
|
|
600
|
+
# range, e.g. ``**Depth Drilled**: 105 **to**: 15437`` or
|
|
601
|
+
# ``Temperature: 32 **to** 100 F``. Without special-casing, the bold-colon
|
|
602
|
+
# value regex stops at the connector's leading ``**`` and only captures the
|
|
603
|
+
# left half. We re-join the trailing value when the bold span between two
|
|
604
|
+
# value chunks is one of these connectors. The connectors are matched whole-
|
|
605
|
+
# word, case-insensitively. Allows leading horizontal whitespace so the
|
|
606
|
+
# splice cursor doesn't have to land exactly on the ``**``.
|
|
607
|
+
_BOLD_CONNECTOR_RE = re.compile(
|
|
608
|
+
r"[ \t]*\*\*[ \t]*(to|and|or|&|thru|through|until)[ \t]*\*\*[ \t]*:?[ \t]*([^\n*]*?)"
|
|
609
|
+
r"(?=[ \t]*\*\*|$)",
|
|
610
|
+
re.IGNORECASE | re.MULTILINE,
|
|
611
|
+
)
|
|
612
|
+
|
|
613
|
+
|
|
614
|
+
def _extend_value_across_bold_connectors(content: str, value_end_offset: int, base_value: str) -> str:
|
|
615
|
+
"""Re-join a bold-colon value that was clipped at a bold connector token.
|
|
616
|
+
|
|
617
|
+
The bold-colon regex terminates the value at the next ``**``. When the
|
|
618
|
+
next bold span is a connector word (``to``, ``and``, ...), the value
|
|
619
|
+
actually continues across it. This helper looks at the content
|
|
620
|
+
immediately following the captured value and, while it sees a bold
|
|
621
|
+
connector followed by more inline content, splices everything into a
|
|
622
|
+
single value string.
|
|
623
|
+
|
|
624
|
+
Stops as soon as the next bold span is anything other than a recognized
|
|
625
|
+
connector — that's a real label boundary, not a continuation.
|
|
626
|
+
"""
|
|
627
|
+
|
|
628
|
+
if not base_value:
|
|
629
|
+
return base_value
|
|
630
|
+
cursor = value_end_offset
|
|
631
|
+
joined = base_value
|
|
632
|
+
while True:
|
|
633
|
+
match = _BOLD_CONNECTOR_RE.match(content, cursor)
|
|
634
|
+
if not match:
|
|
635
|
+
break
|
|
636
|
+
connector = match.group(1)
|
|
637
|
+
extra = match.group(2).strip()
|
|
638
|
+
joined = f"{joined} {connector} {extra}".strip()
|
|
639
|
+
cursor = match.end()
|
|
640
|
+
return joined
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
def _iter_bold_colon_pairs_from_regex(content: str, pattern: re.Pattern[str]) -> list[tuple[str, str]]:
|
|
644
|
+
"""Yield every (label, value) pair surfaced via bold-colon syntax.
|
|
645
|
+
|
|
646
|
+
Generic post-processing applied to every yielded pair: HTML tags
|
|
647
|
+
stripped from both label and value, leading ``[tag]`` prefix removed
|
|
648
|
+
from the label, ``| Next Label:`` boundary trimmed from the value, and
|
|
649
|
+
when the inline value is empty, the next few non-blank list/text lines
|
|
650
|
+
are aggregated into the value (multi-line address pattern).
|
|
651
|
+
"""
|
|
652
|
+
|
|
653
|
+
out: list[tuple[str, str]] = []
|
|
654
|
+
for match in pattern.finditer(content):
|
|
655
|
+
cand_label = match.group(1).strip(": ").strip()
|
|
656
|
+
# Strip trailing markdown line-continuation backslash before whitespace.
|
|
657
|
+
# Some parsers emit ``**Label**: \`` for empty fields; without this
|
|
658
|
+
# strip the value would be ``"\\"``, never matching empty expected.
|
|
659
|
+
raw_value = match.group(2).strip().rstrip("\\").strip()
|
|
660
|
+
# Splice bold connectors (``**to**``, ``**and**``) back into the value
|
|
661
|
+
# so numeric ranges like ``**Depth Drilled**: 105 **to** 15437`` aren't
|
|
662
|
+
# truncated at the connector.
|
|
663
|
+
raw_value = _extend_value_across_bold_connectors(content, match.end(), raw_value)
|
|
664
|
+
cand_label = _strip_html_tags(cand_label).strip()
|
|
665
|
+
cand_label = _LABEL_TAG_PREFIX_RE.sub("", cand_label).strip()
|
|
666
|
+
cand_value = _strip_html_tags(raw_value).strip()
|
|
667
|
+
cand_value = _trim_value_at_next_field(cand_value)
|
|
668
|
+
if not cand_value:
|
|
669
|
+
cand_value = _aggregate_following_lines(content, match.end())
|
|
670
|
+
if cand_label:
|
|
671
|
+
out.append((cand_label, cand_value))
|
|
672
|
+
# Recover any sibling pipe-concatenated pairs riding the same line.
|
|
673
|
+
out.extend(_split_pipe_concatenated_pairs(raw_value))
|
|
674
|
+
return out
|
|
675
|
+
|
|
676
|
+
|
|
677
|
+
def _iter_bold_colon_pairs(content: str) -> list[tuple[str, str]]:
|
|
678
|
+
"""Yield every (label, value) pair surfaced via bold-colon syntax.
|
|
679
|
+
|
|
680
|
+
Kept intentionally backward-compatible with older generated rules: this
|
|
681
|
+
accepts both ``**Label:** value`` / ``**Label**: value`` and the historical
|
|
682
|
+
no-colon form. New rule generation should use
|
|
683
|
+
:func:`_iter_explicit_bold_colon_pairs` so bold emphasis on values is not
|
|
684
|
+
mistaken for a label.
|
|
685
|
+
"""
|
|
686
|
+
|
|
687
|
+
return _iter_bold_colon_pairs_from_regex(content, _BOLD_COLON_RE)
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def _iter_explicit_bold_colon_pairs(content: str) -> list[tuple[str, str]]:
|
|
691
|
+
"""Yield only explicit ``**Label:** value`` / ``**Label**: value`` pairs."""
|
|
692
|
+
|
|
693
|
+
return _iter_bold_colon_pairs_from_regex(content, _EXPLICIT_BOLD_COLON_RE)
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
# Plain-colon pattern. Inter-token whitespace is restricted to horizontal
|
|
697
|
+
# whitespace ([ \t]) so a colon at end-of-line cannot consume the next line as
|
|
698
|
+
# the value (parallel to the bold-colon regex; same blank-line crossing bug).
|
|
699
|
+
_PLAIN_COLON_RE = re.compile(r"^[ \t]*([^:\n*][^:\n]{0,200}?)[ \t]*:[ \t]*(.+?)[ \t]*$", re.MULTILINE)
|
|
700
|
+
_LIST_MARKER_RE = re.compile(r"^\\?[-*+]\s+")
|
|
701
|
+
|
|
702
|
+
# Underscore blank field: ``Processor's Name _________________``. The label
|
|
703
|
+
# sits before a run of three or more underscores acting as a fill-in line for
|
|
704
|
+
# an empty field. No colon, no bold, just a label-then-underscore-blank.
|
|
705
|
+
_UNDERSCORE_BLANK_RE = re.compile(r"^\s*([^_\n]+?)\s+_{3,}\s*$", re.MULTILINE)
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
def _iter_plain_colon_pairs(content: str) -> list[tuple[str, str]]:
|
|
709
|
+
"""Yield (label, value) pairs from `Label: value` lines (plain text).
|
|
710
|
+
|
|
711
|
+
Plain bullet items with the ``Label: value`` shape (``- Defendant: Devon``)
|
|
712
|
+
are stripped of their leading marker and yielded — markdown task lists
|
|
713
|
+
(``- [x] Foo``) are still skipped because they are handled by the checkbox
|
|
714
|
+
scanners. Headings, fenced code, blockquotes, and bold-formatted lines
|
|
715
|
+
are skipped here too.
|
|
716
|
+
|
|
717
|
+
Generic post-processing on every yielded pair: HTML tags stripped from
|
|
718
|
+
both label and value, leading ``[tag]`` prefix removed from the label,
|
|
719
|
+
and the value trimmed at any ``| Next Label:`` boundary so a single line
|
|
720
|
+
like ``A: x | B: y`` yields two pairs instead of one with a run-on
|
|
721
|
+
value.
|
|
722
|
+
"""
|
|
723
|
+
|
|
724
|
+
out: list[tuple[str, str]] = []
|
|
725
|
+
for match in _PLAIN_COLON_RE.finditer(content):
|
|
726
|
+
cand_label = match.group(1).strip()
|
|
727
|
+
raw_value = match.group(2).strip()
|
|
728
|
+
# Skip multi-cell HTML table rows: a single line that opens more than
|
|
729
|
+
# one ``<th>`` / ``<td>`` is a wide table row, not a single
|
|
730
|
+
# ``label: value`` line. Without this guard
|
|
731
|
+
# ``<tr><th>API NO:</th><th>WELL</th><th>LEHMAN #1</th></tr>`` matches
|
|
732
|
+
# the plain-colon regex and yields ``("API NO", "WELLLEHMAN #1")``
|
|
733
|
+
# because HTML-tag stripping collapses adjacent cells into a single
|
|
734
|
+
# value run. Single-cell rows (``<th>Company: CIMARRON ...</th>``)
|
|
735
|
+
# carry exactly one inline KV pair and stay on this path — wide HTML
|
|
736
|
+
# form tables are handled by ``_iter_html_table_kv_rows`` and
|
|
737
|
+
# ``_iter_html_cell_neighbor_pairs``.
|
|
738
|
+
raw_line = match.group(0)
|
|
739
|
+
if len(re.findall(r"<t[hd]\b", raw_line)) > 1:
|
|
740
|
+
continue
|
|
741
|
+
if cand_label.startswith(("#", "`", ">")):
|
|
742
|
+
continue
|
|
743
|
+
if "**" in cand_label:
|
|
744
|
+
continue
|
|
745
|
+
if cand_label.startswith(("\\-", "-", "*", "+")):
|
|
746
|
+
stripped = _LIST_MARKER_RE.sub("", cand_label).strip()
|
|
747
|
+
# Tasklist-shaped bullets (``[x] ...``) belong to the checkbox path.
|
|
748
|
+
if stripped.startswith(("\\[", "[")):
|
|
749
|
+
continue
|
|
750
|
+
if not stripped:
|
|
751
|
+
continue
|
|
752
|
+
cand_label = stripped
|
|
753
|
+
cand_label = _strip_html_tags(cand_label).strip()
|
|
754
|
+
cand_label = _LABEL_TAG_PREFIX_RE.sub("", cand_label).strip()
|
|
755
|
+
cand_value = _strip_html_tags(raw_value).strip()
|
|
756
|
+
cand_value = _trim_value_at_next_field(cand_value)
|
|
757
|
+
if cand_label:
|
|
758
|
+
out.append((cand_label, cand_value))
|
|
759
|
+
# Recover any sibling pipe-concatenated pairs riding the same line.
|
|
760
|
+
out.extend(_split_pipe_concatenated_pairs(raw_value))
|
|
761
|
+
return out
|
|
762
|
+
|
|
763
|
+
|
|
764
|
+
_LABEL_BEFORE_BOLD_KNOWN_SUFFIXES = (
|
|
765
|
+
"Name of Field in which well is located",
|
|
766
|
+
"Date well was plugged",
|
|
767
|
+
"Name of Company or Operator",
|
|
768
|
+
"Name of Farm or Lease",
|
|
769
|
+
"Name of Party Plugging Well",
|
|
770
|
+
"Name of Lease",
|
|
771
|
+
"No. of Acres",
|
|
772
|
+
"Well No.",
|
|
773
|
+
"Sec. No.",
|
|
774
|
+
"Blk No.",
|
|
775
|
+
"Company",
|
|
776
|
+
"Address",
|
|
777
|
+
"Survey",
|
|
778
|
+
"County",
|
|
779
|
+
"Has this well ever produced oil or gas?",
|
|
780
|
+
"Located",
|
|
781
|
+
"Oil",
|
|
782
|
+
"Gas",
|
|
783
|
+
"Dry",
|
|
784
|
+
"Total Depth",
|
|
785
|
+
"Top of each producing sand",
|
|
786
|
+
"Name",
|
|
787
|
+
"Title",
|
|
788
|
+
)
|
|
789
|
+
_LABEL_BEFORE_BOLD_REJECT_PREFIXES = (
|
|
790
|
+
"i,",
|
|
791
|
+
"i ",
|
|
792
|
+
"if you answered",
|
|
793
|
+
"subscribed ",
|
|
794
|
+
"being ",
|
|
795
|
+
)
|
|
796
|
+
_LABEL_BEFORE_BOLD_REJECT_SUFFIXES = (
|
|
797
|
+
" and",
|
|
798
|
+
" at",
|
|
799
|
+
" by",
|
|
800
|
+
" for",
|
|
801
|
+
" from",
|
|
802
|
+
" in",
|
|
803
|
+
" of",
|
|
804
|
+
" on",
|
|
805
|
+
" or",
|
|
806
|
+
" to",
|
|
807
|
+
)
|
|
808
|
+
|
|
809
|
+
|
|
810
|
+
def _clean_label_before_bold(raw: str) -> str:
|
|
811
|
+
"""Normalize the label chunk immediately before a bold value span."""
|
|
812
|
+
|
|
813
|
+
label = raw
|
|
814
|
+
# Empty fill lines often sit between two fields:
|
|
815
|
+
# ``Blk No. ______ Survey **T.E.&L.**``. The label for the bold value is the
|
|
816
|
+
# suffix after the blank, not the earlier empty field.
|
|
817
|
+
label = re.split(r"(?:\\?_){3,}", label)[-1]
|
|
818
|
+
label = re.sub(r"<br\s*/?>", " ", label, flags=re.IGNORECASE)
|
|
819
|
+
label = re.sub(r"^[\s\\*_#>\-+|]+", "", label)
|
|
820
|
+
label = _LABEL_TAG_PREFIX_RE.sub("", label)
|
|
821
|
+
label = _strip_html_tags(label).strip()
|
|
822
|
+
label = label.strip("*_ \t:-,;")
|
|
823
|
+
if ":" in label:
|
|
824
|
+
label = label.rsplit(":", 1)[-1].strip()
|
|
825
|
+
|
|
826
|
+
lowered = label.lower()
|
|
827
|
+
best: str | None = None
|
|
828
|
+
for suffix in _LABEL_BEFORE_BOLD_KNOWN_SUFFIXES:
|
|
829
|
+
idx = lowered.rfind(suffix.lower())
|
|
830
|
+
if idx < 0:
|
|
831
|
+
continue
|
|
832
|
+
if lowered[idx:].strip() != suffix.lower():
|
|
833
|
+
continue
|
|
834
|
+
if best is None or len(suffix) > len(best):
|
|
835
|
+
best = suffix
|
|
836
|
+
if best is not None:
|
|
837
|
+
label = label[-len(best) :].strip()
|
|
838
|
+
|
|
839
|
+
return label.strip("*_ \t:-,;")
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
def _looks_like_label_before_bold(label: str) -> bool:
|
|
843
|
+
if not label or len(label) > 100:
|
|
844
|
+
return False
|
|
845
|
+
lowered = label.lower()
|
|
846
|
+
if lowered.startswith(_LABEL_BEFORE_BOLD_REJECT_PREFIXES):
|
|
847
|
+
return False
|
|
848
|
+
if not any(ch.isalpha() for ch in label):
|
|
849
|
+
return False
|
|
850
|
+
if ")" in label and "(" not in label:
|
|
851
|
+
return False
|
|
852
|
+
if len(label) <= 2:
|
|
853
|
+
return False
|
|
854
|
+
if lowered.endswith(_LABEL_BEFORE_BOLD_REJECT_SUFFIXES):
|
|
855
|
+
return False
|
|
856
|
+
if re.fullmatch(r"(?:19|20)?\d{1,2}", label):
|
|
857
|
+
return False
|
|
858
|
+
first_alpha = next((ch for ch in label if ch.isalpha()), "")
|
|
859
|
+
if first_alpha and first_alpha.islower():
|
|
860
|
+
return False
|
|
861
|
+
return True
|
|
862
|
+
|
|
863
|
+
|
|
864
|
+
def _extend_inline_year_suffix(
|
|
865
|
+
value: str,
|
|
866
|
+
between_this_and_next: str,
|
|
867
|
+
next_match: re.Match[str] | None,
|
|
868
|
+
) -> str:
|
|
869
|
+
"""Join ``**June 17,** 194 **3**`` into ``June 17, 1943``."""
|
|
870
|
+
|
|
871
|
+
if next_match is None:
|
|
872
|
+
return value
|
|
873
|
+
year_prefix = re.sub(r"\s+", "", between_this_and_next)
|
|
874
|
+
if not re.fullmatch(r"(?:19|20)?\d{0,2}", year_prefix):
|
|
875
|
+
return value
|
|
876
|
+
suffix = next_match.group(1).strip()
|
|
877
|
+
if not re.fullmatch(r"\d{1,2}", suffix):
|
|
878
|
+
return value
|
|
879
|
+
year = f"{year_prefix}{suffix}"
|
|
880
|
+
if not re.fullmatch(r"(?:19|20)\d{2}", year):
|
|
881
|
+
return value
|
|
882
|
+
return re.sub(r"\s+", " ", f"{value} {year}").strip()
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
def _iter_label_before_bold_value_pairs(content: str) -> list[tuple[str, str]]:
|
|
886
|
+
"""Yield ``(label, value)`` for visual rows shaped as ``Label **Value**``.
|
|
887
|
+
|
|
888
|
+
Gemini-style parse output often preserves the printed form row as
|
|
889
|
+
``Well No. **Z-5** Name of Lease **W. H. Portwood**`` rather than normalizing
|
|
890
|
+
it to ``**Well No.**: Z-5``. This matcher treats the text immediately before
|
|
891
|
+
each bold span as the label and the bold span as the value. It is deliberately
|
|
892
|
+
lower priority than explicit colon/table sources.
|
|
893
|
+
"""
|
|
894
|
+
|
|
895
|
+
out: list[tuple[str, str]] = []
|
|
896
|
+
for line in content.splitlines():
|
|
897
|
+
if "**" not in line:
|
|
898
|
+
continue
|
|
899
|
+
stripped = line.strip()
|
|
900
|
+
if not stripped or stripped.startswith(("#", ">", "|", "[Figure")):
|
|
901
|
+
continue
|
|
902
|
+
if re.search(r"\binstructions?\b", stripped, flags=re.IGNORECASE):
|
|
903
|
+
continue
|
|
904
|
+
if _LIST_LINE_RE.match(line):
|
|
905
|
+
continue
|
|
906
|
+
if _has_bold_colon_label(line):
|
|
907
|
+
continue
|
|
908
|
+
matches = list(_BOLD_VALUE_RE.finditer(line))
|
|
909
|
+
if not matches:
|
|
910
|
+
continue
|
|
911
|
+
prev_end = 0
|
|
912
|
+
for idx, match in enumerate(matches):
|
|
913
|
+
label = _clean_label_before_bold(line[prev_end : match.start()])
|
|
914
|
+
value = _strip_html_tags(match.group(1)).strip()
|
|
915
|
+
next_match = matches[idx + 1] if idx + 1 < len(matches) else None
|
|
916
|
+
next_start = next_match.start() if next_match is not None else len(line)
|
|
917
|
+
value = _extend_inline_year_suffix(value, line[match.end() : next_start], next_match)
|
|
918
|
+
if _looks_like_label_before_bold(label) and value:
|
|
919
|
+
out.append((label, value))
|
|
920
|
+
prev_end = match.end()
|
|
921
|
+
return out
|
|
922
|
+
|
|
923
|
+
|
|
924
|
+
# Underline fill-in pattern: parsers that preserve the form's "fill in the
|
|
925
|
+
# blank" layout emit the filled value wrapped in ``<u>...</u>`` tags inline
|
|
926
|
+
# in the surrounding prose, e.g.
|
|
927
|
+
#
|
|
928
|
+
# **2. PROPERTY:** Lot <u>12</u>, Block <u>C</u>, City of <u>Austin</u>...
|
|
929
|
+
#
|
|
930
|
+
# The label sits immediately before the underline span, terminated by a
|
|
931
|
+
# punctuation/whitespace boundary on its left side. We yield (label, value)
|
|
932
|
+
# for each such span so the standard ``_label_matches`` fuzzy-matcher can
|
|
933
|
+
# bridge GT labels like "Block" or "City of (Street Address and City)".
|
|
934
|
+
_UNDERLINE_FILL_RE = re.compile(r"<u>([^<\n]+)</u>")
|
|
935
|
+
_LABEL_LEFT_TERMINATORS = ".,;:()\n>"
|
|
936
|
+
|
|
937
|
+
|
|
938
|
+
def _iter_underline_fill_pairs(content: str) -> list[tuple[str, str]]:
|
|
939
|
+
"""Yield (preceding_label, underlined_value) pairs from ``<u>...</u>`` runs."""
|
|
940
|
+
|
|
941
|
+
out: list[tuple[str, str]] = []
|
|
942
|
+
for match in _UNDERLINE_FILL_RE.finditer(content):
|
|
943
|
+
value = match.group(1).strip()
|
|
944
|
+
if not value:
|
|
945
|
+
continue
|
|
946
|
+
before = content[max(0, match.start() - 100) : match.start()]
|
|
947
|
+
# Walk backward to the nearest sentence/clause terminator. Anything
|
|
948
|
+
# left of that terminator belongs to a different label (or to a
|
|
949
|
+
# heading/inline header), so we stop there.
|
|
950
|
+
cut = -1
|
|
951
|
+
for ch in _LABEL_LEFT_TERMINATORS:
|
|
952
|
+
cut = max(cut, before.rfind(ch))
|
|
953
|
+
label_chunk = before[cut + 1 :]
|
|
954
|
+
# Strip markdown noise: leading bullet, bold/italic markers, stray
|
|
955
|
+
# backslashes, and trailing whitespace. The bracketed-tag prefix
|
|
956
|
+
# (``[FORM FIELD] ``) is dropped here too so it never bleeds into
|
|
957
|
+
# candidate labels.
|
|
958
|
+
label_chunk = re.sub(r"^[\s\\*_#>\-]+", "", label_chunk)
|
|
959
|
+
label_chunk = _LABEL_TAG_PREFIX_RE.sub("", label_chunk)
|
|
960
|
+
label_chunk = _strip_html_tags(label_chunk).strip()
|
|
961
|
+
label_chunk = label_chunk.strip("*_ \t").strip()
|
|
962
|
+
if not label_chunk:
|
|
963
|
+
continue
|
|
964
|
+
# Only the trailing 1-6 words can plausibly be the label — the rest
|
|
965
|
+
# is sentence context.
|
|
966
|
+
words = label_chunk.split()
|
|
967
|
+
if not words:
|
|
968
|
+
continue
|
|
969
|
+
label = " ".join(words[-6:])
|
|
970
|
+
if label:
|
|
971
|
+
out.append((label, value))
|
|
972
|
+
return out
|
|
973
|
+
|
|
974
|
+
|
|
975
|
+
def _iter_underscore_blank_pairs(content: str) -> list[tuple[str, str]]:
|
|
976
|
+
"""Yield (label, "") pairs for ``Label ____`` underscore-blank fields."""
|
|
977
|
+
|
|
978
|
+
out: list[tuple[str, str]] = []
|
|
979
|
+
for match in _UNDERSCORE_BLANK_RE.finditer(content):
|
|
980
|
+
cand_label = match.group(1).strip()
|
|
981
|
+
if not cand_label:
|
|
982
|
+
continue
|
|
983
|
+
if cand_label.startswith(("#", "-", "*", "`", ">", "|")):
|
|
984
|
+
continue
|
|
985
|
+
if "**" in cand_label or ":" in cand_label:
|
|
986
|
+
continue
|
|
987
|
+
out.append((cand_label, ""))
|
|
988
|
+
return out
|
|
989
|
+
|
|
990
|
+
|
|
991
|
+
# Italic line shape: ``*Plaintiff*`` or ``_Address_`` (optionally with a
|
|
992
|
+
# trailing space + ``)`` from court-form layouts like ``*Plaintiff* )``).
|
|
993
|
+
_ITALIC_LABEL_LINE_RE = re.compile(r"^\s*([*_])\s*(\S.*?\S)\s*\1[\s)\\]*$")
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
def _find_text_value_adjacent_line(
|
|
997
|
+
content: str,
|
|
998
|
+
label: str,
|
|
999
|
+
label_max_diffs: int | float = 0,
|
|
1000
|
+
) -> tuple[bool, str | None]:
|
|
1001
|
+
"""Fallback for label-on-its-own-line layouts adjacent to an unlabelled value.
|
|
1002
|
+
|
|
1003
|
+
Two layouts share this scanner:
|
|
1004
|
+
|
|
1005
|
+
- **Italic caption below value** (federal court forms — AO398):
|
|
1006
|
+
``Anthony Cole Jackson )\\n*Plaintiff* )``. The label sits italicized on
|
|
1007
|
+
the line below the value.
|
|
1008
|
+
- **Numbered/heading-style label above value** (UCC5, gemini sub-sections):
|
|
1009
|
+
``1a. INITIAL FINANCING STATEMENT FILE NUMBER\\nOR-UCC-2025-00532600``.
|
|
1010
|
+
The label is its own line above the value.
|
|
1011
|
+
|
|
1012
|
+
Conservative heuristic: only fires for short label-shaped lines (≤ 80
|
|
1013
|
+
chars after stripping markers, no ``:`` and no ``**``) and only on a
|
|
1014
|
+
*strict* ratio match (≥ ``CELL_FUZZY_MATCH_THRESHOLD``). This means a
|
|
1015
|
+
long paragraph that *contains* the label as a substring is **not**
|
|
1016
|
+
treated as the label line — partial-ratio matching is reserved for the
|
|
1017
|
+
other (label-then-value) scanners.
|
|
1018
|
+
|
|
1019
|
+
Direction: italic line → look ABOVE first (caption convention); plain
|
|
1020
|
+
line → look BELOW first (label-then-value convention). Whichever
|
|
1021
|
+
direction lands a non-blank line wins.
|
|
1022
|
+
"""
|
|
1023
|
+
|
|
1024
|
+
lines = content.splitlines()
|
|
1025
|
+
lbl_norm = normalize_text(label)
|
|
1026
|
+
if not lbl_norm:
|
|
1027
|
+
return False, None
|
|
1028
|
+
for i, raw in enumerate(lines):
|
|
1029
|
+
stripped = raw.strip()
|
|
1030
|
+
if not stripped or len(stripped) > 100:
|
|
1031
|
+
continue
|
|
1032
|
+
if stripped.startswith(("#", ">", "|", "<", "`")):
|
|
1033
|
+
continue
|
|
1034
|
+
if "**" in stripped or ":" in stripped:
|
|
1035
|
+
continue
|
|
1036
|
+
italic_match = _ITALIC_LABEL_LINE_RE.match(raw)
|
|
1037
|
+
if italic_match:
|
|
1038
|
+
cleaned = italic_match.group(2).strip()
|
|
1039
|
+
else:
|
|
1040
|
+
cleaned = re.sub(r"\s*[)\\]+\s*$", "", stripped)
|
|
1041
|
+
cleaned = re.sub(r"^\\?[-*+]\s+", "", cleaned).strip()
|
|
1042
|
+
cleaned = cleaned.strip("*_ \t").strip()
|
|
1043
|
+
if not cleaned or len(cleaned) > 80:
|
|
1044
|
+
continue
|
|
1045
|
+
cand_norm = normalize_text(cleaned)
|
|
1046
|
+
if not cand_norm:
|
|
1047
|
+
continue
|
|
1048
|
+
strict_match = fuzz.ratio(cand_norm, lbl_norm) / 100.0 >= CELL_FUZZY_MATCH_THRESHOLD
|
|
1049
|
+
tolerant_match = label_max_diffs > 0 and _label_match_score(cleaned, label, label_max_diffs) > 0.0
|
|
1050
|
+
if not strict_match and not tolerant_match:
|
|
1051
|
+
continue
|
|
1052
|
+
if italic_match:
|
|
1053
|
+
search_orders = [
|
|
1054
|
+
range(i - 1, max(i - 4, -1), -1),
|
|
1055
|
+
range(i + 1, min(i + 4, len(lines))),
|
|
1056
|
+
]
|
|
1057
|
+
else:
|
|
1058
|
+
search_orders = [
|
|
1059
|
+
range(i + 1, min(i + 4, len(lines))),
|
|
1060
|
+
range(i - 1, max(i - 4, -1), -1),
|
|
1061
|
+
]
|
|
1062
|
+
for order in search_orders:
|
|
1063
|
+
for j in order:
|
|
1064
|
+
cand_line = lines[j].strip()
|
|
1065
|
+
if not cand_line:
|
|
1066
|
+
continue
|
|
1067
|
+
if cand_line.startswith(("#", "|", ">")):
|
|
1068
|
+
break
|
|
1069
|
+
if "**" in cand_line and ":" in cand_line:
|
|
1070
|
+
break
|
|
1071
|
+
value = re.sub(r"^\\?[-*+]\s+", "", cand_line).strip()
|
|
1072
|
+
value = re.sub(r"\s*[)\\]+\s*$", "", value).strip()
|
|
1073
|
+
value = value.strip("*_ \t").strip()
|
|
1074
|
+
if value:
|
|
1075
|
+
return True, value
|
|
1076
|
+
return True, ""
|
|
1077
|
+
return False, None
|
|
1078
|
+
|
|
1079
|
+
|
|
1080
|
+
def _build_cell_text_counts(data, rows: int, cols: int) -> dict[str, int]: # type: ignore[no-untyped-def]
|
|
1081
|
+
"""Per-table map of text → number of distinct *origin* cells.
|
|
1082
|
+
|
|
1083
|
+
``parse_html_tables`` expands ``colspan``/``rowspan`` by duplicating cell
|
|
1084
|
+
text across every covered grid position, so a single ``<th
|
|
1085
|
+
colspan="4">KEBO</th>`` looks like four ``"KEBO"`` cells in the expanded
|
|
1086
|
+
grid. Counting raw grid cells would mis-classify any spanned value as a
|
|
1087
|
+
repeated label. Dedupe by skipping cells whose text equals the left or
|
|
1088
|
+
above neighbor — those are colspan / rowspan runs of the same origin.
|
|
1089
|
+
"""
|
|
1090
|
+
|
|
1091
|
+
counts: dict[str, int] = {}
|
|
1092
|
+
for r in range(rows):
|
|
1093
|
+
for c in range(cols):
|
|
1094
|
+
t = str(data[r, c]).strip()
|
|
1095
|
+
if not t:
|
|
1096
|
+
continue
|
|
1097
|
+
if c > 0 and str(data[r, c - 1]).strip() == t:
|
|
1098
|
+
continue
|
|
1099
|
+
if r > 0 and str(data[r - 1, c]).strip() == t:
|
|
1100
|
+
continue
|
|
1101
|
+
counts[t] = counts.get(t, 0) + 1
|
|
1102
|
+
return counts
|
|
1103
|
+
|
|
1104
|
+
|
|
1105
|
+
def _neighbor_is_label_like(neighbor: str, text_counts: dict[str, int]) -> bool:
|
|
1106
|
+
if _cell_text_is_label_like(neighbor):
|
|
1107
|
+
return True
|
|
1108
|
+
# Short text that repeats elsewhere in the same table → structural label.
|
|
1109
|
+
if len(neighbor) <= 30 and text_counts.get(neighbor, 0) >= 2:
|
|
1110
|
+
return True
|
|
1111
|
+
return False
|
|
1112
|
+
|
|
1113
|
+
|
|
1114
|
+
def _is_value_shaped_cell(neighbor: str | None) -> bool:
|
|
1115
|
+
if not neighbor:
|
|
1116
|
+
return False
|
|
1117
|
+
s = neighbor.strip()
|
|
1118
|
+
if len(s) < 2:
|
|
1119
|
+
return False
|
|
1120
|
+
# Lone checkbox glyphs aren't useful values for text rules.
|
|
1121
|
+
if _GLYPH_RE.search(s) and len(s) <= 2:
|
|
1122
|
+
return False
|
|
1123
|
+
return True
|
|
1124
|
+
|
|
1125
|
+
|
|
1126
|
+
def _iter_html_cell_neighbor_pairs(content: str) -> list[tuple[str, str]]:
|
|
1127
|
+
"""Yield ``(cell_text, neighbor_value)`` pairs for wide (>2 col) HTML
|
|
1128
|
+
tables, intended as a low-priority fallback source for
|
|
1129
|
+
``_find_text_value_for_label``.
|
|
1130
|
+
|
|
1131
|
+
Targets form-style layouts where labels and values are spatially
|
|
1132
|
+
interleaved inside a single wide ``<table>`` rather than separated into
|
|
1133
|
+
a clean header row + data rows, e.g. well-log report headers::
|
|
1134
|
+
|
|
1135
|
+
<tr><th colspan="2">FILE NO:</th>
|
|
1136
|
+
<th colspan="2">COMPANY</th>
|
|
1137
|
+
<th colspan="4">KEBO OIL & GAS, INC.</th></tr>
|
|
1138
|
+
<tr><th colspan="2">API NO:</th>
|
|
1139
|
+
<th colspan="2">WELL</th>
|
|
1140
|
+
<th colspan="4">LEHMAN #1</th></tr>
|
|
1141
|
+
<tr><th colspan="2">42-157-33282</th>
|
|
1142
|
+
<th colspan="2">FIELD</th>
|
|
1143
|
+
<th colspan="4">NEEDVILLE</th></tr>
|
|
1144
|
+
|
|
1145
|
+
For each non-empty cell in the expanded grid the iterator looks at two
|
|
1146
|
+
candidate neighbors:
|
|
1147
|
+
|
|
1148
|
+
* the first non-empty cell to the right in the same row, skipping
|
|
1149
|
+
colspan duplicates (cell text equal to the cell itself);
|
|
1150
|
+
* the first non-empty cell below in the same column, similarly skipping
|
|
1151
|
+
rowspan duplicates.
|
|
1152
|
+
|
|
1153
|
+
The chosen neighbor is the first one that is *value-shaped* (length ≥ 2,
|
|
1154
|
+
not a lone checkbox glyph) and *not label-shaped* per
|
|
1155
|
+
``_cell_text_is_label_like`` or structural repetition in the same table.
|
|
1156
|
+
Right is preferred over below (matches left-to-right reading).
|
|
1157
|
+
|
|
1158
|
+
Cells with no value-shaped neighbor still emit ``(cell_text, "")`` so the
|
|
1159
|
+
caller's ``_collect`` records ``label_seen=True`` for empty-expected
|
|
1160
|
+
rules — same contract as the other pair sources.
|
|
1161
|
+
|
|
1162
|
+
The caller scores ``cell_text`` against the rule label via
|
|
1163
|
+
``_label_match_score`` and picks the best candidate. We don't filter by
|
|
1164
|
+
label here so the caller can resolve adjacent-label collisions (the
|
|
1165
|
+
same way #978 made other sources do).
|
|
1166
|
+
"""
|
|
1167
|
+
|
|
1168
|
+
if "<table" not in content.lower():
|
|
1169
|
+
return []
|
|
1170
|
+
|
|
1171
|
+
out: list[tuple[str, str]] = []
|
|
1172
|
+
|
|
1173
|
+
for table in parse_html_tables(content):
|
|
1174
|
+
rows, cols = table.data.shape
|
|
1175
|
+
if cols <= 2 or rows == 0:
|
|
1176
|
+
continue
|
|
1177
|
+
|
|
1178
|
+
# Per-table text-count map — a short text that exactly repeats in
|
|
1179
|
+
# ≥2 *distinct origin* cells (after collapsing colspan/rowspan runs
|
|
1180
|
+
# via ``_build_cell_text_counts``) is structurally likely to be a
|
|
1181
|
+
# column label / section header (e.g. ``KB`` / ``DF`` / ``GL`` rows
|
|
1182
|
+
# in well-log elevation blocks). Used as a tie-breaker for which
|
|
1183
|
+
# neighbor cell is value-shaped.
|
|
1184
|
+
text_counts = _build_cell_text_counts(table.data, rows, cols)
|
|
1185
|
+
|
|
1186
|
+
for r in range(rows):
|
|
1187
|
+
for c in range(cols):
|
|
1188
|
+
cell = str(table.data[r, c]).strip()
|
|
1189
|
+
if not cell:
|
|
1190
|
+
continue
|
|
1191
|
+
|
|
1192
|
+
# Right scan: first non-empty cell to the right that is not
|
|
1193
|
+
# a colspan duplicate (text != label cell text).
|
|
1194
|
+
right_val: str | None = None
|
|
1195
|
+
for cc in range(c + 1, cols):
|
|
1196
|
+
nxt = str(table.data[r, cc]).strip()
|
|
1197
|
+
if nxt and nxt != cell:
|
|
1198
|
+
right_val = nxt
|
|
1199
|
+
break
|
|
1200
|
+
|
|
1201
|
+
# Below scan: first non-empty cell directly below that is
|
|
1202
|
+
# not a rowspan duplicate.
|
|
1203
|
+
below_val: str | None = None
|
|
1204
|
+
for rr in range(r + 1, rows):
|
|
1205
|
+
nxt = str(table.data[rr, c]).strip()
|
|
1206
|
+
if nxt and nxt != cell:
|
|
1207
|
+
below_val = nxt
|
|
1208
|
+
break
|
|
1209
|
+
|
|
1210
|
+
# Score each candidate. Want a value-shaped neighbor that
|
|
1211
|
+
# does *not* itself look label-like. Right is preferred over
|
|
1212
|
+
# below when both qualify (matches left-to-right reading).
|
|
1213
|
+
right_ok = _is_value_shaped_cell(right_val) and not _neighbor_is_label_like(
|
|
1214
|
+
right_val or "", text_counts
|
|
1215
|
+
)
|
|
1216
|
+
below_ok = _is_value_shaped_cell(below_val) and not _neighbor_is_label_like(
|
|
1217
|
+
below_val or "", text_counts
|
|
1218
|
+
)
|
|
1219
|
+
|
|
1220
|
+
if right_ok:
|
|
1221
|
+
out.append((cell, right_val or ""))
|
|
1222
|
+
elif below_ok:
|
|
1223
|
+
out.append((cell, below_val or ""))
|
|
1224
|
+
else:
|
|
1225
|
+
# No value-shaped neighbor at this position. Still emit
|
|
1226
|
+
# an empty-value pair so a matching label sets the
|
|
1227
|
+
# caller's ``label_seen`` flag (mirrors the other pair
|
|
1228
|
+
# iterators that surface ``""`` for label-only hits).
|
|
1229
|
+
out.append((cell, ""))
|
|
1230
|
+
|
|
1231
|
+
return out
|
|
1232
|
+
|
|
1233
|
+
|
|
1234
|
+
def _find_text_value_for_label(
|
|
1235
|
+
content: str,
|
|
1236
|
+
label: str,
|
|
1237
|
+
expected_values: list[str] | None = None,
|
|
1238
|
+
max_diffs: int | float = 0,
|
|
1239
|
+
label_max_diffs: int | float = 0,
|
|
1240
|
+
) -> tuple[bool, str | None]:
|
|
1241
|
+
"""Look up the value for *label*. Returns (label_found, value).
|
|
1242
|
+
|
|
1243
|
+
The boolean tracks whether the label was located **at all** — useful for
|
|
1244
|
+
distinguishing "label missing from content" from "label present but value
|
|
1245
|
+
blank" (signature evaluation depends on this distinction). When the label
|
|
1246
|
+
is found only with empty values, returns ``(True, "")`` so callers can
|
|
1247
|
+
decide what to do (text rules with empty expected values pass; signature
|
|
1248
|
+
rules treat it as unsigned).
|
|
1249
|
+
|
|
1250
|
+
Matching strategy is **best-score across all sources**: every candidate
|
|
1251
|
+
KV pair from every source iterator is scored against the target label
|
|
1252
|
+
via :func:`_label_match_score`, and the highest-scoring non-empty value
|
|
1253
|
+
wins. Tie-breaks fall back to source priority (bold-colon > plain-colon
|
|
1254
|
+
> md-table > html-table > underline-fill) and then document order. This
|
|
1255
|
+
eliminates the classic ``Country`` / ``County`` adjacent-label
|
|
1256
|
+
collision: an exact ``Country`` hit (score 1.0) always wins over a
|
|
1257
|
+
fuzzy ``County`` hit (score ~0.86–0.95) no matter which comes first.
|
|
1258
|
+
|
|
1259
|
+
Multi-occurrence disambiguation via ``expected_values``
|
|
1260
|
+
-------------------------------------------------------
|
|
1261
|
+
A label text can legitimately appear multiple times at the **same**
|
|
1262
|
+
best score: ``KB`` / ``DF`` / ``GL`` are exact-match labels in well-
|
|
1263
|
+
log elevation blocks while also appearing as values of
|
|
1264
|
+
``LOG MEASURED FROM`` / ``DRILL. MEAS. FROM`` (where the cell-
|
|
1265
|
+
neighbor matcher surfaces them with score 1.0). Without
|
|
1266
|
+
disambiguation, source-priority + doc-order tie-breaks would lock
|
|
1267
|
+
onto an arbitrary occurrence — which one happens to come first
|
|
1268
|
+
has no relation to which page occurrence the GT refers to.
|
|
1269
|
+
|
|
1270
|
+
When ``expected_values`` is supplied, the matcher applies the rule's
|
|
1271
|
+
expected value(s) as an oracle **among candidates at the top score
|
|
1272
|
+
level only**. That is: it picks the highest score; collects every
|
|
1273
|
+
candidate at that score; and returns the first one whose value
|
|
1274
|
+
matches any expected via :func:`_values_match_text`. If no
|
|
1275
|
+
top-score candidate matches, the legacy source-priority / doc-order
|
|
1276
|
+
tie-break fires — same as without ``expected_values``.
|
|
1277
|
+
|
|
1278
|
+
The "top score only" gate is what keeps the GT oracle from leaking
|
|
1279
|
+
across adjacent labels: ``Country`` (score 1.0) vs ``County``
|
|
1280
|
+
(partial 0.95) live at *different* score levels, so even if a
|
|
1281
|
+
``County`` row's value coincidentally equals the GT's expected
|
|
1282
|
+
``Country`` value, ``County`` is not eligible. Only when two
|
|
1283
|
+
candidates are equally good *label matches* does the value oracle
|
|
1284
|
+
intervene.
|
|
1285
|
+
"""
|
|
1286
|
+
|
|
1287
|
+
label_seen = False
|
|
1288
|
+
# (negated_score, negated_priority, doc_order, value, source_name)
|
|
1289
|
+
# — we'll sort ascending so the *best* candidate (highest score, then
|
|
1290
|
+
# highest priority, then earliest doc order) sits at the top.
|
|
1291
|
+
candidates: list[tuple[float, int, int, str, str]] = []
|
|
1292
|
+
|
|
1293
|
+
def _collect(
|
|
1294
|
+
pairs: list[tuple[str, str]],
|
|
1295
|
+
priority: int,
|
|
1296
|
+
source_name: str,
|
|
1297
|
+
) -> None:
|
|
1298
|
+
nonlocal label_seen
|
|
1299
|
+
for idx, (cand_label, cand_value) in enumerate(pairs):
|
|
1300
|
+
score = _label_match_score(cand_label, label, label_max_diffs)
|
|
1301
|
+
if score <= 0.0:
|
|
1302
|
+
continue
|
|
1303
|
+
label_seen = True
|
|
1304
|
+
if cand_value:
|
|
1305
|
+
candidates.append((-score, -priority, idx, cand_value, source_name))
|
|
1306
|
+
|
|
1307
|
+
# Higher priority numbers = more confident sources. The ordering matches
|
|
1308
|
+
# the original first-match-wins precedence so tie-breaks preserve legacy
|
|
1309
|
+
# behavior on documents where multiple sources produce equally strong
|
|
1310
|
+
# label matches.
|
|
1311
|
+
_collect(_iter_bold_colon_pairs(content), priority=4, source_name="bold_colon")
|
|
1312
|
+
_collect(_iter_plain_colon_pairs(content), priority=3, source_name="plain_colon")
|
|
1313
|
+
_collect(_iter_md_table_kv_rows(content), priority=2, source_name="md_table")
|
|
1314
|
+
_collect(_iter_html_table_kv_rows(content), priority=1, source_name="html_table")
|
|
1315
|
+
_collect(_iter_label_before_bold_value_pairs(content), priority=0, source_name="label_before_bold")
|
|
1316
|
+
|
|
1317
|
+
# Underscore blank fields (``Label ____``) — label seen, value empty.
|
|
1318
|
+
# The yielded value is always ""; we just record label presence so an
|
|
1319
|
+
# empty-expected text rule can pass via the ``label_seen`` short-circuit
|
|
1320
|
+
# below.
|
|
1321
|
+
for cand_label, _ in _iter_underscore_blank_pairs(content):
|
|
1322
|
+
if _label_matches(cand_label, label, label_max_diffs):
|
|
1323
|
+
label_seen = True
|
|
1324
|
+
|
|
1325
|
+
# Last-resort sources — only fire when no higher-confidence source
|
|
1326
|
+
# surfaced a non-empty value, so they never overwrite a strong-source
|
|
1327
|
+
# extraction. Both are gated on ``not candidates`` and added with
|
|
1328
|
+
# priorities below the strong sources; if both fire and both produce
|
|
1329
|
+
# candidates, ``priority`` breaks the tie in favor of underline_fill.
|
|
1330
|
+
if not candidates:
|
|
1331
|
+
# Underline fill-in (``Label <u>value</u>`` inline in prose).
|
|
1332
|
+
if "<u>" in content:
|
|
1333
|
+
_collect(
|
|
1334
|
+
_iter_underline_fill_pairs(content),
|
|
1335
|
+
priority=0,
|
|
1336
|
+
source_name="underline_fill",
|
|
1337
|
+
)
|
|
1338
|
+
# Wide-form HTML table cell-neighbor fallback. Targets layouts
|
|
1339
|
+
# where labels and values are spatially interleaved inside a single
|
|
1340
|
+
# wide ``<table>`` (well-log report headers, rotated form pages),
|
|
1341
|
+
# which neither the 2-col nor the multi-col header×data pair
|
|
1342
|
+
# iterator covers. Priority -1 keeps it strictly below
|
|
1343
|
+
# underline_fill on tie-breaks.
|
|
1344
|
+
_collect(
|
|
1345
|
+
_iter_html_cell_neighbor_pairs(content),
|
|
1346
|
+
priority=-1,
|
|
1347
|
+
source_name="html_cell_neighbor",
|
|
1348
|
+
)
|
|
1349
|
+
|
|
1350
|
+
if candidates:
|
|
1351
|
+
candidates.sort()
|
|
1352
|
+
# ``candidates`` is sorted ascending by (-score, -priority, doc_order,
|
|
1353
|
+
# ...), so the head is the best (label-match-score, source-priority,
|
|
1354
|
+
# doc-order) tuple. We use the rule's expected value as a tie-breaker
|
|
1355
|
+
# **only among candidates at the head score**, which keeps the GT
|
|
1356
|
+
# oracle from leaking across adjacent labels (Country score 1.0 vs
|
|
1357
|
+
# County score ~0.95 live at different levels, so County is never
|
|
1358
|
+
# eligible when Country is present).
|
|
1359
|
+
best_score_key = candidates[0][0]
|
|
1360
|
+
if expected_values:
|
|
1361
|
+
for neg_score, _prio, _doc, value, _src in candidates:
|
|
1362
|
+
if neg_score != best_score_key:
|
|
1363
|
+
break
|
|
1364
|
+
for exp in expected_values:
|
|
1365
|
+
if _values_match_text(value, exp, max_diffs):
|
|
1366
|
+
return True, value
|
|
1367
|
+
return True, candidates[0][3]
|
|
1368
|
+
|
|
1369
|
+
# Adjacent-line fallback (italic caption below value, or numbered label
|
|
1370
|
+
# above value). Only fires when no other matcher located the label.
|
|
1371
|
+
if not label_seen:
|
|
1372
|
+
adj_seen, adj_value = _find_text_value_adjacent_line(content, label, label_max_diffs)
|
|
1373
|
+
if adj_seen:
|
|
1374
|
+
return True, adj_value
|
|
1375
|
+
|
|
1376
|
+
if label_seen:
|
|
1377
|
+
return True, ""
|
|
1378
|
+
return False, None
|
|
1379
|
+
|
|
1380
|
+
|
|
1381
|
+
def _tokenize_checkbox_line(line: str) -> list[tuple[str, bool]]:
|
|
1382
|
+
"""Pair every checkbox marker on *line* with its associated label.
|
|
1383
|
+
|
|
1384
|
+
Markers may be Unicode glyphs (``☐``/``☑``/``◉``/``○``/...) OR ASCII
|
|
1385
|
+
bracket pairs (``[x]``, ``\\[x\\]``, ``[ ]``). Handles both orderings:
|
|
1386
|
+
|
|
1387
|
+
- marker-first: ``☐ Single ☑ Married`` or ``\\[x] A \\[ ] B`` — each
|
|
1388
|
+
label sits between a marker and the next marker (or end of line).
|
|
1389
|
+
- label-first: ``Single ☐ Married ☑`` or ``Checking \\[x] Savings \\[ ]``
|
|
1390
|
+
— each label sits between the previous marker (or start) and the
|
|
1391
|
+
next marker.
|
|
1392
|
+
|
|
1393
|
+
Direction is decided by what comes before the first marker: if the line
|
|
1394
|
+
starts with the marker (after optional whitespace), use marker-first;
|
|
1395
|
+
otherwise use label-first. This covers the inline mid-line bracket
|
|
1396
|
+
pattern ``**Inaccuracy in financing statement** \\[ ]`` since the
|
|
1397
|
+
closing bracket is treated as a marker and the bold-label segment to
|
|
1398
|
+
its left becomes the label.
|
|
1399
|
+
"""
|
|
1400
|
+
|
|
1401
|
+
marker_matches = list(_MARKER_RE.finditer(line))
|
|
1402
|
+
if not marker_matches:
|
|
1403
|
+
return []
|
|
1404
|
+
|
|
1405
|
+
text_before_first = line[: marker_matches[0].start()].strip()
|
|
1406
|
+
pairs: list[tuple[str, bool]] = []
|
|
1407
|
+
|
|
1408
|
+
if not text_before_first:
|
|
1409
|
+
# marker-first: label runs from marker end to next marker start (or EOL).
|
|
1410
|
+
for i, m in enumerate(marker_matches):
|
|
1411
|
+
label_start = m.end()
|
|
1412
|
+
label_end = marker_matches[i + 1].start() if i + 1 < len(marker_matches) else len(line)
|
|
1413
|
+
label_text = line[label_start:label_end].strip()
|
|
1414
|
+
if label_text:
|
|
1415
|
+
pairs.append((label_text, _marker_is_checked(m.group())))
|
|
1416
|
+
else:
|
|
1417
|
+
# label-first: label runs from previous marker end (or 0) to current marker.
|
|
1418
|
+
prev_end = 0
|
|
1419
|
+
for m in marker_matches:
|
|
1420
|
+
label_text = line[prev_end : m.start()].strip()
|
|
1421
|
+
prev_end = m.end()
|
|
1422
|
+
if label_text:
|
|
1423
|
+
pairs.append((label_text, _marker_is_checked(m.group())))
|
|
1424
|
+
return pairs
|
|
1425
|
+
|
|
1426
|
+
|
|
1427
|
+
_TRAILING_BOOL_OPTION_RE = re.compile(
|
|
1428
|
+
r"^(?P<context>.*?)(?P<option>\b(?:yes|no|y|n)\b)\s*$",
|
|
1429
|
+
re.IGNORECASE,
|
|
1430
|
+
)
|
|
1431
|
+
|
|
1432
|
+
|
|
1433
|
+
def _tokenize_checkbox_line_with_context(line: str) -> list[tuple[list[str], bool]]:
|
|
1434
|
+
"""Pair inline yes/no checkbox markers with their question context.
|
|
1435
|
+
|
|
1436
|
+
Typical form parsers render a yes/no row as one line:
|
|
1437
|
+
|
|
1438
|
+
``Multistage cement? Yes [ ] No [x]``
|
|
1439
|
+
|
|
1440
|
+
The plain tokenizer sees ``"Multistage cement? Yes" -> False`` and
|
|
1441
|
+
``"No" -> True``. For a multi-label checkbox rule we want the more precise
|
|
1442
|
+
candidates ``["Multistage cement?", "Yes"] -> False`` and
|
|
1443
|
+
``["Multistage cement?", "No"] -> True`` so repeated Yes/No options can be
|
|
1444
|
+
disambiguated by the surrounding question without adding a new schema.
|
|
1445
|
+
"""
|
|
1446
|
+
|
|
1447
|
+
marker_matches = list(_MARKER_RE.finditer(line))
|
|
1448
|
+
if not marker_matches:
|
|
1449
|
+
return []
|
|
1450
|
+
|
|
1451
|
+
text_before_first = line[: marker_matches[0].start()].strip()
|
|
1452
|
+
if not text_before_first:
|
|
1453
|
+
return []
|
|
1454
|
+
|
|
1455
|
+
pairs: list[tuple[list[str], bool]] = []
|
|
1456
|
+
current_context: str | None = None
|
|
1457
|
+
prev_end = 0
|
|
1458
|
+
for marker in marker_matches:
|
|
1459
|
+
label_text = line[prev_end : marker.start()].strip()
|
|
1460
|
+
prev_end = marker.end()
|
|
1461
|
+
if not label_text:
|
|
1462
|
+
continue
|
|
1463
|
+
|
|
1464
|
+
labels = [label_text]
|
|
1465
|
+
option_match = _TRAILING_BOOL_OPTION_RE.match(label_text)
|
|
1466
|
+
if option_match:
|
|
1467
|
+
context = option_match.group("context").strip(" :;-")
|
|
1468
|
+
option = option_match.group("option").strip()
|
|
1469
|
+
if context:
|
|
1470
|
+
current_context = context
|
|
1471
|
+
labels = [context, option]
|
|
1472
|
+
elif current_context:
|
|
1473
|
+
labels = [current_context, option]
|
|
1474
|
+
|
|
1475
|
+
parts = _dedupe_nonempty_text(labels)
|
|
1476
|
+
if parts:
|
|
1477
|
+
pairs.append((parts, _marker_is_checked(marker.group())))
|
|
1478
|
+
|
|
1479
|
+
return pairs
|
|
1480
|
+
|
|
1481
|
+
|
|
1482
|
+
def _checkbox_label_list_match_score(
|
|
1483
|
+
candidate_labels: list[str],
|
|
1484
|
+
required_labels: list[str],
|
|
1485
|
+
label_max_diffs: list[int],
|
|
1486
|
+
) -> float:
|
|
1487
|
+
if len(candidate_labels) < len(required_labels):
|
|
1488
|
+
return 0.0
|
|
1489
|
+
|
|
1490
|
+
used: set[int] = set()
|
|
1491
|
+
total = 0.0
|
|
1492
|
+
for req_idx, required in enumerate(required_labels):
|
|
1493
|
+
allowed = label_max_diffs[req_idx] if req_idx < len(label_max_diffs) else 0
|
|
1494
|
+
best_idx = -1
|
|
1495
|
+
best_score = 0.0
|
|
1496
|
+
for cand_idx, candidate in enumerate(candidate_labels):
|
|
1497
|
+
if cand_idx in used:
|
|
1498
|
+
continue
|
|
1499
|
+
score = _label_match_score(candidate, required, allowed)
|
|
1500
|
+
if score > best_score:
|
|
1501
|
+
best_idx = cand_idx
|
|
1502
|
+
best_score = score
|
|
1503
|
+
if best_idx < 0 or best_score <= 0.0:
|
|
1504
|
+
return 0.0
|
|
1505
|
+
used.add(best_idx)
|
|
1506
|
+
total += best_score
|
|
1507
|
+
|
|
1508
|
+
return total / max(len(required_labels), 1)
|
|
1509
|
+
|
|
1510
|
+
|
|
1511
|
+
# Markdown task-list. Allows optional ``\`` escapes around the list marker
|
|
1512
|
+
# AND the brackets — some parsers emit ``\[x\]`` (or even ``\- \[x\]`` for a
|
|
1513
|
+
# nested escaped bullet) so the markdown source survives literal-character
|
|
1514
|
+
# rendering. The marker accepts ``-``/``*``/``+`` and numbered-list ``\d+.`` —
|
|
1515
|
+
# USCIS citizenship attestations are rendered as ``1. [x] A citizen ...``.
|
|
1516
|
+
_LIST_MARKER_RE_INLINE = r"(?:[-*+]|\d+\.)"
|
|
1517
|
+
_MD_TASKLIST_RE = re.compile(
|
|
1518
|
+
rf"^\s*\\?{_LIST_MARKER_RE_INLINE}\s*\\?\[([ xX])\\?\]\s*(.+?)\s*$",
|
|
1519
|
+
re.MULTILINE,
|
|
1520
|
+
)
|
|
1521
|
+
|
|
1522
|
+
# Label-first bullet checkbox: ``* Checking \[x]`` or ``\- Savings [ ]``. The
|
|
1523
|
+
# label sits between the list marker and the bracket. Common in forms where
|
|
1524
|
+
# the parser surfaces the option label as the bullet text and the state as a
|
|
1525
|
+
# trailing widget marker. Both the bullet marker and the brackets may be
|
|
1526
|
+
# preceded by a literal backslash escape.
|
|
1527
|
+
_MD_BULLET_LABEL_FIRST_RE = re.compile(
|
|
1528
|
+
rf"^\s*\\?{_LIST_MARKER_RE_INLINE}\s+([^\[\n]+?)\s+\\?\[([ xX])\\?\]\s*$",
|
|
1529
|
+
re.MULTILINE,
|
|
1530
|
+
)
|
|
1531
|
+
|
|
1532
|
+
# Bullet-less task-list: a line that starts with ``\[x]`` / ``[ ]`` directly
|
|
1533
|
+
# with no leading bullet marker. ours_cost_effective and gemini render IRS
|
|
1534
|
+
# W-9 / USCIS / UCC5 checkboxes this way (``\[x] Individual/sole proprietor``
|
|
1535
|
+
# on its own line). The label group disallows ``[`` so a line with multiple
|
|
1536
|
+
# inline bracket markers (``\[ ] A \[x] B``) does NOT match here — those go
|
|
1537
|
+
# through the per-line tokenizer below where each bracket is paired with its
|
|
1538
|
+
# own label.
|
|
1539
|
+
_MD_BARE_TASKLIST_RE = re.compile(
|
|
1540
|
+
r"^\s*\\?\[([ xX])\\?\]\s*([^\[\n]+?)\s*$",
|
|
1541
|
+
re.MULTILINE,
|
|
1542
|
+
)
|
|
1543
|
+
|
|
1544
|
+
|
|
1545
|
+
def _find_checkbox_state_for_label(
|
|
1546
|
+
content: str,
|
|
1547
|
+
label: str,
|
|
1548
|
+
label_max_diffs: int | float = 0,
|
|
1549
|
+
) -> bool | None:
|
|
1550
|
+
"""Return True/False if *label* has a checkbox-style state nearby, else None.
|
|
1551
|
+
|
|
1552
|
+
Like :func:`_find_text_value_for_label`, this collects every candidate
|
|
1553
|
+
``(label, state)`` across every checkbox source, scores the label, and
|
|
1554
|
+
returns the state attached to the highest-scoring candidate. Avoids
|
|
1555
|
+
adjacent-label collisions where two visually similar labels share a
|
|
1556
|
+
line and the wrong one gets picked just because it came first.
|
|
1557
|
+
"""
|
|
1558
|
+
|
|
1559
|
+
# (negated_score, negated_priority, doc_order, state)
|
|
1560
|
+
candidates: list[tuple[float, int, int, bool]] = []
|
|
1561
|
+
|
|
1562
|
+
def _try_add(cand_label: str, state: bool, priority: int, idx: int) -> None:
|
|
1563
|
+
score = _label_match_score(cand_label, label, label_max_diffs)
|
|
1564
|
+
if score > 0.0:
|
|
1565
|
+
candidates.append((-score, -priority, idx, state))
|
|
1566
|
+
|
|
1567
|
+
# Markdown task-list: - [x] Label or - [ ] Label or 1. [x] Label
|
|
1568
|
+
for idx, match in enumerate(_MD_TASKLIST_RE.finditer(content)):
|
|
1569
|
+
state_char, cand_label = match.group(1), match.group(2).strip()
|
|
1570
|
+
_try_add(cand_label, state_char.strip().lower() == "x", priority=4, idx=idx)
|
|
1571
|
+
|
|
1572
|
+
# Label-first bullet: - Label [x] or * Label \[x]
|
|
1573
|
+
for idx, match in enumerate(_MD_BULLET_LABEL_FIRST_RE.finditer(content)):
|
|
1574
|
+
cand_label, state_char = match.group(1).strip(), match.group(2)
|
|
1575
|
+
_try_add(cand_label, state_char.strip().lower() == "x", priority=3, idx=idx)
|
|
1576
|
+
|
|
1577
|
+
# Bullet-less task-list: \[x] Label (no leading -/*/+/digit.)
|
|
1578
|
+
for idx, match in enumerate(_MD_BARE_TASKLIST_RE.finditer(content)):
|
|
1579
|
+
state_char, cand_label = match.group(1), match.group(2).strip()
|
|
1580
|
+
_try_add(cand_label, state_char.strip().lower() == "x", priority=2, idx=idx)
|
|
1581
|
+
|
|
1582
|
+
# Per-line marker tokenization (handles inline groups in either direction
|
|
1583
|
+
# and mid-line ASCII bracket markers after a bold label).
|
|
1584
|
+
inline_idx = 0
|
|
1585
|
+
for line in content.splitlines():
|
|
1586
|
+
if not _MARKER_RE.search(line):
|
|
1587
|
+
continue
|
|
1588
|
+
for cand_label, state in _tokenize_checkbox_line(line):
|
|
1589
|
+
_try_add(cand_label, state, priority=1, idx=inline_idx)
|
|
1590
|
+
inline_idx += 1
|
|
1591
|
+
|
|
1592
|
+
if candidates:
|
|
1593
|
+
candidates.sort()
|
|
1594
|
+
return candidates[0][3]
|
|
1595
|
+
return None
|
|
1596
|
+
|
|
1597
|
+
|
|
1598
|
+
def _find_checkbox_state_for_label_list(
|
|
1599
|
+
content: str,
|
|
1600
|
+
labels: list[str],
|
|
1601
|
+
label_max_diffs: list[int],
|
|
1602
|
+
) -> bool | None:
|
|
1603
|
+
"""Return the checkbox state for a multi-label yes/no option."""
|
|
1604
|
+
|
|
1605
|
+
candidates: list[tuple[float, int, bool]] = []
|
|
1606
|
+
candidate_idx = 0
|
|
1607
|
+
for line in content.splitlines():
|
|
1608
|
+
if not _MARKER_RE.search(line):
|
|
1609
|
+
continue
|
|
1610
|
+
for candidate_labels, state in _tokenize_checkbox_line_with_context(line):
|
|
1611
|
+
score = _checkbox_label_list_match_score(candidate_labels, labels, label_max_diffs)
|
|
1612
|
+
if score > 0.0:
|
|
1613
|
+
candidates.append((-score, candidate_idx, state))
|
|
1614
|
+
candidate_idx += 1
|
|
1615
|
+
|
|
1616
|
+
if candidates:
|
|
1617
|
+
candidates.sort()
|
|
1618
|
+
return candidates[0][2]
|
|
1619
|
+
return None
|
|
1620
|
+
|
|
1621
|
+
|
|
1622
|
+
# Strikethrough span — match a ``~~...~~`` block AND its contents so an edit
|
|
1623
|
+
# history like ``~~old~~ new`` collapses to just ``new``. ``normalize_text``
|
|
1624
|
+
# only strips the ``~~`` markers (leaving the crossed-out text behind), which
|
|
1625
|
+
# is the wrong shape when the GT records the final clean value. The pattern
|
|
1626
|
+
# is non-greedy and bounded to a single line so it can't span paragraphs.
|
|
1627
|
+
_STRIKETHROUGH_SPAN_RE = re.compile(r"~~[^~\n]+~~")
|
|
1628
|
+
|
|
1629
|
+
|
|
1630
|
+
def _strip_strikethrough_spans(s: str) -> str:
|
|
1631
|
+
return _STRIKETHROUGH_SPAN_RE.sub("", s).strip()
|
|
1632
|
+
|
|
1633
|
+
|
|
1634
|
+
def _compact_value_text_for_distance(s: str) -> str:
|
|
1635
|
+
"""Keep only alphanumeric content after the normal form-value cleanup."""
|
|
1636
|
+
|
|
1637
|
+
return re.sub(r"[^0-9a-z]+", "", normalize_text(s))
|
|
1638
|
+
|
|
1639
|
+
|
|
1640
|
+
def _levenshtein_distance_at_most(left: str, right: str, max_distance: int) -> int:
|
|
1641
|
+
"""Compute edit distance, stopping once it is already above the limit."""
|
|
1642
|
+
|
|
1643
|
+
if left == right:
|
|
1644
|
+
return 0
|
|
1645
|
+
if abs(len(left) - len(right)) > max_distance:
|
|
1646
|
+
return max_distance + 1
|
|
1647
|
+
if len(left) < len(right):
|
|
1648
|
+
left, right = right, left
|
|
1649
|
+
|
|
1650
|
+
previous = list(range(len(right) + 1))
|
|
1651
|
+
for i, lch in enumerate(left, start=1):
|
|
1652
|
+
current = [i]
|
|
1653
|
+
row_min = current[0]
|
|
1654
|
+
for j, rch in enumerate(right, start=1):
|
|
1655
|
+
cost = 0 if lch == rch else 1
|
|
1656
|
+
current.append(
|
|
1657
|
+
min(
|
|
1658
|
+
previous[j] + 1,
|
|
1659
|
+
current[j - 1] + 1,
|
|
1660
|
+
previous[j - 1] + cost,
|
|
1661
|
+
)
|
|
1662
|
+
)
|
|
1663
|
+
row_min = min(row_min, current[-1])
|
|
1664
|
+
if row_min > max_distance:
|
|
1665
|
+
return max_distance + 1
|
|
1666
|
+
previous = current
|
|
1667
|
+
return previous[-1]
|
|
1668
|
+
|
|
1669
|
+
|
|
1670
|
+
def _values_match_text(found: str, expected: str, max_diffs: int | float = 0) -> bool:
|
|
1671
|
+
"""Compare two form-field text values with optional character tolerance.
|
|
1672
|
+
|
|
1673
|
+
Form values default to strict comparison on the meaningful text:
|
|
1674
|
+
|
|
1675
|
+
1. Exact match after normalization (``normalize_text`` already case-folds
|
|
1676
|
+
and collapses whitespace), which handles ``Madison`` vs ``madison``,
|
|
1677
|
+
trailing whitespace, and unicode quote variants.
|
|
1678
|
+
2. Strict numeric equality via ``normalize_number_string``, which lets
|
|
1679
|
+
``1,234`` match ``1234`` and ``$1,234.00`` match ``1234`` (the same
|
|
1680
|
+
value written differently) — but rejects ``53703`` vs ``53704``.
|
|
1681
|
+
3. Exact match after dropping separators/punctuation from the normalized
|
|
1682
|
+
value, which handles handwritten/date separators such as ``9-29`` vs
|
|
1683
|
+
``9 29`` without making any character substitutions.
|
|
1684
|
+
|
|
1685
|
+
When ``max_diffs`` is positive, the compact normalized values may differ
|
|
1686
|
+
by that many Levenshtein edits. This is intended for hard-to-read form
|
|
1687
|
+
values where one digit/letter may be ambiguous; the default remains 0.
|
|
1688
|
+
|
|
1689
|
+
Strikethrough spans (``~~old~~ new``) are stripped from the *found*
|
|
1690
|
+
value before comparison so the parser's edit-history rendering matches
|
|
1691
|
+
the GT's clean final value. The expected side is left untouched on the
|
|
1692
|
+
assumption GT never contains ``~~``.
|
|
1693
|
+
"""
|
|
1694
|
+
|
|
1695
|
+
found_stripped = _strip_strikethrough_spans(found)
|
|
1696
|
+
|
|
1697
|
+
f_norm = normalize_text(found_stripped)
|
|
1698
|
+
e_norm = normalize_text(expected)
|
|
1699
|
+
if f_norm == e_norm:
|
|
1700
|
+
return True
|
|
1701
|
+
if not f_norm or not e_norm:
|
|
1702
|
+
return False
|
|
1703
|
+
|
|
1704
|
+
f_num = normalize_number_string(found_stripped)
|
|
1705
|
+
e_num = normalize_number_string(expected)
|
|
1706
|
+
if f_num is not None and e_num is not None and f_num == e_num:
|
|
1707
|
+
return True
|
|
1708
|
+
|
|
1709
|
+
f_compact = _compact_value_text_for_distance(found_stripped)
|
|
1710
|
+
e_compact = _compact_value_text_for_distance(expected)
|
|
1711
|
+
if f_compact and e_compact and f_compact == e_compact:
|
|
1712
|
+
return True
|
|
1713
|
+
|
|
1714
|
+
allowed = int(max_diffs) if max_diffs and max_diffs > 0 else 0
|
|
1715
|
+
if allowed > 0 and f_compact and e_compact:
|
|
1716
|
+
return _levenshtein_distance_at_most(f_compact, e_compact, allowed) <= allowed
|
|
1717
|
+
|
|
1718
|
+
return False
|
|
1719
|
+
|
|
1720
|
+
|
|
1721
|
+
# Trailing ``(row N)`` annotation used by the form-field test generator to
|
|
1722
|
+
# point a label at a specific data row of a multi-column table. The column is
|
|
1723
|
+
# named by the prefix; ``N`` is 1-indexed over data rows (header rows are
|
|
1724
|
+
# skipped). This also covers simple repeated inline rows shaped as
|
|
1725
|
+
# ``FROM value TO value`` because some form parsers flatten ruled tables that
|
|
1726
|
+
# way instead of emitting a markdown table.
|
|
1727
|
+
_ROW_LABEL_RE = re.compile(r"\s*\(row\s+(\d+)\)\s*$", re.IGNORECASE)
|
|
1728
|
+
_INLINE_FROM_TO_RE = re.compile(r"^FROM\s*(?P<from>.*?)\s+TO\s*(?P<to>.*?)\s*$", re.IGNORECASE)
|
|
1729
|
+
|
|
1730
|
+
|
|
1731
|
+
def _split_row_label(label: str) -> tuple[str, int] | None:
|
|
1732
|
+
"""Return ``(column_label, row_index_1based)`` if *label* has a ``(row N)``
|
|
1733
|
+
suffix, else None."""
|
|
1734
|
+
|
|
1735
|
+
m = _ROW_LABEL_RE.search(label)
|
|
1736
|
+
if not m:
|
|
1737
|
+
return None
|
|
1738
|
+
col_label = label[: m.start()].strip()
|
|
1739
|
+
if not col_label:
|
|
1740
|
+
return None
|
|
1741
|
+
return col_label, int(m.group(1))
|
|
1742
|
+
|
|
1743
|
+
|
|
1744
|
+
def _column_header_for_index(table: TableData, col_idx: int) -> str:
|
|
1745
|
+
"""Concatenate every header cell stacked above column *col_idx* into one
|
|
1746
|
+
label. If the table has no recorded column headers (e.g. a markdown table
|
|
1747
|
+
where row 0 is the de facto header), fall back to row 0 of that column."""
|
|
1748
|
+
|
|
1749
|
+
parts: list[str] = []
|
|
1750
|
+
seen: set[str] = set()
|
|
1751
|
+
headers = getattr(table, "col_headers", {}) or {}
|
|
1752
|
+
for _, text in headers.get(col_idx, []):
|
|
1753
|
+
clean = (text or "").strip()
|
|
1754
|
+
if clean and clean not in seen:
|
|
1755
|
+
parts.append(clean)
|
|
1756
|
+
seen.add(clean)
|
|
1757
|
+
if parts:
|
|
1758
|
+
return " ".join(parts)
|
|
1759
|
+
if table.data.size and col_idx < table.data.shape[1]:
|
|
1760
|
+
return str(table.data[0, col_idx]).strip()
|
|
1761
|
+
return ""
|
|
1762
|
+
|
|
1763
|
+
|
|
1764
|
+
def _clean_inline_from_to_line(line: str) -> str:
|
|
1765
|
+
clean = _strip_html_tags(line)
|
|
1766
|
+
clean = re.sub(r"[*_`]+", "", clean)
|
|
1767
|
+
clean = clean.replace("\\", "")
|
|
1768
|
+
clean = re.sub(r"^\s*(?:[-+]\s+|\d+\.\s+)", "", clean)
|
|
1769
|
+
clean = re.sub(r"\s+", " ", clean).strip()
|
|
1770
|
+
return clean
|
|
1771
|
+
|
|
1772
|
+
|
|
1773
|
+
def _clean_inline_from_to_value(value: str) -> str:
|
|
1774
|
+
value = re.sub(r"(?:_|\s){3,}", " ", value)
|
|
1775
|
+
return value.strip(" :;-_")
|
|
1776
|
+
|
|
1777
|
+
|
|
1778
|
+
def _find_inline_from_to_row_label(
|
|
1779
|
+
content: str,
|
|
1780
|
+
col_label: str,
|
|
1781
|
+
row_n: int,
|
|
1782
|
+
label_max_diffs: int | float = 0,
|
|
1783
|
+
) -> tuple[bool, str | None]:
|
|
1784
|
+
"""Look up ``FROM``/``TO`` values in flattened interval rows.
|
|
1785
|
+
|
|
1786
|
+
Example parser output:
|
|
1787
|
+
|
|
1788
|
+
``FROM none reported TO RRC``
|
|
1789
|
+
|
|
1790
|
+
The rule keeps the same row-label syntax as markdown tables:
|
|
1791
|
+
``FROM (row 1)`` -> ``none reported`` and ``TO (row 1)`` -> ``RRC``.
|
|
1792
|
+
"""
|
|
1793
|
+
|
|
1794
|
+
wants_from = _label_matches("FROM", col_label, label_max_diffs)
|
|
1795
|
+
wants_to = _label_matches("TO", col_label, label_max_diffs)
|
|
1796
|
+
if not wants_from and not wants_to:
|
|
1797
|
+
return False, None
|
|
1798
|
+
|
|
1799
|
+
rows: list[tuple[str, str]] = []
|
|
1800
|
+
for raw_line in content.splitlines():
|
|
1801
|
+
if "|" in raw_line:
|
|
1802
|
+
continue
|
|
1803
|
+
line = _clean_inline_from_to_line(raw_line)
|
|
1804
|
+
if not line:
|
|
1805
|
+
continue
|
|
1806
|
+
match = _INLINE_FROM_TO_RE.match(line)
|
|
1807
|
+
if not match:
|
|
1808
|
+
continue
|
|
1809
|
+
rows.append(
|
|
1810
|
+
(
|
|
1811
|
+
_clean_inline_from_to_value(match.group("from")),
|
|
1812
|
+
_clean_inline_from_to_value(match.group("to")),
|
|
1813
|
+
)
|
|
1814
|
+
)
|
|
1815
|
+
|
|
1816
|
+
if not rows:
|
|
1817
|
+
return False, None
|
|
1818
|
+
if row_n < 1 or row_n > len(rows):
|
|
1819
|
+
return True, ""
|
|
1820
|
+
row_from, row_to = rows[row_n - 1]
|
|
1821
|
+
return True, row_from if wants_from else row_to
|
|
1822
|
+
|
|
1823
|
+
|
|
1824
|
+
def _find_table_cell_for_row_label(
|
|
1825
|
+
content: str,
|
|
1826
|
+
label: str,
|
|
1827
|
+
label_max_diffs: int | float = 0,
|
|
1828
|
+
) -> tuple[bool, str | None]:
|
|
1829
|
+
"""Look up ``"<col_label> (row N)"`` in any multi-column table.
|
|
1830
|
+
|
|
1831
|
+
Returns ``(label_seen, value_or_None)``. ``label_seen`` is True if a
|
|
1832
|
+
matching column was found in some table, even when the data row is out
|
|
1833
|
+
of range or the cell is empty — that distinction lets text rules with
|
|
1834
|
+
empty expected values pass on real empty cells without giving signature
|
|
1835
|
+
rules a free pass for missing labels.
|
|
1836
|
+
"""
|
|
1837
|
+
|
|
1838
|
+
parsed = _split_row_label(label)
|
|
1839
|
+
if parsed is None:
|
|
1840
|
+
return False, None
|
|
1841
|
+
col_label, row_n = parsed
|
|
1842
|
+
|
|
1843
|
+
label_seen = False
|
|
1844
|
+
for table in parse_html_tables(content) + parse_markdown_tables(content):
|
|
1845
|
+
if table.data.size == 0:
|
|
1846
|
+
continue
|
|
1847
|
+
rows, cols = table.data.shape
|
|
1848
|
+
# Determine which rows are headers. For HTML tables, header_rows is
|
|
1849
|
+
# populated from <thead>/<th>. For markdown tables, parse_markdown_tables
|
|
1850
|
+
# records header_rows={0} when a separator row is present.
|
|
1851
|
+
header_rows = getattr(table, "header_rows", set()) or set()
|
|
1852
|
+
n_header = (max(header_rows) + 1) if header_rows else 0
|
|
1853
|
+
data_row_idx = n_header + (row_n - 1)
|
|
1854
|
+
|
|
1855
|
+
for col_idx in range(cols):
|
|
1856
|
+
header_text = _column_header_for_index(table, col_idx)
|
|
1857
|
+
if not header_text:
|
|
1858
|
+
continue
|
|
1859
|
+
if not _label_matches(header_text, col_label, label_max_diffs):
|
|
1860
|
+
continue
|
|
1861
|
+
label_seen = True
|
|
1862
|
+
if 0 <= data_row_idx < rows:
|
|
1863
|
+
cell_value = str(table.data[data_row_idx, col_idx]).strip()
|
|
1864
|
+
if cell_value:
|
|
1865
|
+
return True, cell_value
|
|
1866
|
+
# Column matched but cell out of range or empty — keep looking
|
|
1867
|
+
# in case a sibling table has the same header populated.
|
|
1868
|
+
|
|
1869
|
+
inline_seen, inline_value = _find_inline_from_to_row_label(content, col_label, row_n, label_max_diffs)
|
|
1870
|
+
if inline_seen:
|
|
1871
|
+
return True, inline_value
|
|
1872
|
+
|
|
1873
|
+
if label_seen:
|
|
1874
|
+
return True, ""
|
|
1875
|
+
return False, None
|
|
1876
|
+
|
|
1877
|
+
|
|
1878
|
+
def _table_anchor_labels_for_cell(table: TableData, row_idx: int, col_idx: int) -> list[str]:
|
|
1879
|
+
"""Return visible row/column labels that identify a table value cell."""
|
|
1880
|
+
|
|
1881
|
+
labels: list[str] = []
|
|
1882
|
+
header_rows = getattr(table, "header_rows", set()) or set()
|
|
1883
|
+
row_headers = getattr(table, "row_headers", {}) or {}
|
|
1884
|
+
|
|
1885
|
+
labels.append(_column_header_for_index(table, col_idx))
|
|
1886
|
+
|
|
1887
|
+
for _, text in row_headers.get(row_idx, []):
|
|
1888
|
+
labels.append(str(text))
|
|
1889
|
+
|
|
1890
|
+
# Keep markdown and simple HTML tables robust when header metadata is
|
|
1891
|
+
# sparse: add the header cells above the value and the cells to its left.
|
|
1892
|
+
for rr in sorted(header_rows):
|
|
1893
|
+
if rr < row_idx and col_idx < table.data.shape[1]:
|
|
1894
|
+
labels.append(str(table.data[rr, col_idx]))
|
|
1895
|
+
for cc in range(col_idx):
|
|
1896
|
+
labels.append(str(table.data[row_idx, cc]))
|
|
1897
|
+
|
|
1898
|
+
return _dedupe_nonempty_text(labels)
|
|
1899
|
+
|
|
1900
|
+
|
|
1901
|
+
def _compact_anchor_label(s: str) -> str:
|
|
1902
|
+
return _compact_label_text_for_distance(s)
|
|
1903
|
+
|
|
1904
|
+
|
|
1905
|
+
def _compact_contains_with_diffs(haystack: str, needle: str, allowed: int) -> bool:
|
|
1906
|
+
"""Return True when any compact substring matches within edit distance."""
|
|
1907
|
+
|
|
1908
|
+
if allowed <= 0 or not haystack or not needle:
|
|
1909
|
+
return False
|
|
1910
|
+
if len(needle) > len(haystack):
|
|
1911
|
+
return _levenshtein_distance_at_most(haystack, needle, allowed) <= allowed
|
|
1912
|
+
|
|
1913
|
+
min_len = max(1, len(needle) - allowed)
|
|
1914
|
+
max_len = min(len(haystack), len(needle) + allowed)
|
|
1915
|
+
for start in range(0, len(haystack) - min_len + 1):
|
|
1916
|
+
for size in range(min_len, max_len + 1):
|
|
1917
|
+
end = start + size
|
|
1918
|
+
if end > len(haystack):
|
|
1919
|
+
break
|
|
1920
|
+
if _levenshtein_distance_at_most(haystack[start:end], needle, allowed) <= allowed:
|
|
1921
|
+
return True
|
|
1922
|
+
return False
|
|
1923
|
+
|
|
1924
|
+
|
|
1925
|
+
def _table_anchor_label_matches(candidate: str, required: str, max_diffs: int | float = 0) -> bool:
|
|
1926
|
+
"""Stricter label match for multi-label table anchors.
|
|
1927
|
+
|
|
1928
|
+
The legacy fuzzy label scorer is intentionally broad for single-key form
|
|
1929
|
+
labels, but it is too broad for sibling columns like PLUG #1 / PLUG #2.
|
|
1930
|
+
Multi-key table lookup needs exact or containment-style evidence instead.
|
|
1931
|
+
"""
|
|
1932
|
+
|
|
1933
|
+
cand = normalize_text(candidate)
|
|
1934
|
+
req = normalize_text(required)
|
|
1935
|
+
if not cand or not req:
|
|
1936
|
+
return False
|
|
1937
|
+
if _strip_label_punct(cand) == _strip_label_punct(req):
|
|
1938
|
+
return True
|
|
1939
|
+
if req in cand or cand in req:
|
|
1940
|
+
return True
|
|
1941
|
+
|
|
1942
|
+
cand_compact = _compact_anchor_label(candidate)
|
|
1943
|
+
req_compact = _compact_anchor_label(required)
|
|
1944
|
+
if not cand_compact or not req_compact:
|
|
1945
|
+
return False
|
|
1946
|
+
if cand_compact == req_compact:
|
|
1947
|
+
return True
|
|
1948
|
+
min_containment_len = 4
|
|
1949
|
+
if (
|
|
1950
|
+
len(req_compact) >= min_containment_len
|
|
1951
|
+
and req_compact in cand_compact
|
|
1952
|
+
or len(cand_compact) >= min_containment_len
|
|
1953
|
+
and cand_compact in req_compact
|
|
1954
|
+
):
|
|
1955
|
+
return True
|
|
1956
|
+
|
|
1957
|
+
allowed = int(max_diffs) if max_diffs and max_diffs > 0 else 0
|
|
1958
|
+
return allowed > 0 and (
|
|
1959
|
+
_levenshtein_distance_at_most(cand_compact, req_compact, allowed) <= allowed
|
|
1960
|
+
or _compact_contains_with_diffs(cand_compact, req_compact, allowed)
|
|
1961
|
+
)
|
|
1962
|
+
|
|
1963
|
+
|
|
1964
|
+
def _labels_match_all(
|
|
1965
|
+
candidate_labels: list[str],
|
|
1966
|
+
required_labels: list[str],
|
|
1967
|
+
label_max_diffs: list[int],
|
|
1968
|
+
) -> bool:
|
|
1969
|
+
"""Order-insensitive match: every required label must match one anchor."""
|
|
1970
|
+
|
|
1971
|
+
for idx, required in enumerate(required_labels):
|
|
1972
|
+
allowed = label_max_diffs[idx] if idx < len(label_max_diffs) else 0
|
|
1973
|
+
if not any(_table_anchor_label_matches(candidate, required, allowed) for candidate in candidate_labels):
|
|
1974
|
+
return False
|
|
1975
|
+
return True
|
|
1976
|
+
|
|
1977
|
+
|
|
1978
|
+
def _is_table_value_cell(table: TableData, row_idx: int, col_idx: int) -> bool:
|
|
1979
|
+
header_rows = getattr(table, "header_rows", set()) or set()
|
|
1980
|
+
header_cols = getattr(table, "header_cols", set()) or set()
|
|
1981
|
+
header_cells = getattr(table, "header_cells", set()) or set()
|
|
1982
|
+
|
|
1983
|
+
if row_idx in header_rows:
|
|
1984
|
+
return False
|
|
1985
|
+
if header_cells:
|
|
1986
|
+
if (row_idx, col_idx) in header_cells:
|
|
1987
|
+
return False
|
|
1988
|
+
elif col_idx in header_cols:
|
|
1989
|
+
return False
|
|
1990
|
+
|
|
1991
|
+
# In row/column form tables, the first data column is normally the row
|
|
1992
|
+
# label stub ("Cementing Date", "Depth...", etc.), not a value cell.
|
|
1993
|
+
if col_idx == 0 and table.data.shape[1] > 1:
|
|
1994
|
+
return False
|
|
1995
|
+
return True
|
|
1996
|
+
|
|
1997
|
+
|
|
1998
|
+
def _find_table_cell_for_label_list(
|
|
1999
|
+
content: str,
|
|
2000
|
+
labels: list[str],
|
|
2001
|
+
expected_values: list[str] | None = None,
|
|
2002
|
+
max_diffs: int | float = 0,
|
|
2003
|
+
label_max_diffs: list[int] | None = None,
|
|
2004
|
+
) -> tuple[bool, str | None]:
|
|
2005
|
+
"""Look up a value cell by multiple row/column-style form labels.
|
|
2006
|
+
|
|
2007
|
+
This is deliberately narrower than full table evaluation: it only asks
|
|
2008
|
+
whether the same table cell is anchored by all requested visible labels.
|
|
2009
|
+
The rule JSON does not need row/column roles; labels are matched as an
|
|
2010
|
+
unordered set against the nearby table anchors.
|
|
2011
|
+
"""
|
|
2012
|
+
|
|
2013
|
+
label_seen = False
|
|
2014
|
+
fallback_value: str | None = None
|
|
2015
|
+
label_diffs = label_max_diffs or [0] * len(labels)
|
|
2016
|
+
for table in parse_html_tables(content) + parse_markdown_tables(content):
|
|
2017
|
+
if table.data.size == 0:
|
|
2018
|
+
continue
|
|
2019
|
+
rows, cols = table.data.shape
|
|
2020
|
+
|
|
2021
|
+
for row_idx in range(rows):
|
|
2022
|
+
for col_idx in range(cols):
|
|
2023
|
+
if not _is_table_value_cell(table, row_idx, col_idx):
|
|
2024
|
+
continue
|
|
2025
|
+
anchors = _table_anchor_labels_for_cell(table, row_idx, col_idx)
|
|
2026
|
+
if not _labels_match_all(anchors, labels, label_diffs):
|
|
2027
|
+
continue
|
|
2028
|
+
|
|
2029
|
+
label_seen = True
|
|
2030
|
+
value = str(table.data[row_idx, col_idx]).strip()
|
|
2031
|
+
if expected_values:
|
|
2032
|
+
for exp in expected_values:
|
|
2033
|
+
if _values_match_text(value, exp, max_diffs):
|
|
2034
|
+
return True, value
|
|
2035
|
+
if fallback_value is None or (not fallback_value and value):
|
|
2036
|
+
fallback_value = value
|
|
2037
|
+
|
|
2038
|
+
if label_seen:
|
|
2039
|
+
return True, fallback_value or ""
|
|
2040
|
+
return False, None
|
|
2041
|
+
|
|
2042
|
+
|
|
2043
|
+
def _scope_to_page(content: str, parse_output, page: int | None) -> str: # type: ignore[no-untyped-def]
|
|
2044
|
+
"""Return per-page markdown when ``parse_output`` and ``page`` are both set.
|
|
2045
|
+
|
|
2046
|
+
Fail-closed: once per-page IR is present (``pages`` or ``layout_pages``),
|
|
2047
|
+
scoping is strict — if the requested page has no entry (or its markdown
|
|
2048
|
+
is empty), return ``""`` rather than the full document. The old lenient
|
|
2049
|
+
fallback let repeated header/footer fields satisfy page-N rules on the
|
|
2050
|
+
wrong page and silently masked page-level extraction failures (see
|
|
2051
|
+
PR #897 for the reducto/extend variant of the same bug).
|
|
2052
|
+
|
|
2053
|
+
Only when no per-page IR is available (both lists empty) do we fall
|
|
2054
|
+
back to the document-level ``content``. Providers that emit neither
|
|
2055
|
+
list never had fair per-page scoring; the fallback preserves prior
|
|
2056
|
+
behavior rather than introducing a silent regression.
|
|
2057
|
+
|
|
2058
|
+
When ``layout_pages`` carries the per-page split but ``md`` is empty,
|
|
2059
|
+
synthesize from ``items`` (priority ``md > html > value``). ``html``
|
|
2060
|
+
ranks above ``value`` so table items keep their structure for the
|
|
2061
|
+
HTML cell-neighbor matcher.
|
|
2062
|
+
|
|
2063
|
+
``parse_output`` is typed as ``ParseOutput`` upstream but kept loose
|
|
2064
|
+
here to avoid an import cycle.
|
|
2065
|
+
"""
|
|
2066
|
+
|
|
2067
|
+
if parse_output is None or page is None:
|
|
2068
|
+
return content
|
|
2069
|
+
|
|
2070
|
+
pages = getattr(parse_output, "pages", None) or []
|
|
2071
|
+
layout_pages = getattr(parse_output, "layout_pages", None) or []
|
|
2072
|
+
|
|
2073
|
+
if not pages and not layout_pages:
|
|
2074
|
+
# Provider produced no per-page IR at all — fall back to full doc.
|
|
2075
|
+
return content
|
|
2076
|
+
|
|
2077
|
+
if pages:
|
|
2078
|
+
for p in pages:
|
|
2079
|
+
# PageIR.page_index is 0-indexed; rule.page is 1-indexed.
|
|
2080
|
+
if getattr(p, "page_index", None) == page - 1:
|
|
2081
|
+
return getattr(p, "markdown", "") or ""
|
|
2082
|
+
# ``pages`` populated but no matching page — fail closed.
|
|
2083
|
+
if not layout_pages:
|
|
2084
|
+
return ""
|
|
2085
|
+
# Fall through to ``layout_pages`` lookup; some providers populate
|
|
2086
|
+
# only one of the two lists per page.
|
|
2087
|
+
|
|
2088
|
+
for lp in layout_pages:
|
|
2089
|
+
if getattr(lp, "page_number", None) != page:
|
|
2090
|
+
continue
|
|
2091
|
+
md = getattr(lp, "md", "") or ""
|
|
2092
|
+
if md:
|
|
2093
|
+
return md
|
|
2094
|
+
# Synthesize from items: md > html > value (html ranks above value
|
|
2095
|
+
# so table items keep their structure for the HTML cell matcher).
|
|
2096
|
+
parts: list[str] = []
|
|
2097
|
+
for it in getattr(lp, "items", None) or []:
|
|
2098
|
+
text = getattr(it, "md", "") or getattr(it, "html", "") or getattr(it, "value", "")
|
|
2099
|
+
if text:
|
|
2100
|
+
parts.append(text)
|
|
2101
|
+
return "\n\n".join(parts)
|
|
2102
|
+
|
|
2103
|
+
# Per-page IR present but page not found — fail closed.
|
|
2104
|
+
return ""
|
|
2105
|
+
|
|
2106
|
+
|
|
2107
|
+
class FormFieldRule(ParseTestRule):
|
|
2108
|
+
"""Test rule for form-field key-value extraction.
|
|
2109
|
+
|
|
2110
|
+
Locates a labeled field by its visible label in the parsed markdown/HTML
|
|
2111
|
+
and checks the extracted value matches the expected one.
|
|
2112
|
+
"""
|
|
2113
|
+
|
|
2114
|
+
def __init__(self, rule_data: ParseFormFieldRule | dict):
|
|
2115
|
+
super().__init__(rule_data)
|
|
2116
|
+
rule_data = cast(ParseFormFieldRule, self._rule_data)
|
|
2117
|
+
|
|
2118
|
+
if self.type != TestType.FORM_FIELD.value:
|
|
2119
|
+
raise ValueError(f"Invalid type for FormFieldRule: {self.type}")
|
|
2120
|
+
|
|
2121
|
+
self.label = rule_data.label
|
|
2122
|
+
label_parts = _label_parts_with_indexes(rule_data.label)
|
|
2123
|
+
self.labels = [part for part, _ in label_parts]
|
|
2124
|
+
label_indexes = [idx for _, idx in label_parts]
|
|
2125
|
+
self.label_max_diffs = _label_max_diffs_parts(
|
|
2126
|
+
rule_data.label_max_diffs,
|
|
2127
|
+
len(self.labels),
|
|
2128
|
+
label_indexes,
|
|
2129
|
+
)
|
|
2130
|
+
self.value_max_diffs = rule_data.value_max_diffs
|
|
2131
|
+
self.value = rule_data.value
|
|
2132
|
+
self.value_type = rule_data.value_type
|
|
2133
|
+
|
|
2134
|
+
if not self.labels:
|
|
2135
|
+
raise ValueError("label field cannot be empty")
|
|
2136
|
+
|
|
2137
|
+
def _content_for_match(self, md_content: str) -> str:
|
|
2138
|
+
return _scope_to_page(md_content, self.parse_output, self.page)
|
|
2139
|
+
|
|
2140
|
+
def run(
|
|
2141
|
+
self,
|
|
2142
|
+
md_content: str,
|
|
2143
|
+
normalized_content: str | None = None,
|
|
2144
|
+
) -> tuple[bool, str, float]:
|
|
2145
|
+
scoped = self._content_for_match(md_content)
|
|
2146
|
+
if self.value_type == "text":
|
|
2147
|
+
return self._run_text(scoped)
|
|
2148
|
+
if self.value_type == "checkbox":
|
|
2149
|
+
return self._run_checkbox(scoped)
|
|
2150
|
+
if self.value_type == "signature":
|
|
2151
|
+
return self._run_signature(scoped)
|
|
2152
|
+
return False, f"unknown value_type: {self.value_type}", 0.0
|
|
2153
|
+
|
|
2154
|
+
def _run_text(self, content: str) -> tuple[bool, str, float]:
|
|
2155
|
+
# `self.value` can be a list of acceptable alternatives — pass if any matches.
|
|
2156
|
+
expected_alternatives = _value_alternatives(self.value)
|
|
2157
|
+
label = self.labels[0]
|
|
2158
|
+
# Multi-col table cell lookup ("Column Name (row N)") takes precedence
|
|
2159
|
+
# over the bold-colon / 2-col / glyph paths because the suffix
|
|
2160
|
+
# explicitly names a tabular position.
|
|
2161
|
+
if len(self.labels) > 1:
|
|
2162
|
+
label_found, value = _find_table_cell_for_label_list(
|
|
2163
|
+
content,
|
|
2164
|
+
self.labels,
|
|
2165
|
+
expected_alternatives,
|
|
2166
|
+
self.value_max_diffs,
|
|
2167
|
+
self.label_max_diffs,
|
|
2168
|
+
)
|
|
2169
|
+
elif _split_row_label(label) is not None:
|
|
2170
|
+
label_found, value = _find_table_cell_for_row_label(content, label, self.label_max_diffs[0])
|
|
2171
|
+
else:
|
|
2172
|
+
# Thread expected_alternatives so the matcher can disambiguate
|
|
2173
|
+
# among candidates at the same top label-match score. The rule's
|
|
2174
|
+
# position in the test list says nothing about which page
|
|
2175
|
+
# occurrence the GT refers to when a label legitimately repeats
|
|
2176
|
+
# (e.g. ``KB`` appearing as both an elevation label and as the
|
|
2177
|
+
# value of ``LOG MEASURED FROM`` in well-log headers). The GT
|
|
2178
|
+
# oracle is applied **only** to candidates tied at the head
|
|
2179
|
+
# label-match score, so it never leaks across adjacent labels
|
|
2180
|
+
# like Country vs County which live at different score levels.
|
|
2181
|
+
label_found, value = _find_text_value_for_label(
|
|
2182
|
+
content,
|
|
2183
|
+
label,
|
|
2184
|
+
expected_alternatives,
|
|
2185
|
+
self.value_max_diffs,
|
|
2186
|
+
self.label_max_diffs[0],
|
|
2187
|
+
)
|
|
2188
|
+
if not label_found:
|
|
2189
|
+
return False, f"label not found: {_format_label_for_message(self.label)}", 0.0
|
|
2190
|
+
# value may be "" (label found, cell/value empty); _values_match_text
|
|
2191
|
+
# handles empty == empty correctly so empty-expected rules can pass.
|
|
2192
|
+
for expected in expected_alternatives:
|
|
2193
|
+
if _values_match_text(value or "", expected, self.value_max_diffs):
|
|
2194
|
+
return True, "match", 1.0
|
|
2195
|
+
if len(expected_alternatives) == 1:
|
|
2196
|
+
return False, f"expected {expected_alternatives[0]!r}, got {(value or '')!r}", 0.0
|
|
2197
|
+
return False, f"expected any of {expected_alternatives!r}, got {(value or '')!r}", 0.0
|
|
2198
|
+
|
|
2199
|
+
def _run_checkbox(self, content: str) -> tuple[bool, str, float]:
|
|
2200
|
+
expected_bool = _coerce_bool(self.value)
|
|
2201
|
+
if expected_bool is None:
|
|
2202
|
+
return False, f"checkbox value must be coercible to bool, got {self.value!r}", 0.0
|
|
2203
|
+
|
|
2204
|
+
# Prefer a real checkbox-shaped match.
|
|
2205
|
+
state = (
|
|
2206
|
+
_find_checkbox_state_for_label(content, self.labels[0], self.label_max_diffs[0])
|
|
2207
|
+
if len(self.labels) == 1
|
|
2208
|
+
else _find_checkbox_state_for_label_list(content, self.labels, self.label_max_diffs)
|
|
2209
|
+
)
|
|
2210
|
+
if state is None:
|
|
2211
|
+
# Fall back to a text-shaped value (e.g. **Married:** Yes / No).
|
|
2212
|
+
if len(self.labels) > 1:
|
|
2213
|
+
label_found, text_value = _find_table_cell_for_label_list(
|
|
2214
|
+
content,
|
|
2215
|
+
self.labels,
|
|
2216
|
+
label_max_diffs=self.label_max_diffs,
|
|
2217
|
+
)
|
|
2218
|
+
elif _split_row_label(self.labels[0]) is not None:
|
|
2219
|
+
label_found, text_value = _find_table_cell_for_row_label(
|
|
2220
|
+
content,
|
|
2221
|
+
self.labels[0],
|
|
2222
|
+
self.label_max_diffs[0],
|
|
2223
|
+
)
|
|
2224
|
+
else:
|
|
2225
|
+
label_found, text_value = _find_text_value_for_label(
|
|
2226
|
+
content,
|
|
2227
|
+
self.labels[0],
|
|
2228
|
+
label_max_diffs=self.label_max_diffs[0],
|
|
2229
|
+
)
|
|
2230
|
+
if not label_found or not text_value:
|
|
2231
|
+
return False, f"label not found: {_format_label_for_message(self.label)}", 0.0
|
|
2232
|
+
state = _coerce_bool(text_value)
|
|
2233
|
+
if state is None:
|
|
2234
|
+
return False, f"could not interpret {text_value!r} as checkbox state", 0.0
|
|
2235
|
+
|
|
2236
|
+
if state == expected_bool:
|
|
2237
|
+
return True, "match", 1.0
|
|
2238
|
+
return False, f"expected {expected_bool}, got {state}", 0.0
|
|
2239
|
+
|
|
2240
|
+
def _run_signature(self, content: str) -> tuple[bool, str, float]:
|
|
2241
|
+
# Relaxed semantics: the rule's value is treated as a presence indicator,
|
|
2242
|
+
# not a strict bool. A non-empty string (e.g. the actual signed name) is
|
|
2243
|
+
# equivalent to True — the matcher only checks "is something signed here"
|
|
2244
|
+
# rather than the exact handwriting. An empty string / False / None means
|
|
2245
|
+
# "expected unsigned". A list value collapses the same way: any non-empty
|
|
2246
|
+
# alternative means "expected signed".
|
|
2247
|
+
if isinstance(self.value, bool):
|
|
2248
|
+
expected_signed = self.value
|
|
2249
|
+
elif isinstance(self.value, list):
|
|
2250
|
+
expected_signed = any(bool(str(v).strip()) for v in self.value)
|
|
2251
|
+
else:
|
|
2252
|
+
expected_signed = bool(str(self.value).strip())
|
|
2253
|
+
|
|
2254
|
+
# Track label presence separately from value presence — an absent label
|
|
2255
|
+
# must NOT pass an "expected unsigned" rule. A form tuple benchmark
|
|
2256
|
+
# requires the parser to surface the field at all.
|
|
2257
|
+
if len(self.labels) > 1:
|
|
2258
|
+
label_found, text_value = _find_table_cell_for_label_list(
|
|
2259
|
+
content,
|
|
2260
|
+
self.labels,
|
|
2261
|
+
label_max_diffs=self.label_max_diffs,
|
|
2262
|
+
)
|
|
2263
|
+
else:
|
|
2264
|
+
label_found, text_value = _find_text_value_for_label(
|
|
2265
|
+
content,
|
|
2266
|
+
self.labels[0],
|
|
2267
|
+
label_max_diffs=self.label_max_diffs[0],
|
|
2268
|
+
)
|
|
2269
|
+
if not label_found:
|
|
2270
|
+
return False, f"label not found: {_format_label_for_message(self.label)}", 0.0
|
|
2271
|
+
signed = bool(text_value and text_value.strip())
|
|
2272
|
+
if signed == expected_signed:
|
|
2273
|
+
return True, "match", 1.0
|
|
2274
|
+
return False, f"expected signed={expected_signed}, got signed={signed}", 0.0
|