parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
"""Page-decoration rule: running header, running footer and printed page number, per page.
|
|
2
|
+
|
|
3
|
+
LlamaParse lifts page furniture out of the body markdown into per-page structured fields
|
|
4
|
+
(``page_header_markdown``, ``page_footer_markdown``, ``printed_page_number``); the prompts ask
|
|
5
|
+
for ``<page_header>`` / ``<page_footer>`` / ``<page_number>`` tags which post-processing removes.
|
|
6
|
+
This rule scores that contract on one page with four axes:
|
|
7
|
+
|
|
8
|
+
* ``header`` / ``footer`` — predicted text vs annotated text after normalisation (markdown and
|
|
9
|
+
punctuation stripped, whitespace collapsed, lowercase) with ``token_set_ratio`` so that the
|
|
10
|
+
left / centre / right pieces of a header may come in any order. An annotated ``None`` means
|
|
11
|
+
the slot must be empty: a hallucinated header on a cover page fails.
|
|
12
|
+
* ``page_number`` — both sides canonicalised (arabic digits, lowercase roman, ``Page 3 of 10`` →
|
|
13
|
+
``3``, ``– 3 –`` → ``3``, compound ids such as ``A-3`` kept) and compared exactly; a number returned
|
|
14
|
+
inside the header/footer text instead of the dedicated field also counts.
|
|
15
|
+
* ``leak`` — none of the annotated furniture strings may remain in the body markdown; the body
|
|
16
|
+
is the page markdown with any ``<page_*>`` tags removed.
|
|
17
|
+
|
|
18
|
+
Prediction source: the structured page fields when ``parse_output.layout_pages`` is available,
|
|
19
|
+
else the markdown tags — the same preference order as the ``is_header`` / ``is_footer`` rules.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import re
|
|
25
|
+
from typing import Any, cast
|
|
26
|
+
|
|
27
|
+
import numpy as np
|
|
28
|
+
from rapidfuzz import fuzz
|
|
29
|
+
|
|
30
|
+
from parse_bench.evaluation.metrics.parse.rules_base import ParseTestRule
|
|
31
|
+
from parse_bench.evaluation.metrics.parse.test_types import TestType
|
|
32
|
+
from parse_bench.evaluation.metrics.parse.utils import normalize_text
|
|
33
|
+
from parse_bench.test_cases.parse_rule_schemas import ParsePageDecorationRule
|
|
34
|
+
|
|
35
|
+
_TAG_RE = {
|
|
36
|
+
"header": re.compile(r"<page_header>(.*?)</page_header>", re.S | re.I),
|
|
37
|
+
"footer": re.compile(r"<page_footer>(.*?)</page_footer>", re.S | re.I),
|
|
38
|
+
"page_number": re.compile(r"<page_number>(.*?)</page_number>", re.S | re.I),
|
|
39
|
+
}
|
|
40
|
+
_ANY_TAG_RE = re.compile(r"</?page_(?:header|footer|number)>", re.I)
|
|
41
|
+
_ROMAN_RE = re.compile(r"^[ivxlcdm]+$")
|
|
42
|
+
_ROMAN_VALUES = {"i": 1, "v": 5, "x": 10, "l": 50, "c": 100, "d": 500, "m": 1000}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def norm_furniture(text: str | None) -> str:
|
|
46
|
+
"""Comparison form of a header/footer: bench text normalisation, no punctuation, one space."""
|
|
47
|
+
s = normalize_text(text or "").lower()
|
|
48
|
+
s = re.sub(r"[|•·–—\-_/\\,:;.()\[\]{}\"'`*#>]+", " ", s)
|
|
49
|
+
return re.sub(r"\s+", " ", s).strip()
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def roman_to_int(s: str) -> int | None:
|
|
53
|
+
s = s.lower()
|
|
54
|
+
if not s or not _ROMAN_RE.match(s):
|
|
55
|
+
return None
|
|
56
|
+
total, prev = 0, 0
|
|
57
|
+
for ch in reversed(s):
|
|
58
|
+
v = _ROMAN_VALUES[ch]
|
|
59
|
+
total = total - v if v < prev else total + v
|
|
60
|
+
prev = max(prev, v)
|
|
61
|
+
return total
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def canonical_page_number(raw: str | None) -> str | None:
|
|
65
|
+
"""``Page 12 of 48`` → ``12``; ``– iv –`` → ``iv``; ``A-3`` → ``a-3``; ``3 / 10`` → ``3``."""
|
|
66
|
+
if raw is None:
|
|
67
|
+
return None
|
|
68
|
+
s = normalize_text(str(raw)).strip().lower()
|
|
69
|
+
s = re.sub(r"^(page|p\.?|pg\.?|seite|página|pagina)\s*", "", s)
|
|
70
|
+
s = re.sub(r"\s*(of|/|sur|de|von)\s*\d+\s*$", "", s)
|
|
71
|
+
s = s.strip(" -–—|.·•")
|
|
72
|
+
if not s:
|
|
73
|
+
return None
|
|
74
|
+
m = re.fullmatch(r"0*(\d+)", s)
|
|
75
|
+
if m:
|
|
76
|
+
return str(int(m.group(1)))
|
|
77
|
+
if _ROMAN_RE.match(s):
|
|
78
|
+
return s
|
|
79
|
+
m = re.fullmatch(r"([a-z]{1,3})[\s\-–.]?0*(\d+)", s)
|
|
80
|
+
if m:
|
|
81
|
+
return f"{m.group(1)}-{int(m.group(2))}"
|
|
82
|
+
return s
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def page_number_equal(a: str | None, b: str | None) -> bool:
|
|
86
|
+
ca, cb = canonical_page_number(a), canonical_page_number(b)
|
|
87
|
+
if ca is None or cb is None:
|
|
88
|
+
return ca == cb
|
|
89
|
+
if ca == cb:
|
|
90
|
+
return True
|
|
91
|
+
ra, rb = roman_to_int(ca), roman_to_int(cb)
|
|
92
|
+
# ``iv`` printed vs ``4`` predicted (or the reverse) counts as the same number.
|
|
93
|
+
return ra is not None and str(ra) == cb or rb is not None and str(rb) == ca
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _page_markdown(md_content: str, page: int | None) -> str:
|
|
97
|
+
if page is None or "\f" not in md_content:
|
|
98
|
+
return md_content
|
|
99
|
+
parts = md_content.split("\f")
|
|
100
|
+
if 1 <= page <= len(parts):
|
|
101
|
+
return parts[page - 1]
|
|
102
|
+
return md_content
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
class PageDecorationRule(ParseTestRule):
|
|
106
|
+
"""Header / footer / printed page number for one page, plus leakage into the body."""
|
|
107
|
+
|
|
108
|
+
def __init__(self, rule_data: ParsePageDecorationRule | dict):
|
|
109
|
+
super().__init__(rule_data)
|
|
110
|
+
if self.type != TestType.PAGE_DECORATION.value:
|
|
111
|
+
raise ValueError(f"Invalid type for PageDecorationRule: {self.type}")
|
|
112
|
+
self.rule = cast(ParsePageDecorationRule, self._rule_data)
|
|
113
|
+
|
|
114
|
+
def _predicted(self, page_md: str) -> tuple[dict[str, str | None], str]:
|
|
115
|
+
"""Predicted slots and the source they came from (``structured`` or ``markdown_tags``)."""
|
|
116
|
+
if self.parse_output is not None and self.parse_output.layout_pages:
|
|
117
|
+
pages = self.parse_output.layout_pages
|
|
118
|
+
if self.page is not None:
|
|
119
|
+
pages = [p for p in pages if p.page_number == self.page] or pages[:1]
|
|
120
|
+
if pages:
|
|
121
|
+
p = pages[0]
|
|
122
|
+
return {
|
|
123
|
+
"header": p.page_header_markdown or None,
|
|
124
|
+
"footer": p.page_footer_markdown or None,
|
|
125
|
+
"page_number": p.printed_page_number or None,
|
|
126
|
+
}, "structured"
|
|
127
|
+
out: dict[str, str | None] = {}
|
|
128
|
+
for slot, pattern in _TAG_RE.items():
|
|
129
|
+
found = [m.group(1).strip() for m in pattern.finditer(page_md)]
|
|
130
|
+
out[slot] = " | ".join(f for f in found if f) or None
|
|
131
|
+
return out, "markdown_tags"
|
|
132
|
+
|
|
133
|
+
def _text_axis(
|
|
134
|
+
self, expected: str | None, predicted: str | None, threshold: int, ignore: set[str] | None = None
|
|
135
|
+
) -> dict[str, Any]:
|
|
136
|
+
ne, npd = norm_furniture(expected), norm_furniture(predicted)
|
|
137
|
+
if ignore:
|
|
138
|
+
# The printed page number may sit in the header/footer line on either side; it is scored by
|
|
139
|
+
# its own axis and must not cost precision or recall here.
|
|
140
|
+
ne = " ".join(t for t in ne.split() if t not in ignore)
|
|
141
|
+
npd = " ".join(t for t in npd.split() if t not in ignore)
|
|
142
|
+
if not ne and not npd:
|
|
143
|
+
return {
|
|
144
|
+
"passed": True,
|
|
145
|
+
"score": 1.0,
|
|
146
|
+
"expected": expected,
|
|
147
|
+
"predicted": predicted,
|
|
148
|
+
"reason": "none expected, none predicted",
|
|
149
|
+
}
|
|
150
|
+
if not ne:
|
|
151
|
+
return {
|
|
152
|
+
"passed": False,
|
|
153
|
+
"score": 0.0,
|
|
154
|
+
"expected": expected,
|
|
155
|
+
"predicted": predicted,
|
|
156
|
+
"reason": "hallucinated",
|
|
157
|
+
}
|
|
158
|
+
if not npd:
|
|
159
|
+
return {"passed": False, "score": 0.0, "expected": expected, "predicted": predicted, "reason": "missing"}
|
|
160
|
+
# Token-level F1 with fuzzy token matching: order-free (header pieces may be reordered), but a
|
|
161
|
+
# dropped piece lowers recall and body text swept into the header lowers precision. A plain
|
|
162
|
+
# token_set_ratio would score a strict subset 100 and hide missing pieces.
|
|
163
|
+
exp_tokens, pred_tokens = ne.split(), npd.split()
|
|
164
|
+
matched_pred: set[int] = set()
|
|
165
|
+
hits = 0
|
|
166
|
+
for tok in exp_tokens:
|
|
167
|
+
best, best_j = 0.0, -1
|
|
168
|
+
for j, cand in enumerate(pred_tokens):
|
|
169
|
+
if j in matched_pred:
|
|
170
|
+
continue
|
|
171
|
+
score = 100.0 if tok == cand else float(fuzz.ratio(tok, cand))
|
|
172
|
+
if score > best:
|
|
173
|
+
best, best_j = score, j
|
|
174
|
+
if best >= 85 and best_j >= 0:
|
|
175
|
+
matched_pred.add(best_j)
|
|
176
|
+
hits += 1
|
|
177
|
+
recall = hits / len(exp_tokens)
|
|
178
|
+
precision = hits / len(pred_tokens) if pred_tokens else 0.0
|
|
179
|
+
f1 = 2 * precision * recall / (precision + recall) if (precision + recall) else 0.0
|
|
180
|
+
# Both directions must clear the threshold: F1 alone would forgive one dropped piece in four.
|
|
181
|
+
return {
|
|
182
|
+
"passed": recall * 100 >= threshold and precision * 100 >= threshold,
|
|
183
|
+
"score": round(f1, 4),
|
|
184
|
+
"expected": expected,
|
|
185
|
+
"predicted": predicted,
|
|
186
|
+
"recall": round(recall, 3),
|
|
187
|
+
"precision": round(precision, 3),
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
def run(self, md_content: str, normalized_content: str | None = None) -> tuple[bool, str, float]:
|
|
191
|
+
page_md = _page_markdown(md_content, self.page)
|
|
192
|
+
predicted, source = self._predicted(page_md)
|
|
193
|
+
r = self.rule
|
|
194
|
+
page_tokens = {
|
|
195
|
+
t for t in (canonical_page_number(r.page_number), canonical_page_number(predicted["page_number"])) if t
|
|
196
|
+
}
|
|
197
|
+
axes: dict[str, dict[str, Any]] = {
|
|
198
|
+
"header": self._text_axis(r.header, predicted["header"], r.text_threshold, page_tokens),
|
|
199
|
+
"footer": self._text_axis(r.footer, predicted["footer"], r.text_threshold, page_tokens),
|
|
200
|
+
}
|
|
201
|
+
pn_field_ok = page_number_equal(r.page_number, predicted["page_number"])
|
|
202
|
+
# A page number printed inside the running header or footer is legitimately returned as part of
|
|
203
|
+
# that text; the dedicated field is preferred but not required. It counts when the canonical
|
|
204
|
+
# number appears as a whole token in the predicted header/footer.
|
|
205
|
+
in_furniture = False
|
|
206
|
+
if not pn_field_ok and r.page_number and not predicted["page_number"]:
|
|
207
|
+
furniture = norm_furniture(" ".join(v for v in (predicted["header"], predicted["footer"]) if v))
|
|
208
|
+
canon = canonical_page_number(r.page_number) or ""
|
|
209
|
+
in_furniture = bool(canon) and re.search(rf"(?<!\w){re.escape(canon)}(?!\w)", furniture) is not None
|
|
210
|
+
pn_ok = pn_field_ok or in_furniture
|
|
211
|
+
axes["page_number"] = {
|
|
212
|
+
"passed": pn_ok,
|
|
213
|
+
"score": 1.0 if pn_ok else 0.0,
|
|
214
|
+
"expected": r.page_number,
|
|
215
|
+
"predicted": predicted["page_number"],
|
|
216
|
+
"expected_canonical": canonical_page_number(r.page_number),
|
|
217
|
+
"predicted_canonical": canonical_page_number(predicted["page_number"]),
|
|
218
|
+
"in_furniture_text": in_furniture,
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
# leak: annotated furniture must not remain in the body
|
|
222
|
+
body = page_md
|
|
223
|
+
for pattern in _TAG_RE.values(): # tagged furniture is not body
|
|
224
|
+
body = pattern.sub(" ", body)
|
|
225
|
+
body = _ANY_TAG_RE.sub(" ", body) # then any unbalanced stray tag
|
|
226
|
+
body_norm = " " + norm_furniture(body) + " "
|
|
227
|
+
leaked: list[str] = []
|
|
228
|
+
checked = 0
|
|
229
|
+
for slot in ("header", "footer"):
|
|
230
|
+
value = getattr(r, slot)
|
|
231
|
+
if value:
|
|
232
|
+
pieces = [p for p in re.split(r"\s*\|\s*", value) if len(norm_furniture(p)) >= r.leak_min_chars]
|
|
233
|
+
for piece in pieces:
|
|
234
|
+
checked += 1
|
|
235
|
+
if norm_furniture(piece) in body_norm:
|
|
236
|
+
leaked.append(piece)
|
|
237
|
+
if r.page_number_raw or r.page_number:
|
|
238
|
+
raw = norm_furniture(r.page_number_raw or r.page_number)
|
|
239
|
+
if raw:
|
|
240
|
+
checked += 1
|
|
241
|
+
if re.search(rf"(?<!\S){re.escape(raw)}(?!\S)", body_norm):
|
|
242
|
+
leaked.append(r.page_number_raw or r.page_number or "")
|
|
243
|
+
if checked == 0:
|
|
244
|
+
axes["leak"] = {"passed": None, "score": None, "reason": "no furniture expected"}
|
|
245
|
+
else:
|
|
246
|
+
axes["leak"] = {
|
|
247
|
+
"passed": not leaked,
|
|
248
|
+
"score": round(1.0 - len(leaked) / checked, 4),
|
|
249
|
+
"leaked": leaked,
|
|
250
|
+
"checked": checked,
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
self.result_details = {
|
|
254
|
+
"source": source,
|
|
255
|
+
"expected": {
|
|
256
|
+
"header": r.header,
|
|
257
|
+
"footer": r.footer,
|
|
258
|
+
"page_number": r.page_number,
|
|
259
|
+
"page_number_raw": r.page_number_raw,
|
|
260
|
+
},
|
|
261
|
+
"predicted": predicted,
|
|
262
|
+
"axes": axes,
|
|
263
|
+
}
|
|
264
|
+
applicable = [a for a in axes.values() if a.get("passed") is not None]
|
|
265
|
+
passed = all(a["passed"] for a in applicable)
|
|
266
|
+
score = float(np.mean([a["score"] for a in applicable])) if applicable else 1.0
|
|
267
|
+
failed = [k for k, a in axes.items() if a.get("passed") is False]
|
|
268
|
+
|
|
269
|
+
def verdict(k: str) -> str:
|
|
270
|
+
if axes[k].get("passed"):
|
|
271
|
+
return "ok (in header/footer text)" if k == "page_number" and axes[k].get("in_furniture_text") else "ok"
|
|
272
|
+
return axes[k].get("reason") or "mismatch"
|
|
273
|
+
|
|
274
|
+
summary = ", ".join(f"{k}:{verdict(k)}" for k in ("header", "footer", "page_number"))
|
|
275
|
+
expl = f"[{source}] {summary}" + (f"; failed: {', '.join(failed)}" if failed else "; all axes pass")
|
|
276
|
+
return passed, expl, round(score, 4)
|