parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,340 @@
|
|
|
1
|
+
"""Text presence, order, and baseline test rules."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from typing import cast
|
|
5
|
+
|
|
6
|
+
from fuzzysearch import find_near_matches
|
|
7
|
+
from rapidfuzz import fuzz
|
|
8
|
+
|
|
9
|
+
from parse_bench.evaluation.metrics.parse.rules_base import (
|
|
10
|
+
ParseTestRule,
|
|
11
|
+
_strip_and_replace_latex,
|
|
12
|
+
)
|
|
13
|
+
from parse_bench.evaluation.metrics.parse.test_types import TestType
|
|
14
|
+
from parse_bench.evaluation.metrics.parse.utils import normalize_text, normalize_text_light
|
|
15
|
+
from parse_bench.test_cases.parse_rule_schemas import (
|
|
16
|
+
ParseBaselineRule,
|
|
17
|
+
ParseOrderRule,
|
|
18
|
+
ParsePresenceRule,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class TextPresenceRule(ParseTestRule):
|
|
23
|
+
"""Test rule for text presence/absence."""
|
|
24
|
+
|
|
25
|
+
def __init__(self, rule_data: ParsePresenceRule | dict):
|
|
26
|
+
super().__init__(rule_data)
|
|
27
|
+
rule_data = cast(ParsePresenceRule, self._rule_data)
|
|
28
|
+
|
|
29
|
+
if self.type not in {TestType.PRESENT.value, TestType.ABSENT.value}:
|
|
30
|
+
raise ValueError(f"Invalid type for TextPresenceRule: {self.type}")
|
|
31
|
+
|
|
32
|
+
# Check if we should use light normalization that preserves formatting tags
|
|
33
|
+
self.keep_formatting = rule_data.keep_formatting_text_normalisation
|
|
34
|
+
normalize_fn = normalize_text_light if self.keep_formatting else normalize_text
|
|
35
|
+
|
|
36
|
+
self.text = normalize_fn(rule_data.text)
|
|
37
|
+
if not self.text.strip():
|
|
38
|
+
raise ValueError("Text field cannot be empty")
|
|
39
|
+
|
|
40
|
+
self.case_sensitive = rule_data.case_sensitive
|
|
41
|
+
self.first_n = rule_data.first_n
|
|
42
|
+
self.last_n = rule_data.last_n
|
|
43
|
+
self.count = rule_data.count
|
|
44
|
+
|
|
45
|
+
if self.count is not None:
|
|
46
|
+
if not isinstance(self.count, int):
|
|
47
|
+
raise ValueError("Count field must be an integer when provided")
|
|
48
|
+
if self.count < 0:
|
|
49
|
+
raise ValueError("Count field cannot be negative")
|
|
50
|
+
|
|
51
|
+
def _count_non_overlapping_fuzzy_matches(self, query: str, content: str) -> int:
|
|
52
|
+
"""Count non-overlapping fuzzy matches for query in content.
|
|
53
|
+
|
|
54
|
+
We keep matches non-overlapping to avoid over-counting near-duplicate
|
|
55
|
+
windows for the same textual occurrence.
|
|
56
|
+
"""
|
|
57
|
+
max_distance = min(self.max_diffs, 15)
|
|
58
|
+
matches = sorted(
|
|
59
|
+
find_near_matches(query, content, max_l_dist=max_distance),
|
|
60
|
+
key=lambda match: (match.start, match.end),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
non_overlapping_count = 0
|
|
64
|
+
last_end = -1
|
|
65
|
+
for match in matches:
|
|
66
|
+
if match.start >= last_end:
|
|
67
|
+
non_overlapping_count += 1
|
|
68
|
+
last_end = match.end
|
|
69
|
+
return non_overlapping_count
|
|
70
|
+
|
|
71
|
+
def run(self, md_content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
72
|
+
"""Check if text is present or absent in markdown."""
|
|
73
|
+
reference_query = self.text
|
|
74
|
+
|
|
75
|
+
# When keep_formatting is enabled, we must re-normalize content with light normalization
|
|
76
|
+
# since the pre-normalized content from the metric uses standard normalization
|
|
77
|
+
if self.keep_formatting:
|
|
78
|
+
# Always use light normalization when testing formatting
|
|
79
|
+
normalized_content = normalize_text_light(md_content)
|
|
80
|
+
# Backward compatibility: standard normalize_text lowercases by default,
|
|
81
|
+
# so keep-formatting mode mirrors that behavior unless case sensitivity
|
|
82
|
+
# is explicitly disabled below.
|
|
83
|
+
reference_query = reference_query.lower()
|
|
84
|
+
normalized_content = normalized_content.lower()
|
|
85
|
+
elif normalized_content is None:
|
|
86
|
+
# Use pre-normalized content if provided, otherwise normalize
|
|
87
|
+
normalized_content = normalize_text(md_content)
|
|
88
|
+
|
|
89
|
+
if not self.case_sensitive:
|
|
90
|
+
reference_query = reference_query.lower()
|
|
91
|
+
normalized_content = normalized_content.lower()
|
|
92
|
+
|
|
93
|
+
# Apply first_n/last_n if specified
|
|
94
|
+
if self.first_n and self.last_n:
|
|
95
|
+
normalized_content = normalized_content[: self.first_n] + normalized_content[-self.last_n :]
|
|
96
|
+
elif self.first_n:
|
|
97
|
+
normalized_content = normalized_content[: self.first_n]
|
|
98
|
+
elif self.last_n:
|
|
99
|
+
normalized_content = normalized_content[-self.last_n :]
|
|
100
|
+
|
|
101
|
+
# Threshold for fuzzy matching derived from max_diffs.
|
|
102
|
+
# Floor at 0.7 so short queries (e.g. 2-3 chars) don't produce
|
|
103
|
+
# wildly permissive thresholds that match almost anything.
|
|
104
|
+
raw_threshold = 1.0 - (self.max_diffs / (len(reference_query) if len(reference_query) > 0 else 1))
|
|
105
|
+
threshold = max(0.7, min(1.0, raw_threshold))
|
|
106
|
+
best_ratio = fuzz.partial_ratio(reference_query, normalized_content) / 100.0
|
|
107
|
+
|
|
108
|
+
if self.type == TestType.PRESENT.value:
|
|
109
|
+
# Backward compatibility: count=None or count=0 keeps legacy behavior
|
|
110
|
+
# (presence check only, regardless of number of occurrences).
|
|
111
|
+
if self.count not in {None, 0}:
|
|
112
|
+
if self.max_diffs == 0:
|
|
113
|
+
actual_count = normalized_content.count(reference_query)
|
|
114
|
+
else:
|
|
115
|
+
actual_count = self._count_non_overlapping_fuzzy_matches(reference_query, normalized_content)
|
|
116
|
+
|
|
117
|
+
if actual_count == self.count:
|
|
118
|
+
return True, ""
|
|
119
|
+
msg = f"Expected '{reference_query[:40]}...' exactly {self.count} time(s), but found {actual_count}"
|
|
120
|
+
return False, msg
|
|
121
|
+
|
|
122
|
+
if best_ratio >= threshold:
|
|
123
|
+
return True, ""
|
|
124
|
+
else:
|
|
125
|
+
msg = (
|
|
126
|
+
f"Expected '{reference_query[:40]}...' with threshold {threshold} "
|
|
127
|
+
f"but best match ratio was {best_ratio:.3f}"
|
|
128
|
+
)
|
|
129
|
+
return False, msg
|
|
130
|
+
else: # ABSENT
|
|
131
|
+
if best_ratio < threshold:
|
|
132
|
+
return True, ""
|
|
133
|
+
else:
|
|
134
|
+
msg = (
|
|
135
|
+
f"Expected absence of '{reference_query[:40]}...' with threshold {threshold} "
|
|
136
|
+
f"but best match ratio was {best_ratio:.3f}"
|
|
137
|
+
)
|
|
138
|
+
return False, msg
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class BaselineRule(ParseTestRule):
|
|
142
|
+
"""Test rule for baseline quality checks (blank pages, repeats, character sets)."""
|
|
143
|
+
|
|
144
|
+
def __init__(self, rule_data: ParseBaselineRule | dict):
|
|
145
|
+
super().__init__(rule_data)
|
|
146
|
+
rule_data = cast(ParseBaselineRule, self._rule_data)
|
|
147
|
+
|
|
148
|
+
self.max_length = rule_data.max_length
|
|
149
|
+
self.max_length_skips_image_alt_tags = rule_data.max_length_skips_image_alt_tags
|
|
150
|
+
self.max_repeats = rule_data.max_repeats
|
|
151
|
+
self.check_disallowed_characters = rule_data.check_disallowed_characters
|
|
152
|
+
|
|
153
|
+
def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
154
|
+
"""Run baseline quality checks."""
|
|
155
|
+
base_content_len = len("".join(c for c in content if c.isalnum()).strip())
|
|
156
|
+
|
|
157
|
+
# Blank page check
|
|
158
|
+
if self.max_length is not None:
|
|
159
|
+
if self.max_length_skips_image_alt_tags:
|
|
160
|
+
# Remove markdown image tags
|
|
161
|
+
content_for_length_check = re.sub(r"!\[.*?\]\(.*?\)", "", content)
|
|
162
|
+
base_content_len = len("".join(c for c in content_for_length_check if c.isalnum()).strip())
|
|
163
|
+
|
|
164
|
+
if base_content_len > self.max_length:
|
|
165
|
+
return (
|
|
166
|
+
False,
|
|
167
|
+
f"{base_content_len} characters were output for a page we expected to be blank",
|
|
168
|
+
)
|
|
169
|
+
else:
|
|
170
|
+
return True, ""
|
|
171
|
+
|
|
172
|
+
# Check for empty content
|
|
173
|
+
if base_content_len == 0:
|
|
174
|
+
return False, "The text contains no alpha numeric characters"
|
|
175
|
+
|
|
176
|
+
# Check for excessive repetition using sampled windows across the
|
|
177
|
+
# document. Sampling keeps this O(1) even on very large documents.
|
|
178
|
+
if len(content) > 50:
|
|
179
|
+
# Sample up to 5 windows of 200 chars evenly spaced through the doc
|
|
180
|
+
sample_size = min(200, len(content))
|
|
181
|
+
num_samples = min(5, max(1, len(content) // sample_size))
|
|
182
|
+
step = max(1, (len(content) - sample_size) // max(1, num_samples - 1)) if num_samples > 1 else 0
|
|
183
|
+
repetitive_samples = 0
|
|
184
|
+
for i in range(num_samples):
|
|
185
|
+
start = i * step
|
|
186
|
+
window = content[start : start + sample_size]
|
|
187
|
+
if len(set(window)) < 3:
|
|
188
|
+
repetitive_samples += 1
|
|
189
|
+
# Fail if majority of samples are repetitive
|
|
190
|
+
if repetitive_samples > num_samples // 2:
|
|
191
|
+
return False, "Text appears to be excessively repetitive"
|
|
192
|
+
|
|
193
|
+
# Check for disallowed characters (CJK, emoji, etc.)
|
|
194
|
+
if self.check_disallowed_characters:
|
|
195
|
+
pattern = re.compile(
|
|
196
|
+
r"["
|
|
197
|
+
r"\u4e00-\u9FFF" # CJK Unified Ideographs
|
|
198
|
+
r"\u3040-\u309F" # Hiragana
|
|
199
|
+
r"\u30A0-\u30FF" # Katakana
|
|
200
|
+
r"\U0001F600-\U0001F64F" # Emoticons
|
|
201
|
+
r"\U0001F300-\U0001F5FF" # Miscellaneous Symbols
|
|
202
|
+
r"\U0001F680-\U0001F6FF" # Transport and Map Symbols
|
|
203
|
+
r"\U0001F1E0-\U0001F1FF" # Regional Indicator Symbols
|
|
204
|
+
r"]",
|
|
205
|
+
flags=re.UNICODE,
|
|
206
|
+
)
|
|
207
|
+
matches = pattern.findall(content)
|
|
208
|
+
if matches:
|
|
209
|
+
return False, f"Text contains disallowed characters: {matches[:5]}"
|
|
210
|
+
|
|
211
|
+
return True, ""
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
class TextOrderRule(ParseTestRule):
|
|
215
|
+
"""Test rule to verify that one text appears before another."""
|
|
216
|
+
|
|
217
|
+
# Residual HTML markup left behind by ``normalize_text`` (structural tags such
|
|
218
|
+
# as <table>/<tr>/<td>, which it does not touch). Order anchors are authored
|
|
219
|
+
# against plain markdown and routinely span a label and its value, so any page
|
|
220
|
+
# the parser renders as a table puts a `</td><td>` inside the anchor and the
|
|
221
|
+
# rule could never match. Substituting a SPACE — never the empty string —
|
|
222
|
+
# keeps cell contents from welding into one unmatchable token, and matches how
|
|
223
|
+
# the word/sentence bag rules already treat markup.
|
|
224
|
+
_HTML_TAG_PATTERN = re.compile(r"</?[^>]+>")
|
|
225
|
+
|
|
226
|
+
@classmethod
|
|
227
|
+
def _strip_html(cls, text: str) -> str:
|
|
228
|
+
"""Replace residual HTML tags with a space and collapse the runs."""
|
|
229
|
+
return re.sub(r" +", " ", cls._HTML_TAG_PATTERN.sub(" ", text)).strip()
|
|
230
|
+
|
|
231
|
+
def __init__(self, rule_data: ParseOrderRule | dict):
|
|
232
|
+
super().__init__(rule_data)
|
|
233
|
+
rule_data = cast(ParseOrderRule, self._rule_data)
|
|
234
|
+
|
|
235
|
+
if self.type != TestType.ORDER.value:
|
|
236
|
+
raise ValueError(f"Invalid type for TextOrderRule: {self.type}")
|
|
237
|
+
|
|
238
|
+
# Check if we should use light normalization that preserves formatting tags
|
|
239
|
+
self.keep_formatting = rule_data.keep_formatting_text_normalisation
|
|
240
|
+
normalize_fn = normalize_text_light if self.keep_formatting else normalize_text
|
|
241
|
+
|
|
242
|
+
# Canonicalize LaTeX in query strings so "$...$" and "LATEX" authored
|
|
243
|
+
# expectations are matched consistently against content.
|
|
244
|
+
before_for_match = _strip_and_replace_latex(rule_data.before)
|
|
245
|
+
after_for_match = _strip_and_replace_latex(rule_data.after)
|
|
246
|
+
|
|
247
|
+
# Import SentenceBagRule lazily to access _MULTI_DOT_PATTERN
|
|
248
|
+
from parse_bench.evaluation.metrics.parse.rules_bag import SentenceBagRule
|
|
249
|
+
|
|
250
|
+
self.before = re.sub(r" +", " ", SentenceBagRule._MULTI_DOT_PATTERN.sub(" ", normalize_fn(before_for_match)))
|
|
251
|
+
self.after = re.sub(r" +", " ", SentenceBagRule._MULTI_DOT_PATTERN.sub(" ", normalize_fn(after_for_match)))
|
|
252
|
+
# Strip markup on the rule side too, so anchor and content are compared in
|
|
253
|
+
# the same alphabet. Skipped for keep_formatting rules, whose whole point is
|
|
254
|
+
# to assert on the styling tags themselves.
|
|
255
|
+
if not self.keep_formatting:
|
|
256
|
+
self.before = self._strip_html(self.before)
|
|
257
|
+
self.after = self._strip_html(self.after)
|
|
258
|
+
if not self.before.strip():
|
|
259
|
+
raise ValueError("Before field cannot be empty")
|
|
260
|
+
if not self.after.strip():
|
|
261
|
+
raise ValueError("After field cannot be empty")
|
|
262
|
+
if self.max_diffs > len(self.before) // 2 or self.max_diffs > len(self.after) // 2:
|
|
263
|
+
raise ValueError("Max diffs is too large for this test, greater than 50% of the search string")
|
|
264
|
+
|
|
265
|
+
def run(self, md_content: str, normalized_content: str | None = None) -> tuple[bool, str]:
|
|
266
|
+
"""Check if 'before' text appears before 'after' text.
|
|
267
|
+
|
|
268
|
+
When multiple instances exist, we check that the FIRST occurrence of 'before'
|
|
269
|
+
appears before the LAST occurrence of 'after'. This handles cases where the same
|
|
270
|
+
text may appear multiple times in the document.
|
|
271
|
+
"""
|
|
272
|
+
from parse_bench.evaluation.metrics.parse.rules_bag import SentenceBagRule
|
|
273
|
+
|
|
274
|
+
# Order matching must canonicalize LaTeX consistently between rule text and content.
|
|
275
|
+
# Re-normalize from raw markdown so inline math ($...$) becomes a stable LATEX token.
|
|
276
|
+
content_for_match = _strip_and_replace_latex(md_content)
|
|
277
|
+
if self.keep_formatting:
|
|
278
|
+
normalized_content = normalize_text_light(content_for_match)
|
|
279
|
+
else:
|
|
280
|
+
normalized_content = normalize_text(content_for_match)
|
|
281
|
+
normalized_content = re.sub(r" +", " ", SentenceBagRule._MULTI_DOT_PATTERN.sub(" ", normalized_content))
|
|
282
|
+
if not self.keep_formatting:
|
|
283
|
+
normalized_content = self._strip_html(normalized_content)
|
|
284
|
+
|
|
285
|
+
# OPTIMIZATION: Try exact match first (O(n) vs O(n*m) for fuzzy)
|
|
286
|
+
# This provides ~10-100x speedup for rules that match exactly
|
|
287
|
+
# Use find() for FIRST occurrence of before, rfind() for LAST occurrence of after
|
|
288
|
+
before_pos = normalized_content.find(self.before)
|
|
289
|
+
after_pos = normalized_content.rfind(self.after)
|
|
290
|
+
|
|
291
|
+
if before_pos != -1 and after_pos != -1 and before_pos < after_pos:
|
|
292
|
+
# Fast path: both found exactly and in correct order
|
|
293
|
+
return True, ""
|
|
294
|
+
|
|
295
|
+
# Slow path: fall back to fuzzy matching
|
|
296
|
+
# instead of max_diffs = 2, use a relative distance
|
|
297
|
+
# here as the annotation fail quite a lot on typo / unicode chars
|
|
298
|
+
# Cap at 30 to avoid combinatorial explosion in fuzzysearch for long strings
|
|
299
|
+
# (e.g. 7k pattern on 30k text with max_l_dist=350 is ~73B operations)
|
|
300
|
+
before_max_dist = min(max(len(self.before) // 20, self.max_diffs), 15)
|
|
301
|
+
after_max_dist = min(max(len(self.after) // 20, self.max_diffs), 15)
|
|
302
|
+
|
|
303
|
+
# Only do expensive fuzzy search if exact match failed
|
|
304
|
+
if before_pos == -1:
|
|
305
|
+
before_matches = find_near_matches(self.before, normalized_content, max_l_dist=before_max_dist)
|
|
306
|
+
else:
|
|
307
|
+
# Create a fake match object for the exact match (first occurrence)
|
|
308
|
+
before_matches = [type("Match", (), {"start": before_pos, "end": before_pos + len(self.before)})()]
|
|
309
|
+
|
|
310
|
+
if after_pos == -1:
|
|
311
|
+
after_matches = find_near_matches(self.after, normalized_content, max_l_dist=after_max_dist)
|
|
312
|
+
else:
|
|
313
|
+
# Create a fake match object for the exact match (last occurrence)
|
|
314
|
+
after_matches = [type("Match", (), {"start": after_pos, "end": after_pos + len(self.after)})()]
|
|
315
|
+
|
|
316
|
+
if not before_matches:
|
|
317
|
+
return (
|
|
318
|
+
False,
|
|
319
|
+
f"'before' text '{self.before[:40]}...' not found with max_l_dist {before_max_dist}",
|
|
320
|
+
)
|
|
321
|
+
if not after_matches:
|
|
322
|
+
return (
|
|
323
|
+
False,
|
|
324
|
+
f"'after' text '{self.after[:40]}...' not found with max_l_dist {after_max_dist}",
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
# Get FIRST occurrence of before (earliest start position)
|
|
328
|
+
first_before_match = min(before_matches, key=lambda m: m.start)
|
|
329
|
+
# Get LAST occurrence of after (latest start position)
|
|
330
|
+
last_after_match = max(after_matches, key=lambda m: m.start)
|
|
331
|
+
|
|
332
|
+
if first_before_match.start < last_after_match.start:
|
|
333
|
+
return True, ""
|
|
334
|
+
|
|
335
|
+
return (
|
|
336
|
+
False,
|
|
337
|
+
f"Could not find a location where '{self.before}...' appears before '{self.after}...'. "
|
|
338
|
+
f"First 'before' at position {first_before_match.start}, "
|
|
339
|
+
f"last 'after' at position {last_after_match.start}.",
|
|
340
|
+
)
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""Watermark-removal rule with an explicit content-preservation guard."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any, cast
|
|
7
|
+
|
|
8
|
+
from rapidfuzz import fuzz
|
|
9
|
+
|
|
10
|
+
from parse_bench.evaluation.metrics.parse.rules_base import ParseTestRule
|
|
11
|
+
from parse_bench.evaluation.metrics.parse.utils import normalize_text
|
|
12
|
+
from parse_bench.test_cases.parse_rule_schemas import ParseWatermarkRemovalRule
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _partial_similarity(needle: str, content: str) -> float:
|
|
16
|
+
normalized_needle = normalize_text(needle).casefold().strip()
|
|
17
|
+
normalized_content = normalize_text(content).casefold().strip()
|
|
18
|
+
if not normalized_needle or not normalized_content:
|
|
19
|
+
return 0.0
|
|
20
|
+
return float(fuzz.partial_ratio(normalized_needle, normalized_content) / 100.0)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _occurrence_count(needle: str, content: str) -> int:
|
|
24
|
+
"""Count normalized, non-overlapping literal phrase occurrences."""
|
|
25
|
+
normalized_needle = normalize_text(needle).casefold().strip()
|
|
26
|
+
normalized_content = normalize_text(content).casefold().strip()
|
|
27
|
+
if not normalized_needle or not normalized_content:
|
|
28
|
+
return 0
|
|
29
|
+
pattern = re.escape(normalized_needle).replace(r"\ ", r"\s+")
|
|
30
|
+
return sum(1 for _ in re.finditer(pattern, normalized_content))
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class WatermarkRemovalRule(ParseTestRule):
|
|
34
|
+
"""Require watermark text to disappear while sampled body anchors survive."""
|
|
35
|
+
|
|
36
|
+
def __init__(self, rule_data: ParseWatermarkRemovalRule | dict[str, Any]):
|
|
37
|
+
super().__init__(rule_data)
|
|
38
|
+
if not isinstance(self._rule_data, ParseWatermarkRemovalRule):
|
|
39
|
+
raise TypeError("watermark_removal requires ParseWatermarkRemovalRule")
|
|
40
|
+
|
|
41
|
+
def run(
|
|
42
|
+
self,
|
|
43
|
+
md_content: str,
|
|
44
|
+
normalized_content: str | None = None,
|
|
45
|
+
) -> tuple[bool, str, float]:
|
|
46
|
+
rule = cast(ParseWatermarkRemovalRule, self._rule_data)
|
|
47
|
+
actual_markdown = md_content
|
|
48
|
+
if self.parse_output is not None:
|
|
49
|
+
for markdown_page in self.parse_output.pages:
|
|
50
|
+
if markdown_page.page_index + 1 == rule.page:
|
|
51
|
+
actual_markdown = markdown_page.markdown
|
|
52
|
+
break
|
|
53
|
+
|
|
54
|
+
allowed_occurrences = rule.allowed_occurrences or [0] * len(rule.watermark_texts)
|
|
55
|
+
watermark_matches: list[dict[str, Any]] = [
|
|
56
|
+
{
|
|
57
|
+
"text": text,
|
|
58
|
+
"similarity": _partial_similarity(text, actual_markdown),
|
|
59
|
+
"occurrences": _occurrence_count(text, actual_markdown),
|
|
60
|
+
"allowed_occurrences": allowed,
|
|
61
|
+
}
|
|
62
|
+
for text, allowed in zip(rule.watermark_texts, allowed_occurrences, strict=True)
|
|
63
|
+
]
|
|
64
|
+
for match in watermark_matches:
|
|
65
|
+
# Legacy rules have no allowed body occurrences and retain fuzzy
|
|
66
|
+
# leak detection. Occurrence-aware rules distinguish a removed
|
|
67
|
+
# overlay from legitimate copies of the same phrase in body text.
|
|
68
|
+
if rule.allowed_occurrences is None:
|
|
69
|
+
match["removed"] = match["similarity"] < rule.watermark_match_threshold
|
|
70
|
+
else:
|
|
71
|
+
match["removed"] = match["occurrences"] <= match["allowed_occurrences"]
|
|
72
|
+
|
|
73
|
+
preservation_matches: list[dict[str, Any]] = [
|
|
74
|
+
{
|
|
75
|
+
"text": text,
|
|
76
|
+
"similarity": _partial_similarity(text, actual_markdown),
|
|
77
|
+
}
|
|
78
|
+
for text in rule.preserve_texts
|
|
79
|
+
]
|
|
80
|
+
for match in preservation_matches:
|
|
81
|
+
match["preserved"] = match["similarity"] >= rule.preserve_match_threshold
|
|
82
|
+
|
|
83
|
+
removal_score = sum(bool(match["removed"]) for match in watermark_matches) / len(watermark_matches)
|
|
84
|
+
preservation_score = sum(bool(match["preserved"]) for match in preservation_matches) / len(preservation_matches)
|
|
85
|
+
combined_score = min(removal_score, preservation_score)
|
|
86
|
+
passed = removal_score >= rule.removal_pass_threshold and preservation_score >= rule.preservation_pass_threshold
|
|
87
|
+
|
|
88
|
+
self.result_details = {
|
|
89
|
+
"removal_score": removal_score,
|
|
90
|
+
"preservation_score": preservation_score,
|
|
91
|
+
"watermark_match_threshold": rule.watermark_match_threshold,
|
|
92
|
+
"preserve_match_threshold": rule.preserve_match_threshold,
|
|
93
|
+
"watermark_matches": watermark_matches,
|
|
94
|
+
"preservation_matches": preservation_matches,
|
|
95
|
+
}
|
|
96
|
+
explanation = (
|
|
97
|
+
f"watermark_removal={removal_score:.4f} "
|
|
98
|
+
f"body_preservation={preservation_score:.4f} "
|
|
99
|
+
f"removed={sum(bool(match['removed']) for match in watermark_matches)}/{len(watermark_matches)} "
|
|
100
|
+
f"preserved={sum(bool(match['preserved']) for match in preservation_matches)}/{len(preservation_matches)}"
|
|
101
|
+
)
|
|
102
|
+
return passed, explanation, combined_score
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
__all__ = ["WatermarkRemovalRule"]
|